From d7c9bac73632c0af0cc2d473a1fb14e31a083d6b Mon Sep 17 00:00:00 2001 From: Yichao Liang Date: Sat, 26 Sep 2026 15:40:41 -0400 Subject: [PATCH] Remove the submit tools and the solve and learn prompts The phased solve sessions delivered plans through submit_plan and submit_policy, whose capture gate re-ran each plan (repeat rollouts, physics and rule-parameter margins, a necessity check, a robot clearance probe) before accepting it. The continual arms never had these tools. This removes them together with: - the solve and learn prompt templates, their goldens and tests - the attempt deadline and the capture fields on the tool context - the standalone arm's particle override scope - the journal's strategy document and test-phase rollback helpers - the skills' contact_objects hook, which only the clearance probe read - 14 settings flags that lost their last reader The empiric configs stop setting the three removed flags they named; none of them changed behavior. Tool descriptions and probe docstrings the agent can read still mention submit_plan and are left for a separate change, since editing them changes the agent's prompts. --- docs/envs/domino/continuous-perception.md | 5 +- predicators/agent_sdk/belief_probe.py | 31 +- predicators/agent_sdk/bilevel_sketch.py | 8 - predicators/agent_sdk/config.py | 49 +- predicators/agent_sdk/journal.py | 167 +- predicators/agent_sdk/learn_prompts.py | 305 ---- predicators/agent_sdk/local_sandbox.py | 30 - predicators/agent_sdk/parallel_rollouts.py | 5 +- predicators/agent_sdk/plan_execution.py | 2 +- predicators/agent_sdk/play_prompts.py | 42 +- predicators/agent_sdk/policy_execution.py | 12 +- .../agent_sdk/prompts/learn_message.md | 168 -- .../prompts/learn_partial_observability.md | 46 - .../prompts/learn_predicate_invention.md | 140 -- .../prompts/learn_program_message.md | 76 - .../agent_sdk/prompts/learn_program_system.md | 176 -- predicators/agent_sdk/prompts/learn_system.md | 95 - predicators/agent_sdk/prompts/solve_query.md | 163 -- predicators/agent_sdk/prompts/solve_system.md | 372 ---- .../agent_sdk/prompts/subclass_model.md | 20 - predicators/agent_sdk/sketch_prompts.py | 426 ----- predicators/agent_sdk/tools/__init__.py | 15 +- predicators/agent_sdk/tools/assembly.py | 8 +- predicators/agent_sdk/tools/budget.py | 19 +- predicators/agent_sdk/tools/capture.py | 174 -- predicators/agent_sdk/tools/clearance.py | 193 -- predicators/agent_sdk/tools/context.py | 250 +-- predicators/agent_sdk/tools/python_exec.py | 12 - predicators/agent_sdk/tools/registry.py | 19 +- predicators/agent_sdk/tools/synthesis.py | 6 +- predicators/agent_sdk/tools/tasks.py | 59 - predicators/agent_sdk/tools/testing.py | 1597 ----------------- predicators/agent_sdk/tools/verdicts.py | 18 +- .../agent_continual_ablation_approach.py | 2 - .../agent_continual_real_to_sim_approach.py | 2 - .../agent_program_world_model_approach.py | 69 +- .../approaches/agent_sim_learning_approach.py | 103 +- .../agent_sim_predicate_invention_approach.py | 35 +- .../approaches/continual_play_mixin.py | 5 +- .../code_sim_learning/identifiability.py | 10 +- .../code_sim_learning/program_world_model.py | 12 +- .../skill_factories/base.py | 40 +- .../skill_factories/push.py | 44 +- predicators/settings.py | 207 +-- predicators/structs.py | 2 +- predicators/utils.py | 3 +- scripts/configs/empiric/approaches.yaml | 2 - scripts/configs/empiric/common.yaml | 3 - .../agent_sdk/prompt_goldens/explore_query.md | 60 - .../agent_sdk/prompt_goldens/learn_message.md | 72 - .../prompt_goldens/learn_program_message.md | 34 - .../prompt_goldens/learn_program_system.md | 135 -- .../agent_sdk/prompt_goldens/learn_system.md | 81 - .../learn_system_po_invention.md | 177 -- tests/agent_sdk/prompt_goldens/solve_query.md | 70 - .../prompt_goldens/solve_system_explore.md | 59 - .../prompt_goldens/solve_system_plan.md | 57 - .../prompt_goldens/solve_system_policy.md | 70 - .../test_belief_probe_physics_sweep.py | 6 +- .../agent_sdk/test_bilevel_sketch_regions.py | 38 +- tests/agent_sdk/test_capture_decision.py | 263 --- tests/agent_sdk/test_clearance_probe.py | 202 --- tests/agent_sdk/test_prompt_goldens.py | 278 +-- tests/agent_sdk/test_refine_evaluator_gate.py | 1 - tests/agent_sdk/test_session_fatal.py | 4 +- tests/agent_sdk/test_solve_prompt_strategy.py | 190 -- tests/agent_sdk/test_solve_restart_journal.py | 200 +-- tests/agent_sdk/test_submit_plan_capture.py | 1122 ------------ tests/agent_sdk/test_submit_policy_capture.py | 219 --- tests/agent_sdk/test_tool_registry.py | 12 +- tests/agent_sdk/test_trajectory_summary.py | 110 -- ...st_agent_continual_real_to_sim_approach.py | 2 - ...test_agent_program_world_model_approach.py | 60 +- .../test_agent_sim_learning_ablations.py | 29 +- .../test_agent_sim_learning_approach.py | 12 +- ....py => test_evaluate_trajectory_helper.py} | 25 +- .../test_program_world_model.py | 10 +- tests/test_agent_harness_fixes.py | 14 +- tests/test_agent_sdk_tools.py | 213 +-- tests/test_docker_option_plan.py | 2 +- 80 files changed, 354 insertions(+), 8720 deletions(-) delete mode 100644 predicators/agent_sdk/learn_prompts.py delete mode 100644 predicators/agent_sdk/prompts/learn_message.md delete mode 100644 predicators/agent_sdk/prompts/learn_partial_observability.md delete mode 100644 predicators/agent_sdk/prompts/learn_predicate_invention.md delete mode 100644 predicators/agent_sdk/prompts/learn_program_message.md delete mode 100644 predicators/agent_sdk/prompts/learn_program_system.md delete mode 100644 predicators/agent_sdk/prompts/learn_system.md delete mode 100644 predicators/agent_sdk/prompts/solve_query.md delete mode 100644 predicators/agent_sdk/prompts/solve_system.md delete mode 100644 predicators/agent_sdk/sketch_prompts.py delete mode 100644 predicators/agent_sdk/tools/capture.py delete mode 100644 predicators/agent_sdk/tools/clearance.py delete mode 100644 predicators/agent_sdk/tools/tasks.py delete mode 100644 predicators/agent_sdk/tools/testing.py delete mode 100644 tests/agent_sdk/prompt_goldens/explore_query.md delete mode 100644 tests/agent_sdk/prompt_goldens/learn_message.md delete mode 100644 tests/agent_sdk/prompt_goldens/learn_program_message.md delete mode 100644 tests/agent_sdk/prompt_goldens/learn_program_system.md delete mode 100644 tests/agent_sdk/prompt_goldens/learn_system.md delete mode 100644 tests/agent_sdk/prompt_goldens/learn_system_po_invention.md delete mode 100644 tests/agent_sdk/prompt_goldens/solve_query.md delete mode 100644 tests/agent_sdk/prompt_goldens/solve_system_explore.md delete mode 100644 tests/agent_sdk/prompt_goldens/solve_system_plan.md delete mode 100644 tests/agent_sdk/prompt_goldens/solve_system_policy.md delete mode 100644 tests/agent_sdk/test_capture_decision.py delete mode 100644 tests/agent_sdk/test_clearance_probe.py delete mode 100644 tests/agent_sdk/test_solve_prompt_strategy.py delete mode 100644 tests/agent_sdk/test_submit_plan_capture.py delete mode 100644 tests/agent_sdk/test_submit_policy_capture.py delete mode 100644 tests/agent_sdk/test_trajectory_summary.py rename tests/approaches/{test_agent_sim_prompt_formatting.py => test_evaluate_trajectory_helper.py} (86%) diff --git a/docs/envs/domino/continuous-perception.md b/docs/envs/domino/continuous-perception.md index 5cd22e104a..815eb223c1 100644 --- a/docs/envs/domino/continuous-perception.md +++ b/docs/envs/domino/continuous-perception.md @@ -89,9 +89,8 @@ contiguous recorded motion. Re-time an episode after the merge. `after_step` returns `obs` **unchanged**. Shipping is a pure write-only side effect, so *when* it happens is unobservable to the rollout — deferring every chunk to the end produces a bit-identical twin trajectory. Everything reading -state mid-episode (`subgoal_annotations` monitor, -`agent_bilevel_max_execution_replans`, `terminate_on_goal_reached`) reads the -twin's own deterministic simulation either way. +state mid-episode (`terminate_on_goal_reached`) reads the twin's own +deterministic simulation either way. **No protocol change is needed.** `execute_chunks` already packs a list of chunks into one `StepRequest` (`real_robot_bridge.py:176-185`), and `_split_actions` is diff --git a/predicators/agent_sdk/belief_probe.py b/predicators/agent_sdk/belief_probe.py index 2f50b52108..f98aba00de 100644 --- a/predicators/agent_sdk/belief_probe.py +++ b/predicators/agent_sdk/belief_probe.py @@ -38,15 +38,15 @@ Sequence, Set, Tuple, Union from predicators import utils -from predicators.agent_sdk.config import RefinementConfig, ToolSurfaceConfig, \ - ValidationConfig +from predicators.agent_sdk.config import RefinementConfig, ToolSurfaceConfig from predicators.agent_sdk.parallel_rollouts import prefetch_parallel from predicators.agent_sdk.tools.context import absolute_rollout_seed, \ decorrelated_rollout_seed from predicators.agent_sdk.tools.scene import apply_state_modifications, \ draw_pybullet_annotation, render_pybullet_image, render_scene_image from predicators.agent_sdk.tools.verdicts import _EvalStateCollector, \ - evaluate_states_with, load_ground_sampler_fns, make_solved_check + _policy_source_path, evaluate_states_with, load_ground_sampler_fns, \ + make_solved_check from predicators.structs import State, Task, excluded_object_type_names if TYPE_CHECKING: @@ -62,29 +62,17 @@ class ProbeBudgetExceeded(Exception): """A probe call ran past a wall-clock budget. Raised cooperatively at probe checkpoints (every sim call) when the - run_python per-call limit or the solve attempt's wall clock has - expired. ``run_python`` catches it specially: the code's printed + run_python per-call limit has expired. ``run_python`` catches it + specially: the code's printed output so far is returned with the budget message appended, so a stopped sweep still hands the agent its partial results. """ def _check_time_budget(ctx: "ToolContext") -> None: - """Raise :class:`ProbeBudgetExceeded` when a wall-clock budget is up. - - Never fires during the final-submission nudge - (``ctx.capture_best_effort_plan``): with the budget spent, the one - thing left is submitting, and blocking that would forfeit the task. - """ - if ctx.capture_best_effort_plan: - return + """Raise :class:`ProbeBudgetExceeded` when the run_python call's time limit + is up.""" now = time.monotonic() - attempt_dl = ctx.attempt_deadline - if attempt_dl is not None and now > attempt_dl: - raise ProbeBudgetExceeded( - "the attempt's wall-clock exploration budget is exhausted. Stop " - "exploring NOW and submit your single best plan via " - "submit_plan on the current task (omit task_idx).") call_dl = ctx.python_call_deadline if call_dl is not None and now > call_dl: call_timeout = ToolSurfaceConfig.from_cfg().python_call_timeout @@ -930,7 +918,9 @@ def _option_model(self) -> Any: def _fresh_scope(self) -> Optional[Callable[..., Any]]: """Select isolation for the model this probe actually executes.""" - if not ValidationConfig.from_cfg().fresh_env: + # pylint: disable-next=import-outside-toplevel + from predicators.settings import CFG + if not CFG.agent_plan_validation_fresh_env: return None ctx = self._ctx if ctx.probe_option_model_provider is not None: @@ -2561,7 +2551,6 @@ def run_policy( from predicators.agent_sdk.policy_execution import \ build_policy_option_fn, execute_policy_forward - from predicators.agent_sdk.tools.testing import _policy_source_path from predicators.settings import CFG # pylint: enable=import-outside-toplevel diff --git a/predicators/agent_sdk/bilevel_sketch.py b/predicators/agent_sdk/bilevel_sketch.py index ff66682b1c..c9ddee5535 100644 --- a/predicators/agent_sdk/bilevel_sketch.py +++ b/predicators/agent_sdk/bilevel_sketch.py @@ -8,9 +8,6 @@ - ``sketch_types``: shared dataclasses (``GroundSampler``, ``SketchStep``) that parsing constructs and refinement/execution consume. -- ``sketch_prompts``: ``build_solve_system_prompt`` and - ``build_solve_prompt``, the solve/explore system-prompt and query - builders (rendered from ``prompts/*.md``). - ``sketch_parsing``: the sketch-line grammar - step/plan formatters and the parsers for subgoal / ``~`` ground-sampler annotations and continuous params. @@ -27,8 +24,6 @@ parse_region_annotations, parse_sketch_from_text, \ parse_subgoal_annotations, strip_code_fences, strip_region_annotations, \ strip_subgoal_annotations -from predicators.agent_sdk.sketch_prompts import build_early_stop_note, \ - build_solve_prompt, build_solve_system_prompt from predicators.agent_sdk.sketch_refinement import DeepestFailure, \ InfoScorer, RefineOutcome, StepProbeSuggestion, ground_step, \ refine_and_validate_report, refine_sketch, resolve_refine_timeout, \ @@ -44,9 +39,6 @@ "SketchStep", "StepOutcome", "StepProbeSuggestion", - "build_early_stop_note", - "build_solve_prompt", - "build_solve_system_prompt", "execute_plan_forward", "format_plan_lines", "format_sketch_lines", diff --git a/predicators/agent_sdk/config.py b/predicators/agent_sdk/config.py index 65eb8a0da3..4b20e3c85c 100644 --- a/predicators/agent_sdk/config.py +++ b/predicators/agent_sdk/config.py @@ -49,18 +49,9 @@ def from_cfg(cls) -> "SessionConfig": @dataclass(frozen=True) class RefinementConfig: - """Plan-sketch refinement: search budgets, gates, and ground samplers. - - Consumed at handler entry by ``submit_plan`` (tools/testing.py) and - by the probe's ``refine`` (belief_probe.py). - """ + """Plan-sketch refinement settings the probe's ``refine`` reads.""" ground_samplers: bool - refinement_timeout_per_step: float - refinement_timeout_min: float max_samples_per_step: int - check_subgoals: bool - log_state: bool - use_llm_initial_params: bool @classmethod def from_cfg(cls) -> "RefinementConfig": @@ -68,45 +59,7 @@ def from_cfg(cls) -> "RefinementConfig": # Flags keep their names for experiment-yaml compatibility. return cls( ground_samplers=CFG.agent_bilevel_ground_samplers, - refinement_timeout_per_step=( - CFG.agent_bilevel_refinement_timeout_per_step), - refinement_timeout_min=CFG.agent_bilevel_refinement_timeout_min, max_samples_per_step=CFG.agent_bilevel_max_samples_per_step, - check_subgoals=CFG.agent_bilevel_check_subgoals, - log_state=CFG.agent_bilevel_log_state, - use_llm_initial_params=CFG.agent_bilevel_use_llm_initial_params, - ) - - -@dataclass(frozen=True) -class ValidationConfig: - """Capture-validation rollouts and the cross-attempt journal. - - Consumed at handler entry by ``submit_plan`` (tools.py) and the - probe's ``run(trials=N)`` (belief_probe.py); ``use_journal`` gates - the journal / attempt-log channel. - """ - rollouts: int - rollouts_after_flaky: int - fresh_env: bool - physics_margin: bool - rule_param_margin: bool - necessity: bool - use_journal: bool - - @classmethod - def from_cfg(cls) -> "ValidationConfig": - """Read the validation flags from the live ``CFG``.""" - # Flags keep their names for experiment-yaml compatibility. - return cls( - rollouts=CFG.agent_plan_validation_rollouts, - rollouts_after_flaky=( - CFG.agent_plan_validation_rollouts_after_flaky), - fresh_env=CFG.agent_plan_validation_fresh_env, - physics_margin=CFG.agent_plan_validation_physics_margin, - rule_param_margin=CFG.agent_plan_validation_rule_param_margin, - necessity=CFG.agent_plan_validation_necessity, - use_journal=CFG.agent_solve_use_journal, ) diff --git a/predicators/agent_sdk/journal.py b/predicators/agent_sdk/journal.py index 6d45998f3b..16b101a9ad 100644 --- a/predicators/agent_sdk/journal.py +++ b/predicators/agent_sdk/journal.py @@ -1,35 +1,24 @@ -"""Persistent per-run solve journal and attempt log. - -Two markdown files in the sandbox that carry knowledge across solve -attempts, test tasks, and learning cycles: - -- ``journal.md`` is the AGENT's notebook. Solve and learn sessions - append to it with the ordinary file tools (no dedicated tool): short - factual entries - what was tried with exact parameters, what was - measured, what to try differently. The prompts ask for facts and - measurements rather than verdicts: a recorded "X is impossible" - from a failed attempt would re-import exactly the anchoring a fresh - context is meant to shed, while "tried yaws 0-15 deg at x in - [0.50, 0.54], all stopped >=5 cm short" steers the next attempt - without foreclosing it. -- ``attempts.md`` is the HARNESS's log, never edited by the agent: - each task's goal + initial state (once per task) and each - attempt's outcome and captured or best refused plan, so the - essentials of every attempt are on record even when the agent - writes nothing. - -Fresh-context solve sessions read both from their prompt (tail-capped -so recent attempts stay intact), so knowledge travels through these -curated channels instead of raw transcript history. - -Phase lifecycle: learning-phase content persists for the whole run -and accumulates across online-learning cycles, so every evaluation -starts from all learning knowledge so far. Test-phase additions live -only for their own evaluation: at ``end_test_phase`` the approach -archives both files to the run's log dir (outside the sandbox, so the -agent cannot read them) and rolls them back to their pre-test content -via :func:`read_raw` / :func:`restore` - entries written while -solving one evaluation's test tasks must not leak into the next. +"""Persistent per-run journal and round log. + +Two markdown files in the sandbox that carry knowledge across the rounds +and levels of a continual run: + +- ``journal.md`` is the AGENT's notebook. The agent appends to it with + the ordinary file tools (no dedicated tool): short factual entries - + what was tried with exact parameters, what was measured, what to try + differently. The prompts ask for facts and measurements rather than + verdicts: a recorded "X is impossible" from a failed attempt would + re-import exactly the anchoring a fresh context is meant to shed, + while "tried yaws 0-15 deg at x in [0.50, 0.54], all stopped >=5 cm + short" steers the next attempt without foreclosing it. +- ``attempts.md`` is the HARNESS's log, never edited by the agent: one + entry per round (what the agent did in the environment) and per model + change, so the essentials of every round are on record even when the + agent writes nothing. + +Each round's query injects both (tail-capped so recent entries stay +intact), so knowledge travels through these curated channels instead of +raw transcript history. """ from __future__ import annotations @@ -38,133 +27,33 @@ from typing import Optional JOURNAL_FILENAME = "journal.md" -# The harness-owned attempt log (task contexts, attempt outcomes). +# The harness-owned round log. ATTEMPTS_FILENAME = "attempts.md" -# Per-entry cap for harness attempt-log entries: the first entry per -# task embeds the init-state feature dict (the prompt's own -# representation) and a captured plan. The writer orders the layout -# block last, so tail truncation at this cap can only ever cut layout, -# never the outcome or the captured plan. +# Per-entry cap for harness log entries. MAX_ENTRY_CHARS = 4000 -MAX_AUTO_ENTRY_CHARS = MAX_ENTRY_CHARS -# Cap on how much of each file is injected into a solve prompt. -# Tail-biased: recent attempts (usually the same task) matter most. +# Cap on how much of each file is injected into a query. Tail-biased: +# recent rounds matter most. MAX_PROMPT_CHARS = 6000 -# The learn-phase-maintained domain strategy document. Unlike the -# append-only journal (facts and measurements), strategy.md is a LIVING -# document the learn agent rewrites freely each cycle: its best current -# natural-language account of how to solve tasks in this domain. Solve -# prompts inject it as explicitly-advisory reference. -STRATEGY_FILENAME = "strategy.md" - -# Cap on how much strategy is injected into a solve prompt. Head-biased -# (unlike the journal): the document is curated, so its lead carries the -# headline strategy and a tail truncation only cuts detail. -MAX_STRATEGY_PROMPT_CHARS = 4000 - - -def journal_path(sandbox_dir: str) -> str: - """Host path of the run's journal file.""" - return os.path.join(sandbox_dir, JOURNAL_FILENAME) - - -def attempts_path(sandbox_dir: str) -> str: - """Host path of the run's harness-owned attempt log.""" - return os.path.join(sandbox_dir, ATTEMPTS_FILENAME) - - -def strategy_path(sandbox_dir: str) -> str: - """Host path of the run's domain strategy document.""" - return os.path.join(sandbox_dir, STRATEGY_FILENAME) - - -def read_strategy(sandbox_dir: Optional[str], - max_chars: int = MAX_STRATEGY_PROMPT_CHARS) -> str: - """Strategy document content for prompt injection ("" when absent). - - Head-biased truncation: the document is curated by the learn agent, - so the front holds the headline strategy; a truncation notice marks - the cut so readers know detail was dropped. - """ - if not sandbox_dir: - return "" - path = strategy_path(sandbox_dir) - if not os.path.isfile(path): - return "" - with open(path, "r", encoding="utf-8") as f: - content = f.read().strip() - if len(content) > max_chars: - # Cut at a line boundary, never mid-word. - head = content[:max_chars] - cut = head.rfind("\n") - if cut > 0: - head = head[:cut] - content = (head.rstrip() + - "\n[strategy truncated at the prompt cap - read " - f"./{STRATEGY_FILENAME} for the rest]") - return content - def append_entry(sandbox_dir: str, header: str, body: str, max_chars: int = MAX_ENTRY_CHARS, - filename: str = ATTEMPTS_FILENAME) -> Optional[str]: - """Append one harness entry; returns a truncation notice or None. + filename: str = ATTEMPTS_FILENAME) -> None: + """Append one harness entry. ``header`` becomes a ``###
`` line; ``body`` is written - verbatim below it, truncated at ``max_chars`` (default - :data:`MAX_ENTRY_CHARS`; harness auto-entries pass - :data:`MAX_AUTO_ENTRY_CHARS`). + verbatim below it, truncated at ``max_chars``. """ os.makedirs(sandbox_dir, exist_ok=True) - note: Optional[str] = None body = body.strip() if len(body) > max_chars: body = body[:max_chars].rstrip() body += "\n[entry truncated at the per-entry size cap]" - note = (f"entry truncated to {max_chars} chars - keep journal " - "entries short and factual") with open(os.path.join(sandbox_dir, filename), "a", encoding="utf-8") as f: f.write(f"### {header.strip()}\n{body}\n\n") - return note - - -def read_raw(sandbox_dir: Optional[str], - filename: str = JOURNAL_FILENAME) -> Optional[str]: - """Exact file content, or None if the file does not exist. - - Unlike :func:`read_journal` there is no prompt trimming and the - absent-file case is distinguishable from an empty file, so the - result is a faithful snapshot for :func:`restore`. - """ - if not sandbox_dir: - return None - path = os.path.join(sandbox_dir, filename) - if not os.path.isfile(path): - return None - with open(path, "r", encoding="utf-8") as f: - return f.read() - - -def restore(sandbox_dir: str, - snapshot: Optional[str], - filename: str = JOURNAL_FILENAME) -> None: - """Reset the file to a :func:`read_raw` snapshot. - - A ``None`` snapshot means the file did not exist, so it is removed - if present. - """ - path = os.path.join(sandbox_dir, filename) - if snapshot is None: - if os.path.isfile(path): - os.remove(path) - return - os.makedirs(sandbox_dir, exist_ok=True) - with open(path, "w", encoding="utf-8") as f: - f.write(snapshot) def read_journal(sandbox_dir: Optional[str], diff --git a/predicators/agent_sdk/learn_prompts.py b/predicators/agent_sdk/learn_prompts.py deleted file mode 100644 index 75f38a195b..0000000000 --- a/predicators/agent_sdk/learn_prompts.py +++ /dev/null @@ -1,305 +0,0 @@ -"""Prompt construction for the learning (simulator synthesis) phase. - -Rendered from ``learn_system.md``, ``learn_message.md``, -``learn_predicate_invention.md``, and ``learn_partial_observability.md`` -in ``predicators/agent_sdk/prompts`` (see :mod:`prompt_templates`). The -approach classes gather the per-instance values (digests, data roster, -reports, paths) and call these pure builders, so every prompt can be -rendered and reviewed without a live session. -""" -import re -from typing import Any, Mapping, Sequence - -from predicators.agent_sdk.prompt_templates import render - -_BLANK_RUN_RE = re.compile(r"\n{3,}") - - -def _join(parts: Sequence[str]) -> str: - text = "\n\n".join(p.strip("\n") for p in parts if p and p.strip()) - return _BLANK_RUN_RE.sub("\n\n", text).strip("\n") + "\n" - - -# --------------------------------------------------------------------------- -# System prompt -# --------------------------------------------------------------------------- - - -def build_learn_system_prompt( - *, - partially_observable: bool, - residual_rule_signature: str, - scene_viz_hint: str, - physical_params_section: str = "", - extra_sections: Sequence[str] = (), - latent_extra_sections: Sequence[str] = (), - workflow_extra: str = "", - declared_params_only: bool = False, -) -> str: - """Compose the synthesis system prompt. - - The shared subclass contract supplies dynamics, fitting and optional - model-state guidance. Extra sections add predicate invention and - workflow requirements for the particular learning arm. - """ - # Retain the keyword arguments for callers outside this package. - del residual_rule_signature, scene_viz_hint - parts = [ - render("learn_system", "intro"), - render("subclass_model", "simulator"), - render("subclass_model", "dynamics"), - physical_params_section, - render("learn_system", "declared_params") - if declared_params_only else "", - render("subclass_model", "tools"), - *extra_sections, - ] - if partially_observable: - parts.append( - render("subclass_model", - "memory", - base_class="BaseSimulator", - estimate_errors="errors in the model and noisy input")) - parts.extend(latent_extra_sections) - parts += [ - render("learn_system", "plan_format"), - render("learn_system", "deliverables"), - render("learn_system", - "workflow", - workflow_extra=(" " + - workflow_extra) if workflow_extra else ""), - ] - return _join(parts) - - -def render_physical_params_section( - info: Mapping[str, Mapping[str, Any]]) -> str: - """The base-physics parameter menu for a revealed parameter menu. - - ``info`` maps a parameter name to its ``default``, ``lo``, ``hi``, - ``description``, and optional ``scale``; empty input renders - nothing, so envs without a menu never see the feature mentioned. - """ - if not info: - return "" - lines = [] - for name, meta in info.items(): - scale_note = (", fitted in log-space" - if meta.get("scale") == "log" else "") - lines.append(f"- `{name}` (built-in {meta['default']:.4g}, fit " - f"box [{meta['lo']:.4g}, {meta['hi']:.4g}]" - f"{scale_note}): {meta['description']}") - return render("subclass_model", - "physical_params", - param_list="\n".join(lines)) - - -def render_predicate_invention_section(scene_workbench: str) -> str: - """The predicate-invention system-prompt section.""" - return render("learn_predicate_invention", - "system", - scene_workbench=scene_workbench) - - -def render_predicate_latent_section() -> str: - """The predicate-side latent guidance (invention arms, PO only).""" - return render("learn_partial_observability", "predicates") - - -def render_predicate_workflow_extra() -> str: - """The invention arm's addition to the workflow's validation step.""" - return render("learn_predicate_invention", "workflow_extra") - - -# --------------------------------------------------------------------------- -# First message -# --------------------------------------------------------------------------- - - -def build_learn_message( - *, - n_trajs: int, - n_transitions: int, - n_demos: int, - n_interaction: int, - trajectory_listing: str, - structs_ref: str, - inferred_hint: str, - predicate_listing: str, - types_digest: str, - options_digest: str, - simulator_file: str, - objective_block: str = "", - prior_state_block: str = "", - divergence_block: str = "", - base_sim_block: str = "", - tools_block: str = "", - extra_messages: Sequence[str] = (), -) -> str: - """Compose the synthesis session's first message. - - Every block argument is already rendered (see the ``render_*`` - helpers below) or empty. ``extra_messages`` (predicate invention, - partial observability, sampler synthesis) are appended in order. - """ - body = render( - "learn_message", - "skeleton", - n_trajs=str(n_trajs), - n_transitions=str(n_transitions), - n_demos=str(n_demos), - n_interaction=str(n_interaction), - trajectory_listing=trajectory_listing.strip("\n"), - objective_block=objective_block, - prior_state_block=prior_state_block, - divergence_block=divergence_block, - structs_ref=structs_ref, - base_sim_block=base_sim_block, - inferred_hint=inferred_hint, - predicate_listing=predicate_listing, - types_digest=types_digest.strip("\n"), - options_digest=options_digest.strip("\n"), - tools_block=tools_block, - simulator_file=simulator_file, - ) - return _join([body, *extra_messages]) - - -def render_divergence_block(report: str, has_prior_model: bool) -> str: - """The start-of-session residual report section.""" - return render("learn_message", - "divergence_prior" if has_prior_model else "divergence_base", - report=report.strip("\n")) - - -def render_base_sim_block(refs: Sequence[str]) -> str: - """The base-simulator source listing, or empty.""" - if not refs: - return "" - return render("learn_message", - "base_sim", - ref_listing="\n".join(f" - {r}" for r in refs)) - - -def render_tools_block(tool_names: Sequence[str]) -> str: - """The session's tool roster, or empty.""" - if not tool_names: - return "" - return render("learn_message", - "tools", - tool_listing="\n".join(f" - {t}" for t in tool_names)) - - -def render_objective_block(description: str) -> str: - """The env's public task objective section, or empty.""" - if not description: - return "" - return render("learn_message", "objective", description=description) - - -def render_prior_state_block(prior_files: Sequence[str]) -> str: - """The prior-cycle-state paragraph for the artifacts found, or empty.""" - if not prior_files: - return "" - return render("learn_message", - "prior_state", - prior_files=" and ".join(prior_files)) - - -def render_predicate_invention_message(predicates_file: str, - goal_block: str) -> str: - """The invention arm's addition to the first message.""" - return render("learn_predicate_invention", - "message", - predicates_file=predicates_file, - goal_block=goal_block.strip("\n")) - - -def render_partial_observability_message() -> str: - """The short partial-observability note for the first message.""" - return render("learn_partial_observability", "message") - - -def render_zero_shot_message() -> str: - """The no-data note for the first message (ablation A2).""" - return render("learn_message", "zero_shot") - - -# --------------------------------------------------------------------------- -# Program world model arm (C4) -# --------------------------------------------------------------------------- - - -def build_program_learn_system_prompt( - *, - scene_viz_hint: str, - extra_sections: Sequence[str] = (), - workflow_extra: str = "", -) -> str: - """Compose the program-world-model synthesis system prompt. - - ``extra_sections`` (predicate invention) follow the validation - guidance; the plan-format section is shared with the residual arm's - template. ``scene_viz_hint`` is accepted for parity with the - residual builder (the program template names the probe surface - itself) and is not rendered. - """ - del scene_viz_hint - parts = [ - render("learn_program_system", "intro"), - render("learn_program_system", "produce"), - render("learn_program_system", "modeling"), - render("learn_program_system", "tools"), - render("learn_program_system", "validation"), - *extra_sections, - render("learn_system", "plan_format"), - render("learn_program_system", "deliverables"), - render("learn_program_system", - "workflow", - workflow_extra=(" " + - workflow_extra) if workflow_extra else ""), - ] - return _join(parts) - - -def build_program_learn_message( - *, - n_trajs: int, - n_transitions: int, - n_demos: int, - n_interaction: int, - trajectory_listing: str, - structs_ref: str, - predicate_listing: str, - types_digest: str, - options_digest: str, - world_model_file: str, - objective_block: str = "", - prior_state_block: str = "", - tools_block: str = "", - extra_messages: Sequence[str] = (), -) -> str: - """Compose the program-world-model synthesis session's first message.""" - body = render( - "learn_program_message", - "skeleton", - n_trajs=str(n_trajs), - n_transitions=str(n_transitions), - n_demos=str(n_demos), - n_interaction=str(n_interaction), - trajectory_listing=trajectory_listing.strip("\n"), - objective_block=objective_block, - prior_state_block=prior_state_block, - structs_ref=structs_ref, - predicate_listing=predicate_listing, - types_digest=types_digest.strip("\n"), - options_digest=options_digest.strip("\n"), - tools_block=tools_block, - world_model_file=world_model_file, - ) - return _join([body, *extra_messages]) - - -def render_program_zero_shot_message() -> str: - """The no-data note for the program arm's first message.""" - return render("learn_program_message", "zero_shot") diff --git a/predicators/agent_sdk/local_sandbox.py b/predicators/agent_sdk/local_sandbox.py index 00e9a4b93d..a82062f688 100644 --- a/predicators/agent_sdk/local_sandbox.py +++ b/predicators/agent_sdk/local_sandbox.py @@ -37,7 +37,6 @@ import datetime import logging import os -import time from typing import Any, Dict, List, Optional from predicators.agent_sdk.config import SessionConfig @@ -56,11 +55,6 @@ # session run inside one tool call. MCP_TOOL_TIMEOUT_MS = 6 * 3600 * 1000 -# Grace period past the solve-attempt deadline before interrupting a -# still-streaming agent turn (cooperative tool refusals normally end -# the turn well before this). -_DEADLINE_INTERRUPT_SLACK_S = 180 - # Build local-sandbox-specific prompts from shared templates. # CLAUDE.md (sandbox mechanics only; see build_claude_md) is written # into the sandbox when it is populated. @@ -179,33 +173,9 @@ async def query(self, if not self._started: await self.start_session() - # Wall-clock backstop for the solve attempt deadline: the probe - # and run_python enforce it cooperatively (tool calls refuse - # past the deadline), so normally the agent wraps up on its own; - # interrupt only if the turn stream is still going long after. - # The approach clears attempt_deadline before its final-submission - # nudge, so the submission query is never interrupted. - interrupt_sent = False - - async def _maybe_interrupt_on_deadline(_entry: Dict[str, Any]) -> None: - nonlocal interrupt_sent - deadline = getattr(self._tool_context, "attempt_deadline", None) - if (interrupt_sent or deadline is None or time.monotonic() <= - deadline + _DEADLINE_INTERRUPT_SLACK_S): - return - interrupt_sent = True - logger.warning( - "Solve-attempt wall clock exceeded by >%ds mid-query; " - "interrupting the agent turn.", _DEADLINE_INTERRUPT_SLACK_S) - try: - await self._client.interrupt() - except Exception as e: # pylint: disable=broad-except - logger.warning("Interrupt failed: %s", e) - async def _on_entry(entry: Dict[str, Any]) -> None: # The context counters behind the play tools' [context] line. self._tool_context.note_stream_entry(entry) - await _maybe_interrupt_on_deadline(entry) collected = await self._run_streamed_query(message, log_path=log_path, diff --git a/predicators/agent_sdk/parallel_rollouts.py b/predicators/agent_sdk/parallel_rollouts.py index 79b8d6c1d7..4304854c1f 100644 --- a/predicators/agent_sdk/parallel_rollouts.py +++ b/predicators/agent_sdk/parallel_rollouts.py @@ -1,8 +1,7 @@ """Fork-based parallel execution of independent validation rollouts. -The capture gate's margin sweep, its repeat rollouts, and the belief -probe's ``trials=N`` / ``physics_sweep`` modes all run the same shape of -work: N independent rollouts, each on a freshly constructed env under +The belief probe's ``trials=N`` and ``physics_sweep`` modes run the same +shape of work: N independent rollouts, each on a freshly constructed env under its own seed / override scope, in one CPU-bound process. This module fans them out as forked children. diff --git a/predicators/agent_sdk/plan_execution.py b/predicators/agent_sdk/plan_execution.py index 48bcdd1462..d3f284946b 100644 --- a/predicators/agent_sdk/plan_execution.py +++ b/predicators/agent_sdk/plan_execution.py @@ -127,7 +127,7 @@ def execute_plan_forward( """Execute a fully-grounded plan step by step through the option model. Shared forward-execution core behind ``validate_plan_forward`` (used - by ``BeliefProbe.refine``) and the ``submit_plan`` tool. + by ``BeliefProbe.refine``) and the probe's rollouts. State carries forward across options — matching how the real env executes. Per step it mirrors ``run_backtracking_refinement``'s fixed-plan path: check ``initiable``, call diff --git a/predicators/agent_sdk/play_prompts.py b/predicators/agent_sdk/play_prompts.py index 97680accec..4efff28a1f 100644 --- a/predicators/agent_sdk/play_prompts.py +++ b/predicators/agent_sdk/play_prompts.py @@ -8,7 +8,7 @@ """ from __future__ import annotations -from typing import Iterable, List, Sequence +from typing import Any, Iterable, List, Mapping, Sequence from predicators.agent_sdk.prompt_templates import render from predicators.observation_noise import ObservationNoise @@ -354,6 +354,28 @@ def build_play_system_prompt(tool_names: Sequence[str], return "\n\n".join(section.strip() for section in sections) +def render_physical_params_section( + info: Mapping[str, Mapping[str, Any]]) -> str: + """The base-physics parameter menu for a revealed parameter menu. + + ``info`` maps a parameter name to its ``default``, ``lo``, ``hi``, + ``description``, and optional ``scale``; empty input renders + nothing, so envs without a menu never see the feature mentioned. + """ + if not info: + return "" + lines = [] + for name, meta in info.items(): + scale_note = (", fitted in log-space" + if meta.get("scale") == "log" else "") + lines.append(f"- `{name}` (built-in {meta['default']:.4g}, fit " + f"box [{meta['lo']:.4g}, {meta['hi']:.4g}]" + f"{scale_note}): {meta['description']}") + return render("subclass_model", + "physical_params", + param_list="\n".join(lines)) + + def build_model_contract( *, partially_observable: bool, @@ -369,15 +391,15 @@ def build_model_contract( ``partially_observable`` adds the model-state callback contract and the latent-aware classifier note. ``physical_params_section`` is the rendered system-identification section, from - ``render_physical_params_section`` in the learn prompt module; empty - when the env reveals no tunable physics. ``declared_params_only`` - adds the no-harness-fitting section, since the probe then refuses to - fit. ``frozen`` (the zero-shot arm) drops the fitting guidance, - since the model is sealed at the first action. ``supplied_model`` - (scene-only, oracle dynamics) keeps only the predicate contract: the - agent never writes ``simulator.py``. ``scene_built`` (the agentic - real-to-sim arm) replaces the domain-twin subclass contract with the - ``SceneBase`` one: the agent loads the scene itself. + ``render_physical_params_section`` above; empty when the env reveals + no tunable physics. ``declared_params_only`` adds the no-harness- + fitting section, since the probe then refuses to fit. ``frozen`` + (the zero-shot arm) drops the fitting guidance, since the model is + sealed at the first action. ``supplied_model`` (scene-only, oracle + dynamics) keeps only the predicate contract: the agent never writes + ``simulator.py``. ``scene_built`` (the agentic real-to-sim arm) + replaces the domain-twin subclass contract with the ``SceneBase`` + one: the agent loads the scene itself. """ if supplied_model: parts = [ diff --git a/predicators/agent_sdk/policy_execution.py b/predicators/agent_sdk/policy_execution.py index 4615ffd30a..8707a6295c 100644 --- a/predicators/agent_sdk/policy_execution.py +++ b/predicators/agent_sdk/policy_execution.py @@ -1,10 +1,10 @@ -"""Closed-loop execution of agent-written per-task policies. +"""Closed-loop execution of agent-written policies in the belief model. -``agent_solve_policy_mode`` replaces the captured fixed option plan with -a program the solve agent writes to the sandbox (``policy.py``): a -``get_option(state, memory)`` function that returns the NEXT plan line -(same grammar as sketches) from the actual current state, or ``None`` -when finished. This module holds the two shared pieces: +``BeliefProbe.run_policy`` rolls out a program the agent writes to the +sandbox (``policy.py``): a ``get_option(state, memory)`` function that +returns the NEXT plan line (same grammar as sketches) from the actual +current state, or ``None`` when finished. This module holds the two +shared pieces: * :func:`build_policy_option_fn` - execs the policy source once and wraps ``get_option`` into a ``(state, last_failure) -> Optional[ diff --git a/predicators/agent_sdk/prompts/learn_message.md b/predicators/agent_sdk/prompts/learn_message.md deleted file mode 100644 index d4d496858c..0000000000 --- a/predicators/agent_sdk/prompts/learn_message.md +++ /dev/null @@ -1,168 +0,0 @@ -# Synthesis (learn) first message - -Composed by `AgentSimLearningApproach._build_synthesis_learn_message` -through `learn_prompts.build_learn_message`. The message carries this -cycle's data and digests; the rules of the phase live in the system -prompt (`learn_system.md`). - - -Synthesize a residual dynamics simulator for this environment. There -are __N_TRAJS__ trajectories (__N_TRANSITIONS__ step transitions) -available: __N_DEMOS__ oracle demonstration(s), which reached the goal -by construction, and __N_INTERACTION__ interaction trajectory/ies -collected during online learning, some of which may have failed to -reach the goal. - -__TRAJECTORY_LISTING__ - -Each trajectory carries a `train_task_idx`. `is_goal_state(state, -task_idx)` (equivalently `train_tasks[task_idx].goal_holds(state)`) -checks a single state for the goal atoms. Reaching the goal atoms does -not by itself mean an episode is solved; when a task objective is -stated below, score full trajectories with `evaluate_trajectory`. Use -`is_goal_state` to confirm which trajectories reached the goal atoms -and to treat failed interaction trajectories as counterexamples: places -where a predicate or rule said "this should work" and the environment -disagreed. - -__OBJECTIVE_BLOCK__ - -__PRIOR_STATE_BLOCK__ - -__DIVERGENCE_BLOCK__ - -Data-structure source code is at: __STRUCTS_REF__ - -__BASE_SIM_BLOCK__ - -A residual scan between the base simulator's prediction and the -observed next state suggests that these features carry residual -dynamics (a starting hint; it may include base-sim jitter, so refine it -as you go): - -__INFERRED_HINT__ - -## Available Predicates (for subgoal annotations) - -__PREDICATE_LISTING__ - -Subgoal annotations in plans for `sim.refine` / `sim.run` must -reference these predicate names with matching arity and types. Any -threshold or condition you bake into a rule must be consistent with -what the predicate's classifier checks, or refinement rejects -parameter samples that look correct on paper. - -## Object Types - -__TYPES_DIGEST__ - -## Options - -Plans (for `sim.refine` / `sim.run`) and rules must match these typed -signatures and parameter boxes exactly: - -__OPTIONS_DIGEST__ - -__TOOLS_BLOCK__ - -## This session - -Read the data-structures file first, then explore the trajectory data -with `run_python`. Write your simulator to `__SIMULATOR_FILE__`, -exporting a `RESIDUAL_ENV` subclass with `AGENT_PARAM_SPECS` and `RESIDUAL_FEATURES`, and -iterate with `Edit` and re-scoring. Pass `task_idx` explicitly to -`sim.reset`; `sim.task(task_idx)` prints a task digest. Finish with the -deliverables listed in the system prompt: a final `sim.fit()`, the -GO/NO-GO check, the decision record, `./open_questions.md`, and -`./strategy.md`. - - -## Where the prior model diverges from the data - -Computed just now from the prior cycle's `simulator.py` with its -parameters refit to all trajectories above, so the remaining -mismatches need structural fixes, not tuning. Re-score any edit with -`sim.residuals()` (same report, current file): - -__REPORT__ - - -## Where the base simulator diverges from the data - -No prior model exists yet, so every feature below is an unmodeled -mechanism: this is the map of what your `simulator.py` needs to cover, -each with its worst transition located in the data. Re-score any edit -with `sim.residuals()` (same report, current file): - -__REPORT__ - - -The base simulator's own source code is available (read-only): - -__REF_LISTING__ - -These files are byte-identical to the code your base-sim rollouts -execute: scene geometry and constants, body construction, stepping, -and state read/write. They deliberately omit the environment's hidden -domain-specific step, the residual dynamics you are here to model, and -its task generation and goal semantics. Use them to ground hypotheses -(masses, damping, substeps per action, how switches toggle) instead of -re-measuring those from data. - - -## Available Tools - -__TOOL_LISTING__ - - -## Task objective (env ground-truth reward) - -__DESCRIPTION__ - -The trajectory roster above shows each interaction episode's -env-computed reward. In `run_python`, `evaluate_trajectory(states, -actions=None, task_idx=0)` is the task's reward model: it scores any -state sequence with the same rules, a collected trajectory's `states` -and `actions`, or a rollout of your simulator (where a rule that -replays physics runs on your belief simulator at its current fit, so -the verdict is only as trustworthy as the simulator). It returns -`{reward, solved, note}`; `solved` means the episode is scored as a -success, a rollout can reach the goal atoms and still be -`solved=False`, and `note` says what a replaying rule simulated and -on what. Label transitions with `(option, objects, params)` (`None` -for an unlabeled one) so such a rule replays your action rather than -its canonical one. - - -Prior cycle state: __PRIOR_FILES__ already exist in the sandbox from a -previous learning cycle. Read them first: they are the previous cycle's -committed result and a reasonable starting point for incremental -refinement, though a fresh rewrite is fine if the prior approach looks -fundamentally wrong. Structural decisions are not binding across -cycles: re-read the decision record at the top of `simulator.py` and -re-decide the architecture itself (what the base sim carries and what -the rules model, which features the rules own, the latent structure, -whether disclosed base-sim parameters should be identified) rather -than only tuning what exists. In particular, if the trajectory roster -shows goal-reaching episodes scored `solved=0`, suspect a structural -modeling error (for example mis-calibrated base physics that the rules -only paper over near the fit data), not only parameter values. Earlier -versions are in `./simulator_versions/` and `./predicates_versions/` -(named `cycle_XXX_vers_YYY_*.py`); cross-reference the roster's -provenance tags against those files to see which rules and predicates -produced each failed plan. - - -## Zero-shot synthesis - -No trajectory has been recorded and none will be before you finish: -this session is the whole learning phase, and what you write here is -what the planner uses on the test tasks. The trajectory counts above -are zero for that reason, and there is no residual scan or divergence -report to read. Build the simulator and its parameters from the task -description, the object types and options, the scene (`sim.task`, -`sim.reset`, `sim.render`) and your own knowledge of the mechanisms -involved, and validate the result with `sim.refine` / `sim.run` -rollouts of a full plan; `sim.fit` and `sim.residuals` have no data to -work with. State each mechanism you commit to, and the evidence you -would want for it, in the decision record. diff --git a/predicators/agent_sdk/prompts/learn_partial_observability.md b/predicators/agent_sdk/prompts/learn_partial_observability.md deleted file mode 100644 index 0a3982486a..0000000000 --- a/predicators/agent_sdk/prompts/learn_partial_observability.md +++ /dev/null @@ -1,46 +0,0 @@ -# Hidden-state guidance for messages and predicates - - -## Partial observability - -Some causally important quantities may be absent from the observation -entirely (under no name), possibly several, possibly none. Inspect the -trajectories first to judge whether any hidden process is at work and -which observable features are your window into it; then, if latents -are needed, declare subclass `MODEL_STATE_INIT` and implement `update_model_state`. - - -### Predicate signature - -Classifiers may stay observation-only or take an optional `latent` -kwarg. The latent block is available at refinement time too: the -planner threads it through `state.latent` across search nodes, and -`Predicate.holds` routes it into classifiers that opted in. Be -defensive: at the very first step `state.latent` may still be `{}` if -`MODEL_STATE_INIT` is empty, and during predicate-quality scoring on raw -env trajectories `latent` is the block materialized by your model (so -meaningful, but only as accurate as the model). - -```python -# Observation-only (robust to an inaccurate model; preferred when the -# observable carries enough signal): -Predicate("ProcessDone", [widget_type], - lambda s, objs, latent=None: - s.get(objs[0], "progress") > 0.5) - -# Latent-aware (inherits simulator correctness; defend against -# missing keys at step 0): -Predicate("ProcessDone", [widget_type], - lambda s, objs, latent=None: - (latent or {}).get("level", 0.0) >= params["done_thresh"]) -``` - -The kwarg must be named exactly `latent` for the routing to apply. -Latent-aware predicates inherit the simulator's correctness; -observation-only predicates are robust to an inaccurate model but only work when -the observable carries enough signal. - -`sim.predicates()` rolls each trajectory through your simulator to -materialize the latent before scoring classifiers, so latent-aware -predicates get a real block there. Use its report to localize failures -(bad model versus bad threshold). diff --git a/predicators/agent_sdk/prompts/learn_predicate_invention.md b/predicators/agent_sdk/prompts/learn_predicate_invention.md deleted file mode 100644 index 839d964dc9..0000000000 --- a/predicators/agent_sdk/prompts/learn_predicate_invention.md +++ /dev/null @@ -1,140 +0,0 @@ -# Predicate invention (learn phase) - -Appended to the synthesis system prompt and first message by -`AgentSimPredicateInventionApproach`. - - -## Predicate Invention (required for plan subgoals) - -You also invent the symbolic predicates the planner uses as subgoal -atoms in plan sketches. Only `Holding` is provided as a primitive; -placement, device-state, and process-completion predicates do not -exist until you invent them. - -Goals are presented in natural language (see the first message) and -goal achievement is checked externally by the environment through -`is_goal_state(state, task_idx)` / `train_tasks[task_idx].goal_holds(state)`. -You need not invent goal-named predicates or match environment -predicate names: invented predicates exist for plan-sketch subgoals -(gating `Wait`, `Place`, and similar steps) and can be named freely. - -Define them in `predicates.py` (path given in the first message): - -```python -LEARNED_PREDICATES: List[Predicate] -``` - -The exec namespace pre-injects `Predicate`, `np`, and a -`_type` binding for each env type (for example `widget_type`, -`fixture_type`). The names below are illustrative; use the types, -features, and parameter names your digests and the trajectory data -report. - -```python -# Placement: object xy within a learned distance of the fixture's -# functional point, NOT its recorded origin (see "Geometric gates"). -# The local-frame offset is declared as ParamSpecs in simulator.py -# and shared with the rule that gates the same physics. -def _widget_at_fixture(s, objs): - widget, fixture = objs - rot = s.get(fixture, "rot") - cos_r, sin_r = np.cos(rot), np.sin(rot) - rot_mat = np.array([[cos_r, -sin_r], [sin_r, cos_r]]) - local_offset = np.array([params["fixture_local_dx"], - params["fixture_local_dy"]]) - origin = np.array([s.get(fixture, "x"), s.get(fixture, "y")]) - anchor = origin + rot_mat @ local_offset # world-frame point - widget_xy = np.array([s.get(widget, "x"), s.get(widget, "y")]) - dist = np.linalg.norm(widget_xy - anchor) - return dist < params["widget_at_fixture_dist"] - -LEARNED_PREDICATES = [ - Predicate("WidgetAtFixture", [widget_type, fixture_type], - _widget_at_fixture), - # Device state: a feature exceeding a fixed cutoff (no learned param). - Predicate("FixtureActive", [fixture_type], - lambda s, objs: s.get(objs[0], "is_on") > 0.5), - # Process completion: a rule-driven feature reaches a learned threshold. - Predicate("WidgetReady", [widget_type], - lambda s, objs: s.get(objs[0], "progress") >= params["ready_threshold"]), -] -``` - -A pre-injected `params` view is in scope and always reads the current -fitted values of every `ParamSpec` declared in `simulator.py`; after -each refit, predicates reading `params["name"]` see the new values. -Whenever one physical gate drives both a rule's firing condition and a -predicate's "subgoal reached" check, declare its parameters (the -distance threshold and the local-frame anchor offset it is measured -from) once in `PARAM_SPECS` and reference `params["name"]` from both. -That keeps the two anchored to the same point and gives the offset a -fitting signal from the rule's step data. A parameter used only by -predicates has no fitting signal and stays at its `init_value`, so -choose those initial values carefully. - -What you typically need: - -- Placement predicates (object at a target location) for every - open-ended option such as `Place`; without them refinement picks an - arbitrary location. -- Device-state predicates (on/off) for every toggle option. -- Process-completion predicates over the features your rules drive, so - `Wait` steps know when to terminate. Keep classifier thresholds - consistent with the rules' saturation values; an inconsistency makes - `sim.fit` look fine while `sim.refine` gets stuck on the `Wait` - subgoal. -- Coverage: every option you expect in a sketch should have predicates - that express its post-condition, so every sketch step can carry a - subgoal annotation. Annotations are checked against the real state - during execution to detect and replan diverged steps; a step with no - annotatable effect is unmonitored. While drafting sketches, a step - you cannot annotate with any invented predicate is a missing - predicate. - -Verify every classifier against the scene and the data. A classifier -picks features and parameter values, and both can be wrong, so commit -neither from intuition: follow the threshold-fitting protocol in -"Geometric gates" for every numeric cutoff, and use __SCENE_WORKBENCH__ -for geometry and `run_python` for the numeric sweep over trajectory -states. - -`sim.predicates()` validates cheaply (first-flip step, monotonicity, -coverage across all trajectories) and is also the loader: it updates -the predicate set `sim.refine` uses, so call it after every edit to -`predicates.py` and before re-running refinement. On goal-reaching -trajectories (`reached_goal=True` in `describe_trajectory`) a milestone -predicate should flip from false to true exactly once and stay true. -On failed interaction trajectories (`reached_goal=False`) the same -predicate may fire while the rest of the trajectory shows no goal -completion; that is the signature of an over-loose threshold (the -predicate fires, the downstream physics does not follow), so tighten -it or share the gating parameter with the rule so they are fitted -jointly. - -Predicates persist across online cycles: the file is preserved between -synthesis sessions, and every successful `Write`/`Edit` (plus a final -post-session check) is snapshotted to -`predicates_versions/cycle_XXX_vers_YYY_predicates.py`. Each cycle -re-runs synthesis with the full trajectory history, so failed past -attempts remain visible. - - -## Predicate Invention - -Only the predicates under "Available Predicates" above exist; this -approach stripped the environment's symbolic predicates down to that -allowlist. Invent every other subgoal predicate in `__PREDICATES_FILE__` -as `LEARNED_PREDICATES`, following the system prompt's "Predicate -Invention" section. - -__GOAL_BLOCK__ - -Workflow: edit `predicates.py`, call `sim.predicates()` in -`run_python`, then run `sim.refine` / `sim.run` with sketches that -reference your invented names. Any predicate a sketch references must -exist in `predicates.py` first. - - -Step 4's sketches need subgoal predicates that do not exist until you -invent them: before validating, write them to `predicates.py` and load -them with `sim.predicates()` (see "Predicate Invention"). diff --git a/predicators/agent_sdk/prompts/learn_program_message.md b/predicators/agent_sdk/prompts/learn_program_message.md deleted file mode 100644 index d434eb96f2..0000000000 --- a/predicators/agent_sdk/prompts/learn_program_message.md +++ /dev/null @@ -1,76 +0,0 @@ -# Program world model synthesis (learn) first message - -Composed by `AgentProgramWorldModelApproach._build_program_learn_message` -through `learn_prompts.build_program_learn_message`. The message carries -this cycle's data and digests; the rules of the phase live in the system -prompt (`learn_program_system.md`). - - -Synthesize a world model program for this environment. There are -__N_TRAJS__ recorded trajectories (__N_TRANSITIONS__ skill-level -transitions) available: __N_DEMOS__ oracle demonstration(s), which -reached the goal by construction, and __N_INTERACTION__ interaction -trajectory/ies collected during online learning, some of which may -have failed to reach the goal. - -__TRAJECTORY_LISTING__ - -Each trajectory carries a `train_task_idx`. `is_goal_state(state, -task_idx)` (equivalently `train_tasks[task_idx].goal_holds(state)`) -checks a single state for the goal atoms. Reaching the goal atoms does -not by itself mean an episode is solved; when a task objective is -stated below, score full trajectories with `evaluate_trajectory`. Use -`is_goal_state` to confirm which trajectories reached the goal atoms -and to treat failed interaction trajectories as counterexamples: places -where the environment disagreed with what a skill was expected to do. - -__OBJECTIVE_BLOCK__ - -__PRIOR_STATE_BLOCK__ - -Data-structure source code is at: __STRUCTS_REF__ - -## Available Predicates (for subgoal annotations) - -__PREDICATE_LISTING__ - -Subgoal annotations in plans for `sim.refine` / `sim.run` must -reference these predicate names with matching arity and types. - -## Object Types - -__TYPES_DIGEST__ - -## Options - -Plans (for `sim.refine` / `sim.run`) and your `transition` must match -these typed signatures and parameter boxes exactly: - -__OPTIONS_DIGEST__ - -__TOOLS_BLOCK__ - -## This session - -Read the data-structures file first, then explore the trajectory data -with `run_python`. Write your world model to `__WORLD_MODEL_FILE__`, -defining `LATENT_FEATURES`, `initial_latent`, and `transition`, and -iterate with `Edit` and `sim.score()`. Pass `task_idx` explicitly to -`sim.reset`; `sim.task(task_idx)` prints a task digest. Finish with the -deliverables listed in the system prompt: a final `sim.score()`, the -GO/NO-GO check, the decision record, `./open_questions.md`, and -`./strategy.md`. - - -## Zero-shot synthesis - -No trajectory has been recorded and none will be before you finish: -this session is the whole learning phase, and what you write here is -what the planner uses on the test tasks. The trajectory counts above -are zero for that reason, and `sim.score` has no data to score -against. Build the world model from the task description, the object -types and options, the scene (`sim.task`, `sim.reset`, `sim.render`) -and your own knowledge of the mechanisms involved, and validate it -with `sim.refine` / `sim.run` rollouts of a full plan. State each -mechanism you commit to, and the evidence you would want for it, in -the decision record. diff --git a/predicators/agent_sdk/prompts/learn_program_system.md b/predicators/agent_sdk/prompts/learn_program_system.md deleted file mode 100644 index bb818bc894..0000000000 --- a/predicators/agent_sdk/prompts/learn_program_system.md +++ /dev/null @@ -1,176 +0,0 @@ -# Program world model synthesis (learn phase) system prompt - -Composed by `AgentProgramWorldModelApproach._build_synthesis_system_prompt` -through `learn_prompts.build_program_learn_system_prompt`. The paper's -code-world-model arm: an option-level program over the object-centric -state with no physics engine underneath, in the form of Pinductor / -POMDP Coder. The plan-format section is shared with `learn_system.md`. - - -You are synthesizing a world model for a robotic manipulation -environment as a standalone program: given the observed state and the -skill the robot executes, predict the observed state after the skill -completes. There is no physics engine behind your program. Robot -motion, grasping, contact, placement, and every process the -environment runs (delayed effects, gradual changes, propagation between -objects, hidden mechanisms) are yours to model, at the level of one -skill call at a time. - - -## What you produce - -One file, `world_model.py` (path given in the first message), defining -three top-level names: - -```python -LATENT_FEATURES: Dict[str, List[str]] # {type_name: [hidden feature names]} your latent tracks - -def initial_latent(obs: State, rng: np.random.Generator) -> Dict[str, Any]: - """A draw of the hidden state consistent with the first observation.""" - -def transition(obs: State, latent: Dict[str, Any], option: _Option, - rng: np.random.Generator) -> Tuple[State, Dict[str, Any], int]: - """The observed state after `option` runs to completion from `obs`, - the updated hidden state, and the number of low-level steps used.""" -``` - -`obs` is a `State` over the environment's objects with only the -OBSERVABLE features (`obs.get(obj, "x")`; `obs.set(obj, "x", v)` on -the copy you return; `list(obs)` iterates the objects). `option` is a -ground skill: `option.name`, `option.objects` (typed, in signature -order), `option.params` (the continuous parameter vector, in the -option's box), and for `Wait` the target atoms in -`option.memory.get("wait_target_atoms")`. Return a new `State` with -exactly the same objects (start from `obs.copy()`), your updated latent -dict, and a positive step count (the environment's horizon is counted -in low-level steps, so a skill that takes longer must cost more). - -The latent is yours: a plain dict of whatever the environment hides -(process progress, attachments, cure state, per-object counters). -Declare in `LATENT_FEATURES` what it tracks. `initial_latent` may be -stochastic through `rng` - when the first observation leaves the -hidden state genuinely undetermined, return a draw over the -possibilities: the harness keeps a particle belief of several draws, -scores the model with it, and re-validates every plan under every -particle. A deterministic `initial_latent` is a belief with one -particle. - - -## Modeling guidance - -- Model a skill's effect on every feature it changes, not only on the - ones the goal names. Gripper state, the held object's pose while it - is carried, the poses of objects that move together, and the - features a process advances are all read by the planner's - predicates and by the next skill. -- Skills fail. When a skill's parameters put its target out of reach, - into collision, or onto an unsupported spot, return an outcome the - environment would produce (the object drops, stays put, the gripper - closes on nothing), not the intended one; a model that always - succeeds validates plans that fail. -- Processes take time. A hidden process that advances while the robot - does other things advances in your latent on EVERY transition - (including `Wait`), by an amount tied to the step count you return, - so that a plan's timing is checked. Wait terminates when its target - atoms hold or on the first observable change; model its duration - accordingly. -- Ground every mechanism in the recorded data: find the transitions - where a feature changes, characterize when it changes and by how - much, and encode that. A mechanism you suspect but never observed is - a hypothesis; record it in the decision record and, when the goal - requires it, ship it as a labelled hypothesis with the experiment - that would confirm it first in `./open_questions.md`. -- Thresholds and geometric gates (how close is close enough, which - side of a fixture) come from the data too: find the recorded - attempts on both sides of the boundary and place the gate between - them. When in doubt, tighten toward the empirical boundary; a - permissive model passes plans the environment rejects. -- `predicates.py` has no learned parameters in this arm: write - thresholds as literals there, kept consistent with the ones in your - transition, or read the hidden state through `state.latent` (a dict - while the planner rolls your model; `None` on a raw observation, so - a predicate the plan needs on the real robot must not depend on it). - - -## Tools - -`run_python` is the one tool over the data, and it carries the `sim` -probe over your CANDIDATE world model (reloaded whenever the file -changes): - -- `sim.score()`: the model's score on the recorded trajectories - a - particle-filter pseudo-likelihood over your hidden state (0 is a - perfect model; each unit is one feature-std of mean error per - transition), the per-feature error table, and the worst - transitions. The inner-loop signal: re-score after every edit, and - read the worst transitions to find WHICH mechanism is wrong. - `sim.score(traj_idxs=[...])` restricts the data. -- `sim.refine(plan)`: backtracking parameter search on a plan sketch - through your model. -- `sim.run(plan)`: forward rollout through your model with subgoal - checking. -- `sim.reset(task_idx=..., mods={...})` and `sim.render(label, - annotations=[...])`: stage a state and render it with overlays. -- `sim.predicates()`: score `predicates.py` on the recorded data. -- `trajectories`, `describe_trajectory(i)`, `train_tasks`, - `is_goal_state`: the recorded evidence. Each action carries the - skill that produced it (`action.get_option()`), so the option-level - transitions are the spans between skill changes. - -Probe rollouts are candidate predictions; do not confuse them with the -recorded `trajectories`. - - -### Score vs. forward validation - -`sim.score` and refine-then-run test complementary things: pointwise -accuracy on what was recorded versus goal reachability on what a plan -needs. A model can score well on the data and still make a gate wide -enough that refinement accepts a placement the environment rejects, or -advance a process too fast so that a `Wait` looks sufficient in the -model and is not on the robot. Use `sim.score` as the fast inner loop -and refine-then-run as the slow, goal-relevant gate before declaring -done. When `sim.refine` passes but `sim.run` reports a subgoal not -reached, the model is more permissive than the environment's effective -behavior: tighten the threshold toward the empirical boundary, never -loosen it. - - -## Deliverables of a learning session - -- Decision record. Begin `world_model.py` with a short comment stating - your key modeling choices and the evidence behind them: which - mechanisms the data shows, which features each skill writes, what - the latent tracks and how it is initialized, and every hypothesis - shipped without direct evidence. Later cycles read this record - before deciding what to keep. -- Completeness. Work through every mismatch the score's worst - transitions reveal in this one session; each deferred mechanism - costs a full explore-learn-test round trip. -- Final score and GO/NO-GO. Before ending, run `sim.score()` on the - final file and record it in the decision record. Then refine a full - solve of the train task in your model and validate it with several - trials (`sim.refine`, then `sim.run(plan, trials=5)`). Record the - verdict with the plan's weakest margin, the smallest distance from - any step's operating point to a threshold your model enforces. NO-GO - means the next test episode will likely fail: put exactly what is - missing at the top of `./open_questions.md`, and keep `./strategy.md` - current with what the next exploration should collect. - - -## Workflow - -1. Explore the data with `run_python`: for each skill, which features - change between its start and its end, and under what conditions. -2. `Write` `world_model.py`; `Edit` to iterate. -3. Score with `sim.score()` and read the worst transitions to find - the mechanism to fix. Repeat until the remaining error is noise. -4. Propose an option-skeleton plan and validate it: `sim.reset(task_idx=i)`, - `sim.refine(plan, require_goal=True)`, then a continuous `sim.run` - of the refined plan from a fresh `sim.reset(task_idx=i)`. A stuck - refine step means a gate is too tight or a mechanism is missing; a - refine-pass whose `sim.run` diverges means the model is too - permissive. Fix and re-validate; do not declare done until both - pass.__WORKFLOW_EXTRA__ -5. Finish with the deliverables above: final `sim.score()`, GO/NO-GO, - decision record, `./open_questions.md`, `./strategy.md`. diff --git a/predicators/agent_sdk/prompts/learn_system.md b/predicators/agent_sdk/prompts/learn_system.md deleted file mode 100644 index 2beaf0e3ea..0000000000 --- a/predicators/agent_sdk/prompts/learn_system.md +++ /dev/null @@ -1,95 +0,0 @@ -# Simulator learning session instructions - - -You are synthesizing a parameterized residual-dynamics simulator for a -robotic manipulation environment. - -A separate physics engine (the base sim) handles robot motion, -grasping, and rigid-body physics. Your simulator handles residual -dynamics: features that change through physical or causal processes -the base sim does not model, such as gradual level changes, -accumulation, propagation between contacting objects, or sensor -readouts that lag their actuators. - - -### Harness parameter estimation is DISABLED in this run - -The harness fits no parameter from data: `sim.fit` refuses, -`sim.residuals(fit_params=True)` and `sweep_params=` are unavailable, -and the deployed model uses every `ParamSpec` and `AGENT_PARAM_SPECS` -entry exactly as you declared it. That makes the declaration itself -the estimate: - -- `init_value` is the point estimate the planner uses. Choose it from - your knowledge of the mechanism and from what the recorded data - shows (`sim.residuals()` at the declared values, `sim.run` / - `sim.refine` rollouts, `describe_trajectory`, or an estimator you - write yourself against `trajectories`); do not leave a placeholder. -- `lo` / `hi` is the plausible interval. It is used as such: the - validation gate re-rolls plans at values across this interval and the - exploration ensemble is drawn uniformly from it, so a box that is too - wide rejects every plan and one that is too narrow hides your own - uncertainty. Declare a finite box for every parameter. - -Everywhere the rest of this prompt says to fit, score, or refit a -parameter, read "declare it and check the rollouts at the declared -values" instead. - - -## Plan format for `sim.refine` / `sim.run` - -One option call per line, with every option argument supplied as a -typed object reference (`obj:type`), matching the options digest in -your prompt exactly. The parser is strict: an omitted argument is not -auto-filled. Example: - -``` -PickWidget(robot:robot, widget0:widget) -Place(robot:robot) -> {WidgetAtFixture(widget0:widget, fixture0:fixture)} -ActivateFixture(robot:robot, fixture0:fixture) -Wait(robot:robot) -> {WidgetReady(widget0:widget)} -``` - -The names are illustrative; use the options, types, and predicates your -prompt digests list. Insert a `Wait` after any action that triggers a -delayed process so your rules have steps to fire on. - -Subgoal annotations (`-> {Atom(obj:type, ...)}` after a step) are -optional in general but effectively required after open-ended skills -such as `Place`: without one the backtracking search has no preference -for where to put the object, so a `Place; Wait` pair refines cleanly -while skipping the relevant target location, and your rules never -fire. That looks like a rule bug but is a missing subgoal. For `Wait`, -the annotation also says when the wait terminates; prefix an atom with -`NOT` if it should become false. - - -## Deliverables of a learning session - -- Begin `simulator.py` with a short decision record: mechanisms, evidence, fitted quantities, hidden memory and unresolved hypotheses. -- Reconcile every mechanism exercised by the recordings with the model. - Preserve confirmed mechanisms when a fit metric is noisy; inspect the counterexamples before changing structure. -- Ground physical changes in recorded behavior the base mispredicts. - Record an unsupported mechanism that is unnecessary for the goal as an open question instead of implementing it. - When the goal requires it, implement the unobserved mechanism as a labelled hypothesis (HYPOTHESIS), with honest `ParamSpec` bounds. - Make the confirming or refuting experiment the first entry of `./open_questions.md`, naming the observation that distinguishes the alternatives. - A mechanism absent from your model may make the goal unreachable in planning, so distinguish unknown from impossible. -- Declare uncertain constants as `ParamSpec`s with plausible ranges. - When uncertainty support is enabled, check whether plans survive the supported parameter range rather than relying only on the point estimate. -- Run a final explicit `sim.fit()` if the model declares learnable constants, then `sim.validate()` on the full recordings. - Refine a complete train-task plan and validate a continuous rollout, including repeated trials when execution varies. - Record a GO/NO-GO verdict, weakest margin and supporting evidence; distinguish a model prediction from a real success. - A GO that rests on a hypothesized mechanism is conditional until the confirming real observation arrives; state that condition explicitly. -- Write `./open_questions.md` as a ranked list of unresolved mechanisms or parameters. - Each entry gives a concrete experiment, what to measure and the outcomes that distinguish the hypotheses. - Remove questions the new evidence settles. -- Write `./strategy.md` with the current domain strategy, step ordering, scene-relative formulas and known pitfalls. - Update advice when evidence changes; state uncertainty honestly. - - -## Workflow - -1. Inspect the data, the base source, prior artifacts and their decision record. -2. Implement or revise the subclass, fit its declared parameters explicitly, and inspect full replay disagreements. -3. Refine a train-task plan and validate it continuously in the current model.__WORKFLOW_EXTRA__ -4. Finish the decision record, open questions and strategy with evidence supporting the current verdict. diff --git a/predicators/agent_sdk/prompts/solve_query.md b/predicators/agent_sdk/prompts/solve_query.md deleted file mode 100644 index 145f295fef..0000000000 --- a/predicators/agent_sdk/prompts/solve_query.md +++ /dev/null @@ -1,163 +0,0 @@ -# Solve / explore query - -Composed by `sketch_prompts.build_solve_prompt`. The query carries the -task (goal, scene, vocabulary), the run state (records, scheduled -plans, open questions), and this episode's instructions. Rules that -hold for every episode live in the system prompt (`solve_system.md`). - - -__OPENING__ - -__GOAL_NL_SECTION__ - -__SCORING_SECTION__ - -__GOAL_ATOMS_SECTION__ - -__EXPERIMENT_SECTION__ - -## Initial State Atoms - -__ATOMS__ - -## Initial State Features - -__STATE__ - -__IMAGE_SECTION__ - -## Objects - -__OBJECTS__ - -## Available Options - -__OPTIONS__ - -## Available Predicates (for subgoal annotations) - -__PREDICATES__ - -__TRAJECTORY_SUMMARY__ - -__TOOLS_SECTION__ - -__STRATEGY_SECTION__ - -__ATTEMPTS_SECTION__ - -__JOURNAL_SECTION__ - -__SCHEDULED_PLANS_SECTION__ - -## Instructions - -__INSTRUCTIONS__ - - -Solve the task below: produce a plan that reaches its goal. - - -Design this episode's experiment for the task below. Reaching the goal -is the most informative experiment available, so treat solving the task -as part of information gathering. - - -## Goal Description - -__GOAL_NL__ - - -## Scoring (env ground-truth reward) - -__OBJECTIVE__ - -Decode every reward you observe with this rule before hypothesizing any -other mechanism; there are no hidden reward terms. - - -## Goal Atoms - -__GOAL_ATOMS__ - - -## Experiment Guidance - -__GUIDANCE__ - - -(none: no atom of the available predicates holds initially) - - -## Available Tools - -__TOOL_LIST__ - - -## Domain Strategy (advisory, written during learning) - -__STRATEGY__ - - -## Attempt Log (recorded by the harness) - -__ATTEMPTS__ - - -## Solve Journal (./journal.md) - -__JOURNAL__ - - -(no journal entries yet) - - -## Plans Already Scheduled This Cycle - -The plan(s) below are already queued to run on this task before any -learning happens, so their data will be collected regardless of what -you propose now. - -__PLANS__ - -Propose a plan whose data is complementary rather than redundant: cover -the open questions, mechanisms, or parameter regions the plan(s) above -leave unmeasured. A goal-reaching plan is still preferred when it can -carry that coverage; when it cannot, a designed experiment that settles -what the scheduled plans will not is the better use of this episode. -Only when the model is believed correct everywhere and no meaningfully -different goal-reaching plan exists, repeat the best -plan.__CERTIFIED_RULE__ - - -If a scheduled plan is marked belief-certified, this episode is the -second test of the belief model, and one success of one plan is weak -evidence. In order of preference: (1) a STRUCTURALLY different -goal-reaching plan (a different option sequence, order, grasp, or -contact arrangement) validated through the same `submit_plan` gate; (2) -when no structurally different plan exists for this goal, the same -structure with materially different parameters (a different placement -pose, offset, or timing, not a jitter), validated the same way; (3) -only as a last resort, the certified plan resubmitted unchanged. State -which of the three you chose and why. A certified plan that then fails -for real is the most informative outcome this episode can produce, not -a loss. - - -Inspect the environment with your tools, then produce the plan and -deliver it through the capture gate as the system prompt's Deliverable -section specifies. When a step does not reach its subgoal, -__STUCK_ADVICE__ - - -Inspect the environment with your tools, design the episode as the -system prompt's Exploration Setting specifies, and output the plan -lines as your final text. - - -tune that step's parameters from the rendered image and the object -poses (working principles 4 and 5), then re-test it. - - -revise the plan (different objects, a different ordering, an added -intermediate step, or a corrected annotation), then re-test it. diff --git a/predicators/agent_sdk/prompts/solve_system.md b/predicators/agent_sdk/prompts/solve_system.md deleted file mode 100644 index 1ba6cd0a6f..0000000000 --- a/predicators/agent_sdk/prompts/solve_system.md +++ /dev/null @@ -1,372 +0,0 @@ -# Solve / explore system prompt - -Composed by `sketch_prompts.build_solve_system_prompt`. One identity and -one deliverable section are chosen by phase and mode; the remaining -sections are shared. The query prompt (`solve_query.md`) carries the -task data and the run state and never restates these rules. - - -You are a planning agent. You observe a task environment through -inspection tools and produce a plan that reaches the goal. - - -You are an exploration agent in an online learning loop. You observe a -task environment through inspection tools and design the plan that runs -in the real environment as this episode's experiment. - - -## Deliverable - -A plan captured by `submit_plan`. Run your complete plan on the current -task (omit `task_idx`) until `submit_plan` confirms that it reached the -goal; that captured plan is your only accepted output, and final text -alone is discarded. After the capture, repeat the plan lines as your -final text. Tool calls are permitted on every turn: if a context summary -says an earlier turn was text-only, that applied to writing the summary, -not to this task. - - -## Deliverable - -A closed-loop policy in `./policy.py`, validated by `submit_policy`. -Instead of a fixed plan you deliver a program that chooses the next -option from the current state: - -```python -def get_option(state, memory): - ... -``` - -- `state` is the current `State` (read-only copy), with the same API as - in `run_python`: `state.get(obj, "feature")`, `for obj in state`, - `obj.name`, `obj.type`. -- `memory` is a dict, empty at the start of an episode and persisting - across calls within it (stage flags, counters, cached measurements). - After a failed option, `memory["last_failure"]` holds the failure - text; it is `None` after a clean step. Branch on it to recover. -- Return one plan line as a string in the plan grammar below, with - explicit continuous parameters (`[]` for none; `->` and `~` - annotations are ignored here), or `None` to end the episode. -- `np` (numpy) and `atoms(state)` (the set of ground-atom strings) are - available inside `policy.py`. -- Execution semantics are identical in the belief simulator and the - real environment: `get_option` is called at every option boundary - with the actual current state; an option failure does not end the - episode (it is reported through `memory["last_failure"]` and you are - asked again); an exception, an unparsable line, or an ungroundable - line ends it; at most __MAX_OPTIONS__ options run per episode. -- After a failure, change something (parameters, target, or action). - Re-issuing the identical failing line __MAX_REPEATED_FAILURES__ times - in a row ends the episode as a policy bug, and so does re-issuing one - identical line that keeps completing with no observable state change - __MAX_REPEATED_NOOPS__ times in a row. - -Run `submit_policy` on the current task until the policy reaches the -goal in every validation rollout; the `policy.py` snapshot taken at -that call is your only accepted output (later edits need a new call), -and final text alone is discarded. Test recovery first: -`sim.run_policy()` in `run_python` runs `./policy.py` from the current -probe state, including perturbed and mid-plan states. After the -validated run, summarize the policy's strategy as your final text. Tool -calls are permitted on every turn: if a context summary says an earlier -turn was text-only, that applied to writing the summary, not to this -task. - - -## Deliverable - -Your final plan text: the experiment that runs in the real environment. -Output only the plan lines at the end, after any analysis. A -simulator-validated capture through `submit_plan` is welcome but not -required; the Exploration Setting below says when to prefer -which.__CERTIFIED_NOTE__ - - -A plan that passes the `submit_plan` capture gate (goal reached in -every fresh belief rollout) is executed verbatim as this episode's -solve attempt; only an unvalidated plan is treated as an experiment. - - -## Plan grammar - -One option per line: - -``` -OptionName(obj1:type1, obj2:type2)__PARAM_SLOT__ -> {Pred(obj1:type1), NOT Pred2(obj1:type1, obj2:type2)} -Wait(robot:robot)__WAIT_SLOT__ -> {Pred3(obj1:type1)} -``` - -- Every object reference is typed (`obj:type`), in arguments and atoms - alike. Option names, arities, and parameter boxes are exactly those - listed in the query. -- __PARAMS_RULE__ -- `-> {atoms}` is the step's subgoal annotation: the atoms that should - newly hold, or stop holding (`NOT`), once the step succeeds. Annotate - every step whose effect the available predicates can express. - Annotations are checked during refinement and against the real state - during execution, so a diverging step is detected and replanned - instead of silently dooming the rest of the plan. Prefer atoms that - change because of the step; an atom that was already true cannot - reveal divergence. A step without an annotation is checked only for - having executed. -- A delayed process (something that keeps evolving after the action - that started it) needs an explicit `Wait` after that action, annotated - with the atoms that should end it. `Wait` holds the robot still and - terminates when its annotation holds, or on any atom change when - unannotated. Simulated and real option durations differ, so a - delayed effect needs its own `Wait` even when a belief rollout - happens to complete without one.__GROUND_SAMPLER_RULE__ - - -`[p1, p2, ...]` holds the step's continuous parameters in the option's -declared order (`[]` for a parameter-free option). Parameters are -executed exactly as written. - - -Continuous parameters are omitted: a backtracking search finds them -from the subgoal annotations. - - - -- For `sim.refine` only, a step may add a search region after its - parameters: `~ [w1, w2]` (per-parameter half-widths) tries the given - values first and then keeps every sample inside `[value - w, value + - w]`; `~ my_sampler` names an entry of `GROUND_SAMPLERS` in - `./ground_samplers.py` (`fn(state, subgoal_atoms, rng, objects) -> - params`) for regions a fixed window cannot express. The file is - reloaded on every `sim.refine` call. - - -## Tools - -- `submit_plan(plan_text)` runs the plan on the current task in the - belief simulator with your exact parameters, no search. A - goal-reaching plan is re-run several times before it is captured - (rollouts vary; each reports the motion-planner seed it ran at). A - plan reported FLAKY failed one of those rollouts: reproduce that - rollout (`rollout_seed=` to `submit_plan`, or - `sim.run(plan_text, seed=...)`), read why, add margin to the fragile - step, and resubmit. `validation_rollouts=N` requests a stricter gate - up front; `sim.run(plan_text, trials=N)` measures reliability without - submitting.__VALIDATION_GATE__ -- `run_python(code)` exposes the `sim` probe over the belief simulator: - `sim.run(plan_text, seed=..., trials=...)` is a forward rollout with - subgoal checks; `sim.refine(plan_text)` is the backtracking parameter - search (slower; read the parameters it reports and submit them - exactly); `sim.reset(mods={...})` followed by `sim.render(...)` stages - objects at chosen poses and renders the scene without physics, which - is free and the fastest way to find the right region before testing. -- Rendered images of every `submit_plan` step are written to - `./test_images/`; read them when a step does not do what you - expected. - - -Capture also requires the plan to succeed on a grid of perturbations -spanning one standard deviation of the identified physical parameters; -a plan that fails any grid point is reported PARAM-SENSITIVE. Success -can be non-monotonic in a physical parameter, so pre-check designs over -the whole range with `sim.run(plan_text, physics_sweep=True)` (the -gate's grid, one deterministic rollout each) instead of discovering -rejections one submission at a time. - - -Capture additionally re-runs the plan under the posterior members of -the learned rule parameters (the fit's uncertainty about the thresholds -and offsets it learned); failing under any member is reported -PARAM-SENSITIVE. A design that only works at the fitted point estimate -of an uncertain constant fails either this gate or the real -environment, whose true constant lies somewhere in that posterior. - - -Capture also requires every step to be necessary: the plan is re-run -once per step with that step removed, and if the goal is still reached -without a step the plan is reported REDUNDANT naming it. A captured plan -is an explanation of how the goal comes about, so it must not carry -steps whose absence changes nothing (a Wait on atoms that already hold, -an action on an object your model says is uninvolved). Submit the -shortest plan your model needs, and read a REDUNDANT report as evidence -about the model: a step you believed necessary was not. - - -## Working principles - -1. Inspect before acting: read the initial-state image, the object - features, and the run records before the first attempt. -2. Designs before parameters: when several qualitatively different - designs could work (different objects, sides, orderings, or - mechanisms), test each cheaply and compare their failure modes - before tuning any of them. Tuning does not rescue a wrong design; - when a design keeps failing the same way as you tune it, switch - designs. -3. Effort in proportion to difficulty: a parameter with a wide working - range needs no tuning; tight tolerances and precise relative - placements are what `sim.refine` is for. -4. Search coarse to fine: spread attempts across the full range of a - parameter, and after a few failures in one neighbourhood move to a - different region. Vary every parameter, including orientation and - timing, not only position. -5. Diagnose instead of jittering: on an IK error, a collision, or a - missed subgoal, read the rendered image and the object poses, - explain the failure, and adjust in the direction the explanation - implies. -6. Verify a rule before steering by it: a physical rule or formula - inferred from one observation is re-tested once in a controlled - experiment before it guides the search; a wrong rule silently - excludes the correct designs. -7. Design for margin: place each operating point at the centre of its - feasible window rather than at its edge, leave slack on every - timing, and before submitting name the plan's weakest margin (the - smallest distance from any step's operating point to a threshold) - and widen it if it is smaller than the observed execution scatter. -8. Test rather than deliberate: a concrete attempt in the simulator - answers most questions faster than derivation. Keep reasoning - concise. - - -9. Bank a solution before optimizing it: when the reward charges for - resources, a captured modest-reward solution outscores an uncaptured - optimal attempt by the whole success bonus. Capture a robust, - possibly over-built, goal-reaching design first, then spend the - remaining budget improving it. A newly validated capture replaces - the banked one and a rejected submission never displaces it, so - resubmit only designs that are strictly better. - - -## Run records - -These files in your working directory persist across sessions of this -run: - -- `./journal.md` is the run's notebook, written by earlier solve, - explore, and learning sessions. Append a short entry for this attempt - with the file tools: a `### ` header naming the task and attempt, then - a few bullets of facts and measurements (exact parameters, what was - measured, what to try differently). No verdicts such as "impossible". -- `./attempts.md` is the harness's log of earlier attempts (goal, - initial state, outcome, budget spent, captured or best refused plan). - Facts, not advice; do not edit it. -- `./strategy.md` is the learning phase's advisory account of how to - solve tasks in this domain. Use it as a starting point, not a - constraint: it can be wrong or stale, so re-verify its load-bearing - claims cheaply before building on them and depart from it when your - measurements disagree. -- `./open_questions.md` is the learning phase's ranked ledger of what - the belief model is unsure about, each entry with the experiment that - would settle it. -- `./session_logs/` holds earlier queries and tool results. - -Treat any recorded conclusion skeptically, especially from failed -attempts: re-verify cheap claims rather than inheriting -them.__JOURNAL_PROTOCOL__ - - - - -Journal protocol for a solve attempt: - -- A design the attempt log records as having reached the goal in the - real environment is the incumbent: reproduce it unless the record - also shows it failing since, or a model update invalidates one of its - steps. Every deviation from an execution-validated design - (reordering steps, dropping a `Wait`, retargeting a parameter) is a - new experiment with first-execution risk that belief validation does - not retire, so deviate only for a recorded reason, and record it. -- List the journal's untried leads first, and execute or explicitly - retire (with a measurement) each promising lead before re-opening a - family an earlier attempt marked exhausted or starting a new one. -- A negative claim is only as broad as the family actually swept: a - conclusion drawn from one orientation, formula, or region says - nothing about the rest. -- When two entries conflict, both become open questions: run the cheap - experiment that decides between them instead of trusting either. - - -## Exploration setting - -- The loop. Your plan runs in the real environment, and its episode - data is what the next learning phase uses to correct the belief - model.__EARLY_STOP_NOTE__ -- The belief model. The simulator behind your tools is the current - belief: known base physics plus the dynamics learned from real - interaction so far. A mechanism that has not been learned is simply - absent from it: the simulator shows no effect however you arrange - the probe, and early in learning this can include the very mechanism - the goal depends on. Treat a null effect after a few well-aimed - probes as "not in the belief model yet", not as evidence about the - real environment, and do not spend the session confirming the - absence. -- Choosing the experiment. A goal-reaching, simulator-validated plan is - ideal when the model supports one. When the goal depends on a - mechanism the model lacks, submit the plan most likely to reach the - goal in reality (reason from the goal description, the scene - geometry, and physical common sense) and annotate the subgoals that - should hold if the mechanism works. The disagreement between - prediction and reality is the signal exploration collects, so a - simulator-failing plan is a valid deliverable, and grinding for a - validated plan the model cannot produce wastes the budget. -- Verbatim execution. Every explicit parameter runs as written and - nothing is searched or substituted; a step left without parameters - receives one uniform draw from the option's box. Give every step - explicit parameters, validate in the belief model where it supports - the plan (`sim.run`, `sim.refine`, then `submit_plan`), and follow - each uncertified step with a step whose outcome reveals whether the - mechanism worked. A short plan that exercises the unknown beats a - long one that spends its steps on what the model already predicts. -- What a cycle's data must contain. Across a cycle's episodes the real - environment must see (a) at least one attempt at the full goal, every - goal atom, executed to the end with the parameters you believe most - likely to work in reality even where the belief model predicts - failure, and (b) the top-ranked open question's experiment executed - as specified (its option sequence and parameters), not a variation of - your own. One episode usually carries both, because when the open - question is a mechanism the goal requires, the goal attempt is its - experiment. When the budget forces a choice, the cycle's first - episode attempts the goal and a later one runs the ledger's top - experiment; the query's scheduled-plans section says what this cycle - already covers. -- One episode, many measurements. Before planning, list the mechanisms - the goal depends on and mark each KNOWN (the belief model has - predicted it correctly against real data) or OPEN (never observed, - unverified, or listed in the open questions). Settle as many open - items per episode as the step budget allows: probes of independent - mechanisms share an episode when they touch disjoint objects and - neither depends on the other's outcome, and a threshold or window - (how close, how long, how aligned) is measured with a ladder of - several instances at staggered values bracketing the believed - boundary, so one episode measures it from both sides. Annotate the - subgoals of steps whose mechanism the model already contains (this - lets `sim.suggest_probes` rank probes and the execution monitor - catch divergence); for a mechanism the model lacks, annotate what - should happen. Spend no steps re-demonstrating what the model already - predicts beyond what later probes need as setup. -- The first cycle. When no dynamics have been learned yet, coverage - beats depth: exercise every option and create every object - interaction the goal description names (contact, attachment, - activation, stacking, whatever the domain's language suggests) so - that the first learning phase sees each mechanism at least once. - Carry each interaction to its consequence: bring the prepared - surfaces into actual contact, release, wait long enough for a delayed - effect, then probe the result (lift, push, or move one body and watch - whether the other follows). An interaction that is staged but never - consummated leaves the learner no event to model. -- Records. Append measurements to `./journal.md` as you go (a short - entry per experiment, numbers first). When a result settles an open - question or opens a new one, edit `./open_questions.md` directly; - the next learning phase designs its work from that file. Do not edit - `./strategy.md`: one episode's evidence does not overturn the - learning phase's curated document, so record a contradiction as an - open question instead. - - -The loop concludes early once the exploration plans solve training: -__ATTEMPTS_CLAUSE__ must reach the goal for real, and the plan must have -validated in the belief model; a lucky real success from a plan the -model could not certify does not count. Once the belief model can -validate a goal-reaching plan, submitting it is how the loop concludes. - - -The loop concludes early once __PHASES__ every test task. Test attempts -plan with the belief model, so what ends the loop is the belief model -becoming reliably correct; your episodes count toward that only through -the model corrections their data enables, not through reaching the goal -themselves. diff --git a/predicators/agent_sdk/prompts/subclass_model.md b/predicators/agent_sdk/prompts/subclass_model.md index deb637aaa5..d3c081d40d 100644 --- a/predicators/agent_sdk/prompts/subclass_model.md +++ b/predicators/agent_sdk/prompts/subclass_model.md @@ -142,26 +142,6 @@ Execution tracking uses the same callback on real observations; this is an infer Do not treat it as measured truth or as a particle filter. Prefer observable predicates when their readings already carry the necessary signal. - -## Fit and validate complete rollouts - -Edit `./simulator.py`, then explicitly call `sim.fit()` to estimate its declared constants from the recorded trajectories. -Edits are loaded on the next probe call; a rollout does not implicitly fit parameters. -Before fitting, the model uses its carried or declared values and is marked UNFITTED. -If there are no learnable constants, skip fitting and call `sim.validate()`. - -`sim.validate()` replays every selected recording at the values currently deployed for planning, including recordings a robust fit rejected. -`sim.residuals()` uses full simulator replay for subclass models to expose accumulated error; the report labels the parameter values it scores. -`sim.fit(traj_idxs=[...])` and explicit validation parameter overrides are diagnostics and publish nothing. -Compare candidates on the same recordings, inspect per-trajectory failures and preserve counterexamples. -A low fitting error on a selected subset does not establish model fidelity or task solvability. - -Use `sim.refine(plan)` to search skill parameters, then run the resulting plan continuously with `sim.run(plan)` and check each annotated subgoal. -Use `sim.reset(task_idx=..., mods=...)` and `sim.render(label, annotations=[...])` to inspect geometry. -Evaluate trajectory success with the supplied evaluator when available; its verdict on a simulated trajectory depends on the model's fidelity. -Prefer additional simulator checks over spending real steps on a prediction that disagrees with recorded evidence. -Keep speculative mechanisms labeled as hypotheses and state what observation would distinguish competing explanations. - ## Supplied physical parameter menu diff --git a/predicators/agent_sdk/sketch_prompts.py b/predicators/agent_sdk/sketch_prompts.py deleted file mode 100644 index 6959e85448..0000000000 --- a/predicators/agent_sdk/sketch_prompts.py +++ /dev/null @@ -1,426 +0,0 @@ -"""Prompt construction for the solve and explore phases. - -Two builders, both rendered from the Markdown templates in -``predicators/agent_sdk/prompts`` (see :mod:`prompt_templates`): - -- :func:`build_solve_system_prompt` composes ``solve_system.md``: the - agent's identity, its deliverable contract, the plan grammar, the - tool semantics, the working principles, the run-record protocol, - and (explore phase) the exploration setting. Everything that holds - for every episode of a phase lives here. -- :func:`build_solve_prompt` composes ``solve_query.md``: the task - (goal, scene, vocabulary), the run state (records, scheduled plans, - open questions), and this episode's instructions. - -The query never restates a rule from the system prompt; the split is -what keeps each rule stated exactly once. -""" -import re -from typing import List, Optional, Sequence, Set - -from predicators import utils -from predicators.agent_sdk.prompt_templates import render -from predicators.settings import CFG -from predicators.structs import LowLevelTrajectory, ParameterizedOption, \ - Predicate, Task - -_BLANK_RUN_RE = re.compile(r"\n{3,}") - - -def _join(parts: Sequence[str]) -> str: - """Join non-empty prompt parts with blank lines, squeezing runs of blank - lines that empty optional sections leave behind.""" - text = "\n\n".join(p for p in parts if p) - return _BLANK_RUN_RE.sub("\n\n", text).strip("\n") + "\n" - - -def build_early_stop_note() -> str: - """The exploration setting's early-stop sentence, from ``CFG``. - - Empty when early stopping is off. Describes the rule the run uses - (train-driven: every attempt must be certified and solve for real; - test-driven: consecutive perfect test phases). - """ - if not CFG.online_learning_early_stopping: - return "" - if CFG.online_learning_early_stopping_by_test_solve_rate: - n_perfect = CFG.online_learning_early_stopping_consecutive_perfect_tests - phases = ("the next test phase solves" if n_perfect <= 1 else - f"{n_perfect} consecutive test phases each solve") - return render("solve_system", "early_stop_test", phases=phases) - attempts_clause = ("every episode of a cycle" if - CFG.online_learning_early_stopping_require_all_attempts - else "each train task's first episode of a cycle") - return render("solve_system", - "early_stop_train", - attempts_clause=attempts_clause) - - -def build_solve_system_prompt( - *, - explore: bool, - policy_mode: bool = False, - propose_params: bool = True, - ground_samplers: bool = False, - physics_margin: bool = False, - necessity: bool = False, - rule_param_margin: bool = False, - use_journal: bool = True, - execute_certified_plan: bool = True, - early_stop_note: str = "", - policy_max_options: int = 0, - policy_max_repeated_failures: int = 0, - policy_max_repeated_noops: int = 0, -) -> str: - """Compose the solve-phase or explore-phase system prompt. - - ``explore`` selects the exploration identity, deliverable, and - setting (``early_stop_note`` and ``execute_certified_plan`` only - matter there); otherwise ``policy_mode`` selects the closed-loop - policy deliverable over the captured plan. ``propose_params`` and - ``ground_samplers`` shape the grammar; ``physics_margin`` and - ``rule_param_margin`` describe the capture gate's extra checks; - ``use_journal`` includes the run-record protocol. - """ - assert not (explore and policy_mode), ( - "exploration delivers a plan sketch even in policy-mode runs") - identity = render("solve_system", - "identity_explore" if explore else "identity_solve") - if explore: - certified_note = "" - if execute_certified_plan: - certified_note = " " + render("solve_system", "certified_note") - deliverable = render("solve_system", - "deliverable_explore", - certified_note=certified_note) - elif policy_mode: - deliverable = render( - "solve_system", - "deliverable_policy", - max_options=str(policy_max_options), - max_repeated_failures=str(policy_max_repeated_failures), - max_repeated_noops=str(policy_max_repeated_noops)) - else: - deliverable = render("solve_system", "deliverable_plan") - - params_rule = render( - "solve_system", - "params_rule_propose" if propose_params else "params_rule_search") - ground_sampler_rule = "" - if propose_params and ground_samplers: - ground_sampler_rule = "\n" + render("solve_system", - "ground_sampler_rule") - grammar = render("solve_system", - "grammar", - param_slot="[p1, p2]" if propose_params else "", - wait_slot="[]" if propose_params else "", - params_rule=params_rule, - ground_sampler_rule=ground_sampler_rule) - - validation_gate = "" - if physics_margin: - validation_gate += " " + render("solve_system", - "validation_gate_physics") - if rule_param_margin: - validation_gate += " " + render("solve_system", - "validation_gate_rule_params") - if necessity: - validation_gate += " " + render("solve_system", - "validation_gate_necessity") - tools = render("solve_system", "tools", validation_gate=validation_gate) - - principles = render("solve_system", "principles") - if not explore: - principles += "\n" + render("solve_system", "banking") - - run_records = "" - if use_journal: - journal_protocol = "" - if not explore: - journal_protocol = "\n\n" + render("solve_system", - "journal_protocol_solve") - run_records = render("solve_system", - "run_records", - journal_protocol=journal_protocol) - - setting = "" - if explore: - note = (" " + early_stop_note.strip()) if early_stop_note else "" - setting = render("solve_system", - "exploration_setting", - early_stop_note=note) - - return _join([ - identity, deliverable, grammar, tools, principles, run_records, setting - ]) - - -def build_solve_prompt( - task: Task, - *, - all_predicates: Set[Predicate], - all_options: Set[ParameterizedOption], - trajectory_summary: str = "", - tool_names: Optional[Sequence[str]] = None, - experiment_guidance: str = "", - scheduled_plans: Optional[Sequence[str]] = None, - initial_image_section: str = "", - propose_params: bool = False, - require_tool_validation: bool = False, - explore_mode: bool = False, - journal: str = "", - strategy: str = "", - attempts: str = "", -) -> str: - """Compose the solve or explore query for ``task``. - - ``propose_params`` labels each option's parameter box as proposed by - the agent (otherwise as found by the search) and picks the stuck- - step advice. ``require_tool_validation`` selects the capture-gate - instructions; ``explore_mode`` the experiment instructions (the two - are exclusive). ``scheduled_plans`` lists the plans this cycle - already queued, so the agent is asked for a complementary one. - ``journal``, ``attempts``, and ``strategy`` are the run records' - contents (``journal.py``); the protocol for using them is in the - system prompt. - """ - assert not (explore_mode and require_tool_validation), ( - "explore_mode accepts an uncaptured experiment sketch, which " - "contradicts the hard capture gate of require_tool_validation") - - init_state = task.init - objects = sorted(init_state, key=lambda o: o.name) - obj_lines = [f" {obj.name}: {obj.type.name}" for obj in objects] - - # Only expose goal atoms whose predicate is in the agent's current - # predicate set: approaches that strip env predicates rely on - # goal_nl and must not leak atoms of predicates the agent invents. - goal_atoms = [ - str(a) for a in sorted(task.goal, key=str) - if a.predicate in all_predicates - ] - - option_lines = [] - for opt in sorted(all_options, key=lambda o: o.name): - type_sig = ", ".join(t.name for t in opt.types) - params_dim = opt.params_space.shape[0] - param_info = "" - if params_dim > 0: - low = opt.params_space.low.tolist() - high = opt.params_space.high.tolist() - label = "params" if propose_params else "auto-searched params" - desc = (", ".join(opt.params_description) - if opt.params_description else f"{params_dim}d") - param_info = f" [{label}: {desc}, range {low} to {high}]" - option_lines.append(f" {opt.name}({type_sig}){param_info}") - - atoms = utils.abstract(init_state, all_predicates) - atom_lines = [str(a) for a in sorted(atoms, key=str)] - - pred_lines = [] - any_latent = False - for pred in sorted(all_predicates, key=lambda p: p.name): - type_sig = ", ".join(t.name for t in pred.types) - line = f" {pred.name}({type_sig})" - if pred.accepts_latent: - line += " [reads belief latent]" - any_latent = True - if pred.natural_language_assertion is not None: - names = [t.name for t in pred.types] - line += f": {pred.natural_language_assertion(names)}" - pred_lines.append(line) - if any_latent: - pred_lines.append( - " ([reads belief latent]: its truth can depend on belief-only " - "state that real observations never carry, so an annotation " - "using it may be dropped from closed-loop monitoring at " - "capture as execution-unverifiable - prefer predicates over " - "observable features for monitored subgoals.)") - - goal_nl_section = "" - if task.goal_nl: - goal_nl_section = render("solve_query", - "goal_nl", - goal_nl=task.goal_nl) - - # The env's public reward form (success condition plus costs) when - # the task ships an evaluator that states one; never oracle values. - scoring_section = "" - evaluator = getattr(task, "evaluator", None) - if evaluator is not None and evaluator.objective_description(): - scoring_section = render("solve_query", - "scoring", - objective=evaluator.objective_description()) - - goal_atoms_section = "" - if goal_atoms: - goal_atoms_section = render("solve_query", - "goal_atoms", - goal_atoms="\n".join(goal_atoms)) - - experiment_section = "" - if experiment_guidance: - experiment_section = render("solve_query", - "experiment_guidance", - guidance=experiment_guidance) - - tools_section = "" - if tool_names: - tools_section = render("solve_query", - "tools", - tool_list="\n".join(f" - {t}" - for t in tool_names)) - - strategy_section = "" - if strategy: - strategy_section = render("solve_query", "strategy", strategy=strategy) - - attempts_section = "" - if attempts: - attempts_section = render("solve_query", "attempts", attempts=attempts) - - journal_section = "" - if journal or attempts: - journal_section = render("solve_query", - "journal", - journal=journal - or render("solve_query", "no_journal")) - - scheduled_section = "" - if scheduled_plans: - plans = "\n".join(f"Plan {i + 1}:\n{p}" - for i, p in enumerate(scheduled_plans)) - certified_rule = "" - if any("belief-certified" in p for p in scheduled_plans): - certified_rule = "\n\n" + render("solve_query", "certified_rule") - scheduled_section = render("solve_query", - "scheduled_plans", - plans=plans, - certified_rule=certified_rule) - - if explore_mode: - instructions = render("solve_query", "instructions_explore") - elif require_tool_validation: - stuck = render("solve_query", - "stuck_tune" if propose_params else "stuck_revise") - instructions = render("solve_query", - "instructions_capture", - stuck_advice=stuck) - else: - instructions = render("solve_query", - "instructions_capture", - stuck_advice=render("solve_query", - "stuck_revise")) - - body = render( - "solve_query", - "skeleton", - opening=render("solve_query", - "opening_explore" if explore_mode else "opening_solve"), - goal_nl_section=goal_nl_section, - scoring_section=scoring_section, - goal_atoms_section=goal_atoms_section, - experiment_section=experiment_section, - atoms="\n".join(atom_lines) if atom_lines else render( - "solve_query", "no_atoms"), - # 4 decimals (0.1 mm at metric scale), matching the per-step - # state dumps: the default 2 dp misstated mm-scale geometry - # (a 0.025 half-extent printed as 0.03) and sent agents to - # sim.task()/sim.state() to re-derive every number. - state=init_state.dict_str(indent=2, num_decimal_points=4), - image_section=initial_image_section.strip("\n"), - objects="\n".join(obj_lines), - options="\n".join(option_lines), - predicates="\n".join(pred_lines), - trajectory_summary=trajectory_summary.strip("\n"), - tools_section=tools_section, - strategy_section=strategy_section, - attempts_section=attempts_section, - journal_section=journal_section, - scheduled_plans_section=scheduled_section, - instructions=instructions, - ) - return _join([body]) - - -def _executed_options(traj: LowLevelTrajectory) -> List[str]: - """The option sequence a trajectory executed, consecutive repeats - collapsed, from the options its actions carry; empty when the actions carry - none (a demo replayed from raw actions, or a pickle that dropped them).""" - out: List[str] = [] - for act in traj.actions: - if not act.has_option(): - continue - opt = act.get_option() - text = f"{opt.name}({', '.join(o.name for o in opt.objects)})" - if not out or out[-1] != text: - out.append(text) - return out - - -def summarize_trajectories( - trajectories: Sequence[LowLevelTrajectory], - predicates: Set[Predicate], - train_tasks: Optional[Sequence[Task]] = None) -> str: - """The ``## Trajectory Summary`` query section, or "". - - Per recent trajectory (the last - ``CFG.agent_sdk_max_trajectories_in_context``): the option plan it - executed, the env's verdict on it (reward and whether the goal - atoms held at the end, when the episode was evaluated), and the - atoms gained and lost between its first and last state. Shared by - the solve approach and the explorers. - - Trajectories are numbered by their index in the full list, so a - number means the same episode in every session of a run. The - verdict lines are what an agent with no learn phase otherwise never - sees: the model-free arm's cycle-1 session used to read only - "Trajectory 0: 42 steps" for an episode whose plan had failed. - """ - if not trajectories: - return "" - max_trajs = CFG.agent_sdk_max_trajectories_in_context - recent = trajectories[-max_trajs:] - first = len(trajectories) - len(recent) - lines = [ - f"\n## Trajectory Summary ({len(trajectories)} total, " - f"showing last {len(recent)})" - ] - for i, traj in enumerate(recent): - n_steps = len(traj.actions) - init_atoms = utils.abstract(traj.states[0], predicates) - final_atoms = utils.abstract(traj.states[-1], predicates) - new_atoms = final_atoms - init_atoms - lost_atoms = init_atoms - final_atoms - lines.append(f"\nTrajectory {first + i}: {n_steps} steps") - executed = _executed_options(traj) - if executed: - lines.append(" Executed: " + " -> ".join(executed)) - verdict: List[str] = [] - reward = traj.env_reward - if reward is not None: - verdict.append(f"env reward {reward:.2f}") - terminated = traj.env_terminated - if terminated is not None: - verdict.append("goal atoms held at the end" - if terminated else "goal NOT reached") - elif train_tasks is not None: - try: - task = train_tasks[traj.train_task_idx] - except (IndexError, ValueError, AttributeError): - task = None - if task is not None: - held = task.goal_holds(traj.states[-1]) - verdict.append("goal atoms held at the end" - if held else "goal NOT reached") - if verdict: - lines.append(" Outcome: " + ", ".join(verdict)) - if new_atoms: - lines.append( - " Gained: " + - f"{', '.join(str(a) for a in sorted(new_atoms, key=str))}") - if lost_atoms: - lines.append( - " Lost: " + - f"{', '.join(str(a) for a in sorted(lost_atoms, key=str))}") - return "\n".join(lines) diff --git a/predicators/agent_sdk/tools/__init__.py b/predicators/agent_sdk/tools/__init__.py index 55258bbd18..c31998d24f 100644 --- a/predicators/agent_sdk/tools/__init__.py +++ b/predicators/agent_sdk/tools/__init__.py @@ -3,14 +3,14 @@ This package replaces the former single-module ``tools.py``. Layout: - ``registry``: tool-name rosters and the session tool-list surface. -- ``context``: ``ToolContext`` / ``PlanCapture`` shared session state. +- ``context``: ``ToolContext``, the shared session state. - ``results``: tool-result formatting and sandbox-file helpers. - ``sandbox_guard``: sandbox-escape screening for agent-supplied text. -- ``budget``: solve-attempt budget footer and watchdog. +- ``budget``: the round's budget footer and the run_python watchdog. - ``scene``: scene rendering and state-manipulation helpers. - ``verdicts``: task-evaluator verdicts and ground-sampler loading. -- ``testing`` / ``exploration``: the static MCP - tool builders, assembled by ``assembly.create_mcp_tools``. +- ``exploration``: the static ``run_python`` builder, assembled by + ``assembly.create_mcp_tools``. - ``digests``: the type / option / task / trajectory digest renderers shared by the prompts and the probe. - ``snapshots``: versioned write-time snapshots of agent-edited files. @@ -25,14 +25,13 @@ """ # pylint: disable=unused-import from predicators.agent_sdk.tools.assembly import create_mcp_tools -from predicators.agent_sdk.tools.context import PlanCapture, ToolContext +from predicators.agent_sdk.tools.context import ToolContext from predicators.agent_sdk.tools.params_view import _ParamsView from predicators.agent_sdk.tools.predicate_synthesis import \ make_predicate_quality_loader from predicators.agent_sdk.tools.registry import ALL_TOOL_NAMES, \ BUILTIN_TOOLS, EXPLORATION_TOOL_NAMES, MCP_SERVER_NAME, \ - SYNTHESIS_TOOL_NAMES, TESTING_TOOL_NAMES, get_allowed_tool_list, \ - list_session_tool_names + SYNTHESIS_TOOL_NAMES, get_allowed_tool_list, list_session_tool_names from predicators.agent_sdk.tools.results import _make_coercing_tool, \ _make_spilling_text_result, session_log_filename from predicators.agent_sdk.tools.sandbox_guard import \ @@ -56,8 +55,6 @@ "SANDBOX_INTROSPECTION", "SANDBOX_SYSTEM_ROOTS", "SYNTHESIS_TOOL_NAMES", - "TESTING_TOOL_NAMES", - "PlanCapture", "ToolContext", "agent_render_resolution", "apply_state_modifications", diff --git a/predicators/agent_sdk/tools/assembly.py b/predicators/agent_sdk/tools/assembly.py index 29d8ce3116..e8a856d997 100644 --- a/predicators/agent_sdk/tools/assembly.py +++ b/predicators/agent_sdk/tools/assembly.py @@ -5,7 +5,6 @@ from predicators.agent_sdk.tools.exploration import _build_exploration_tools from predicators.agent_sdk.tools.results import _make_coercing_tool, \ _make_spilling_text_result -from predicators.agent_sdk.tools.testing import _build_testing_tools def create_mcp_tools(ctx: ToolContext, @@ -35,11 +34,8 @@ def create_mcp_tools(ctx: ToolContext, # probe in one namespace), so the solve-phase instance is neither # built nor offered there. extra_names = {getattr(t, "name", "") for t in ctx.extra_mcp_tools} - _all = { - **_build_testing_tools(ctx, _text_result, tool), - **({} if "run_python" in extra_names else _build_exploration_tools( - ctx, _text_result, tool)), - } + _all = ({} if "run_python" in extra_names else _build_exploration_tools( + ctx, _text_result, tool)) if tool_names is None: tools = list(_all.values()) else: diff --git a/predicators/agent_sdk/tools/budget.py b/predicators/agent_sdk/tools/budget.py index 7540a32a0e..329f63dce3 100644 --- a/predicators/agent_sdk/tools/budget.py +++ b/predicators/agent_sdk/tools/budget.py @@ -8,25 +8,20 @@ def _budget_footer(ctx: ToolContext, rollouts_before: int = 0) -> str: - """``[budget]`` line appended to tool results during a solve attempt. + """``[budget]`` line appended to tool results during a play round. - Shows attempt wall-clock (elapsed, and the budget when one is set) - and the attempt's cumulative sim-rollout count (plus this call's - delta). Agents pace well when they can see a clock and terribly when - they can't: the 47k-rollout single-call sweep of run_20260717_230436 - ran 7 h with zero cost feedback. Empty when no attempt is in flight. + Shows the round's elapsed wall clock and its cumulative sim-rollout + count (plus this call's delta). Agents pace well when they can see a + clock and terribly when they can't: the 47k-rollout single-call + sweep of run_20260717_230436 ran 7 h with zero cost feedback. Empty + when no attempt is in flight. """ start = ctx.attempt_start if start is None: return "" parts = [] elapsed_min = (time.monotonic() - start) / 60.0 - deadline = ctx.attempt_deadline - if deadline is not None: - total_min = (deadline - start) / 60.0 - parts.append(f"attempt time {elapsed_min:.1f}/{total_min:.0f} min") - else: - parts.append(f"attempt time {elapsed_min:.1f} min") + parts.append(f"attempt time {elapsed_min:.1f} min") total_rollouts = ctx.attempt_rollout_count delta = total_rollouts - rollouts_before rollout_part = f"sim rollouts this attempt: {total_rollouts}" diff --git a/predicators/agent_sdk/tools/capture.py b/predicators/agent_sdk/tools/capture.py deleted file mode 100644 index 6b70de8462..0000000000 --- a/predicators/agent_sdk/tools/capture.py +++ /dev/null @@ -1,174 +0,0 @@ -"""The ``submit_plan`` capture decision, as a pure function. - -:func:`_decide_capture` encodes the run-verified capture gates in one -side-effect-free place: the handler in ``testing.py`` computes the -inputs, then performs the ctx mutations and message formatting its -decision calls for. Keeping the policy pure makes every guard -combination directly unit-testable -(``tests/agent_sdk/test_capture_decision.py``). -""" -import enum -from dataclasses import dataclass -from typing import Optional - - -class CaptureDecision(enum.Enum): - """What ``submit_plan`` does with the submitted plan.""" - # Goal reached, evaluator-certified, every validation rollout passed: - # captured and marked as a validated solve. - VALIDATED_CAPTURE = "validated_capture" - # Final-submission nudge: captured to execute for its honest reward, - # but not marked as a solve (see BestEffortReason). - BEST_EFFORT_CAPTURE = "best_effort_capture" - # Goal reached on rollout 1 but a validation repeat failed: refused, - # and later submissions on the task face the escalated gate. - FLAKY_NO_CAPTURE = "flaky_no_capture" - # Execution validation passed, but a rollout at +-1-posterior-sigma - # perturbed physical params failed: the plan has no margin to the - # physics fit's parameter error, so it is refused (the real env may - # sit anywhere in that range). - PARAM_SENSITIVE_NO_CAPTURE = "param_sensitive_no_capture" - # Every gate above passed, but the plan still reaches the goal with - # one of its steps removed: that step is padding, the plan is not an - # explanation of the goal, and it is refused naming the step. - REDUNDANT_NO_CAPTURE = "redundant_no_capture" - # Goal atoms reached via a route the task evaluator scores as a - # non-solve: refused. - REWARD_HACK_NO_CAPTURE = "reward_hack_no_capture" - # Goal reached on a train task instead of the current task: not - # captured, flagged loudly. - WRONG_TASK_NOTE = "wrong_task_note" - # Nothing to capture and nothing to flag. - NO_CAPTURE = "no_capture" - - -class BestEffortReason(enum.Enum): - """Why a best-effort capture cannot count as a validated solve.""" - GOAL_NOT_REACHED = "goal_not_reached" # honest shortfall - REWARD_HACK = "reward_hack" # evaluator scores the rollout a non-solve - FLAKY = "flaky" # a validation repeat failed - PARAM_SENSITIVE = "param_sensitive" # failed at perturbed physics - REDUNDANT = "redundant" # reaches the goal without one of its steps - - -@dataclass(frozen=True) -class CaptureOutcome: - """A capture decision; ``best_effort_reason`` is set iff the decision is - ``BEST_EFFORT_CAPTURE``.""" - decision: CaptureDecision - best_effort_reason: Optional[BestEffortReason] = None - - @property - def captured(self) -> bool: - """Whether the plan is captured as the current answer.""" - return self.decision in (CaptureDecision.VALIDATED_CAPTURE, - CaptureDecision.BEST_EFFORT_CAPTURE) - - -def _decide_capture(*, - capture_enabled: bool, - is_current_task: bool, - have_plan: bool, - goal_achieved: bool, - evaluator_rejected: bool, - reward_hack: bool, - flaky: bool, - best_effort_mode: bool, - have_validated_capture: bool, - param_sensitive: bool = False, - redundant: bool = False) -> CaptureOutcome: - """Decide what ``submit_plan`` does with an evaluated plan. - - Pure: no ctx access, no I/O - the caller supplies exactly what the - gates read and applies the side effects the decision calls for. - - - ``capture_enabled``: ``ctx.capture_goal_reaching_plans`` - only - approaches that consume captured plans set it. - - ``is_current_task``: the plan ran on the current solve/explore - task (no ``task_idx`` given), the only task whose captures count. - - ``have_plan``: the parsed plan grounded to at least one step. - - ``goal_achieved``: goal reached, clean to goal, and within the - episode horizon on the first rollout. - - ``evaluator_rejected``: a non-coarse evaluator verdict scored the - first rollout illegitimate. - - ``reward_hack``: ``evaluator_rejected`` and the rollout actually - reached the goal atoms (terminated) - an illegitimate route, as - opposed to an honest shortfall. - - ``flaky``: a validation repeat failed. - - ``best_effort_mode``: ``ctx.capture_best_effort_plan`` - set only - for the final-submission nudge after an attempt exhausted its - turn budget. - - ``have_validated_capture``: a validated-solve capture already - exists (``ctx.solved_plan_reached_goal``). - - ``param_sensitive``: execution validation passed but a rollout at - +-1-posterior-sigma perturbed physical params failed - the plan - has no margin to the physics fit's parameter error. - - ``redundant``: every other gate passed but the plan still reaches - the goal with one of its steps removed - that step is padding. - - With ``best_effort_mode`` (final-submission nudge after turn-cap - exhaustion) capture the submission unconditionally: honest - shortfall, evaluator-rejected rollout, and flaky repeat alike. The - budget is spent, so executing the agent's best plan for its honest - reward beats forfeiting the task (run_20260714_145053 task 4: a - goal-reaching but certificate-rejected final submission was refused - and the task forfeited, scoring n/a instead of its honest reward). - Only a validated solve is marked as one; everything else is a - best-effort capture that executes but cannot count as a solve - the - certificate still protects the score. - """ - # A best-effort capture never displaces a validated-solve capture. - best_effort_capture = best_effort_mode and not have_validated_capture - validated_solve = (goal_achieved and not reward_hack and not flaky - and not param_sensitive and not redundant) - if (capture_enabled and is_current_task - and (validated_solve or best_effort_capture) and have_plan): - if validated_solve: - return CaptureOutcome(CaptureDecision.VALIDATED_CAPTURE) - if not goal_achieved: - reason = BestEffortReason.GOAL_NOT_REACHED - elif reward_hack: - reason = BestEffortReason.REWARD_HACK - elif flaky: - reason = BestEffortReason.FLAKY - elif param_sensitive: - reason = BestEffortReason.PARAM_SENSITIVE - else: - reason = BestEffortReason.REDUNDANT - return CaptureOutcome(CaptureDecision.BEST_EFFORT_CAPTURE, reason) - if (capture_enabled and is_current_task and goal_achieved - and not evaluator_rejected and flaky): - # Loudly refuse a flaky capture: the agent still has this - # session to add margin and resubmit, which beats discovering - # the flakiness as a failed real episode. - return CaptureOutcome(CaptureDecision.FLAKY_NO_CAPTURE) - if (capture_enabled and is_current_task and goal_achieved - and not evaluator_rejected and param_sensitive): - # Loudly refuse a physics-margin failure: the fitted params are - # uncertain at the reported sigma, and the real env may sit - # anywhere in that range (run_20260723_091108: a capture - # validated 8/8 at the fitted friction failed deterministically - # at the true value just outside the design's success band). - return CaptureOutcome(CaptureDecision.PARAM_SENSITIVE_NO_CAPTURE) - if (capture_enabled and is_current_task and goal_achieved - and not evaluator_rejected and redundant): - # Loudly refuse padding: the plan reaches the goal without one of - # its steps, so that step explains nothing and only spends real - # episode steps (run_20260902_152811: three presses and a release - # for a goal the model reached with two presses and a Wait). - return CaptureOutcome(CaptureDecision.REDUNDANT_NO_CAPTURE) - if capture_enabled and is_current_task and reward_hack: - # Loudly refuse a reward hack: the rollout reaches the goal atoms - # but the evaluator's certificate rejects the route (e.g. the - # target was knocked over directly), and the real evaluator - # applies the same certificate, so it can never count as a solve. - # (Under a best-effort final submission the same plan is instead - # captured, flagged as a non-solve, to execute for its honest - # reward.) - return CaptureOutcome(CaptureDecision.REWARD_HACK_NO_CAPTURE) - if capture_enabled and not is_current_task and goal_achieved: - # Loudly flag a success that cannot count: agents have burned - # whole sessions validating on a train task, believing they - # were done (run_20260707_112310 test task 0, session 3). - return CaptureOutcome(CaptureDecision.WRONG_TASK_NOTE) - return CaptureOutcome(CaptureDecision.NO_CAPTURE) diff --git a/predicators/agent_sdk/tools/clearance.py b/predicators/agent_sdk/tools/clearance.py deleted file mode 100644 index 33cf81bb41..0000000000 --- a/predicators/agent_sdk/tools/clearance.py +++ /dev/null @@ -1,193 +0,0 @@ -"""Robot-clearance measurement for the capture gate. - -Certification's validation rollouts sample the belief's own execution -variability, but not the gap between the belief's realized poses and -the real executor's: every move phase terminates anywhere within its -pose tolerance, grasp heights vary within the descend tolerance, and -placed objects land with scatter. A plan whose robot passes within -that slop of a bystander certifies 8/8 by luck and loses the ninth -draw for real (2026-09-02 bridge seed3: two belief-certified explore -plans died on 6.5 mm and 9.9 mm real contacts against a block every -rollout had cleared). - -:class:`RobotClearanceProbe` measures, along each validation rollout's -low-level trajectory, the minimum distance between the robot's links -and every body the executor would treat as a bystander - excluding the -held object and its welded attachments (the skill factory's own -planning-scene reconstruction decides that), the current option's -argument objects (a pick's fingers straddle their target by design), -the bodies the option's skill declares it contacts (a push strikes its -appliance's switch, a separate body: boil/fan 2026-09-02 refused every -plan at -3 to -14 mm against the switch being pushed) and the object -held when the option starts (a Place releases it and retreats past it -at a distance set by the grasp, not by the executor's slop: domino -2026-09-02 refused every plan at ~5 mm against the domino just placed). -The verdict compares that minimum against the executor's pose slop, -``sqrt(move_to_pose_tol)``, read from the plan's own skills, so the -bar is the executor's and not a domain constant. -""" -from __future__ import annotations - -import logging -import math -from typing import Any, List, Optional, Sequence, Tuple - -import pybullet as p - -from predicators.structs import _Option - -# Only bodies within this distance of a robot link are queried; anything -# farther cannot be the minimum that matters. -_QUERY_DISTANCE = 0.05 - - -def phase_skill_of(options: Sequence[_Option]) -> Optional[Any]: - """The first PhaseSkill (duck-typed) behind a grounded plan's options. - - A PhaseSkill's built option binds its policy as a bound method, so - the skill - with its planning simulator and executor tolerances - - is reachable through it. ``None`` when no option is skill-factory - built or none carries a planning simulator. - """ - for opt in options: - skill = getattr(getattr(opt.parent, "policy", None), "__self__", None) - config = getattr(skill, "_config", None) - if (skill is not None and hasattr(skill, "_sim_collision_context") - and getattr(config, "simulator", None) is not None): - return skill - return None - - -class RobotClearanceProbe: - """Minimum robot-to-bystander clearance over validation rollouts. - - Feed each executed option's low-level trajectory to - :meth:`observe`; :attr:`min_dist` and :attr:`where` then describe - the closest approach seen so far. ``stride`` subsamples trajectory - states (a contact event spans many control steps, so every third - state loses nothing the verdict needs); the final state of every - option is always probed - phase goals are where the executor's - refusals were measured. - """ - - def __init__(self, skill: Any, stride: int = 3) -> None: - self._skill = skill - self._stride = max(1, stride) - self.min_dist = math.inf - self.where = "" - self.num_probes = 0 - - @property - def threshold(self) -> float: - """The executor's pose slop: its move-phase terminal tolerance.""" - return float(math.sqrt(self._skill._config.move_to_pose_tol)) # pylint: disable=protected-access - - def observe(self, label: str, option: _Option, - states: Sequence[Any]) -> None: - """Probe one option's trajectory states (best-effort diagnostic).""" - exempt = {o.name for o in option.objects} - if states: - exempt |= self._designed_contacts(option, states[0]) - last = len(states) - 1 - for k, state in enumerate(states): - if k % self._stride and k != last: - continue - try: - dist, body = self._min_robot_clearance(state, exempt) - except Exception as e: # pylint: disable=broad-except - logging.debug("Clearance probe skipped a state: %s", e) - continue - self.num_probes += 1 - if dist < self.min_dist: - self.min_dist = dist - self.where = (f"{label}, step {option.name}" - f"({', '.join(o.name for o in option.objects)})" - f" vs {body}") - - def _designed_contacts(self, option: _Option, state: Any) -> set: - """Names of the bodies this option touches by design: the ones its - skill declares (a push's switch, see ``PhaseSkill.contact_objects``) - and the object it holds when it starts (a Place's fingers open around - the block they release, and their retreat clears it by the grasp - geometry, not by the executor's pose slop). - - Best-effort: a failing lookup exempts - nothing. - """ - names: set = set() - skill = getattr(getattr(option.parent, "policy", None), "__self__", - None) - contacts = getattr(skill, "contact_objects", None) - if contacts is not None: - try: - names |= {o.name for o in contacts(state, option.objects)} - except Exception as e: # pylint: disable=broad-except - logging.debug("Clearance probe: contact lookup failed: %s", e) - try: - held = self._held_object_name(state) - except Exception as e: # pylint: disable=broad-except - logging.debug("Clearance probe: held lookup failed: %s", e) - held = None - if held is not None: - names.add(held) - return names - - def _held_object_name(self, state: Any) -> Optional[str]: - """The name of the object held in ``state``, if any.""" - _, _, names, held, _ = self._skill._sim_collision_context(state) # pylint: disable=protected-access - return None if held is None else names.get(held) - - def _min_robot_clearance(self, state: Any, - exempt: set) -> Tuple[float, str]: - skill = self._skill - sim = skill._config.simulator # pylint: disable=protected-access - _, bodies, names, _, _ = skill._sim_collision_context(state) # pylint: disable=protected-access - robot = sim._pybullet_robot # pylint: disable=protected-access - robot.set_joints(state.joint_positions) - client = sim._physics_client_id # pylint: disable=protected-access - best = _QUERY_DISTANCE - best_body = "" - for body in bodies: - name = names.get(body, str(body)) - if name in exempt: - continue - points = p.getClosestPoints(robot.robot_id, - body, - _QUERY_DISTANCE, - physicsClientId=client) - if not points: - continue - dist = min(pt[8] for pt in points) - if dist < best: - best, best_body = dist, name - return best, best_body - - def verdict(self) -> Tuple[bool, str, str]: - """``(ok, summary line, detail)``. - - ``ok`` is False when the closest approach is inside the - executor's pose slop; ``detail`` then names it for the refusal - message (empty when ok or when nothing was probed). - """ - if self.num_probes == 0: - return True, "", "" - thr = self.threshold - if not math.isfinite(self.min_dist): - return True, (f"min robot clearance: >{_QUERY_DISTANCE * 1e3:.0f}" - " mm"), "" - summary = (f"min robot clearance: {self.min_dist * 1e3:.1f} mm " - f"({self.where}; executor pose slop {thr * 1e3:.0f} mm)") - if self.min_dist < thr: - return False, summary, ( - f"the robot passes within {self.min_dist * 1e3:.1f} mm of a " - f"bystander ({self.where}), inside the executor's " - f"{thr * 1e3:.0f} mm pose slop") - return True, summary, "" - - -def clearance_lines(probe: Optional[RobotClearanceProbe]) -> List[str]: - """Report lines for a probe, empty when no probe ran.""" - if probe is None: - return [] - _, summary, _ = probe.verdict() - return [summary] if summary else [] diff --git a/predicators/agent_sdk/tools/context.py b/predicators/agent_sdk/tools/context.py index 26acb4ab90..fe957abbf8 100644 --- a/predicators/agent_sdk/tools/context.py +++ b/predicators/agent_sdk/tools/context.py @@ -12,25 +12,6 @@ ParameterizedOption, Predicate, State, Task, Type -@dataclass(frozen=True) -class PlanCapture: - """A captured plan popped off a :class:`ToolContext` in one piece. - - Returned by :meth:`ToolContext.take_plan_capture` so consumers see - the four ``solved_plan*`` fields as the single value they are: - ``plan`` is falsy when nothing was captured. - """ - plan: Optional[Any] - sketch: Optional[Any] - reached_goal: Optional[bool] - eval_reward: Optional[float] - validation_summary: Optional[str] = None - # Closed-loop policy mode: the validated policy.py SOURCE snapshot - # (mutually exclusive with ``plan``). The capture is falsy when both - # are None. - policy_source: Optional[str] = None - - @dataclass class ToolContext: """Shared mutable state between the approach and MCP tools.""" @@ -157,7 +138,7 @@ class ToolContext: # main.py's ``test_task_idx``. None outside the test phase. Threaded into # the saved session-log filename so test queries are attributable to a task. test_task_idx: Optional[int] = None - test_call_id: int = 0 # incremented per submit_plan call + test_call_id: int = 0 # incremented per probe rollout call # 0-based learning cycle (matching main.py's "ONLINE LEARNING CYCLE i"; # -1 = the offline pass) while a synthesis (learn) session is active, # None otherwise. Set/cleared around the synthesis query so tools that @@ -183,126 +164,40 @@ class ToolContext: # atoms are evaluable on real observations and refinement must keep # them as Wait targets; False (bare observations) strips them. latent_tracking_available: bool = False - # Set by submit_plan / submit_policy when a plan is verified - # to reach the goal on the CURRENT solve task: the simulator-verified plan - # (grounded options with found params) and the parallel subgoal sketch. - # The bilevel approach returns this directly instead of re-refining, so - # the agent's tool-validated answer is exactly what gets executed. None ⇒ - # nothing captured this query. - solved_plan: Optional[Any] = None - solved_sketch: Optional[Any] = None - # Whether the captured solved_plan counts as a validated solve in its - # belief-sim rollout(s): goal reached, evaluator-certified, and every - # validation rollout passed. False ⇒ it was a best-effort capture (see - # below). Cleared together with solved_plan. - solved_plan_reached_goal: Optional[bool] = None - # Gate for the above: only approaches that consume captured plans - # set this True, so other users of submit_plan record no spurious - # captures. - capture_goal_reaching_plans: bool = False - # Set (with capture_goal_reaching_plans) only for the final-submission - # nudge after an attempt exhausted its turn budget: submit_plan - # then captures the agent's submitted plan on the current task even if it - # does not reach the goal, is scored a non-solve by the task evaluator, - # or is flaky, so the approach executes the best-effort plan (for its - # honest reward) instead of paying for another full-budget attempt. A - # best-effort capture never displaces a validated-solve capture. - capture_best_effort_plan: bool = False - # Fresh-physics scope for capture-validation rollouts: a callable - # returning a context manager. While entered, ``ctx.option_model`` - # simulates on a freshly constructed env instance instead of the shared - # session env, whose reset cannot reconstruct state exactly (solver - # warm-start state, velocity residuals), making repeated rollouts - # correlated with each other and systematically offset from the fresh - # real env. Accepts an optional ``physical_overrides`` keyword (a - # param-name -> value dict applied to the fresh env on top of the - # identified params) for the physics-margin rollouts. Installed by + # Fresh-physics scope for validation rollouts: a callable returning a + # context manager. While entered, ``ctx.option_model`` simulates on a + # freshly constructed env instance instead of the shared session env, + # whose reset cannot reconstruct state exactly (solver warm-start + # state, velocity residuals), making repeated rollouts correlated + # with each other and systematically offset from the fresh real env. + # Accepts an optional ``physical_overrides`` keyword (a param-name -> + # value dict applied to the fresh env on top of the identified + # params) for the physics-sweep rollouts. Installed by # AgentSimLearningApproach (see ``_fresh_validation_env_scope``); - # None ⇒ validation rollouts share the session env. Gated by + # None ⇒ probe rollouts share the session env. Gated by # agent_plan_validation_fresh_env. validation_env_scope: Optional[Callable[..., Any]] = None # Candidate-aware counterpart: loads the deployed candidate before # cloning its physics and rebinds its option model for the whole rollout. - # Never substitute the solve-time model for a synthesis candidate. + # Never substitute the deployed model for a synthesis candidate. probe_validation_env_scope: Optional[Callable[..., Any]] = None - # Physics-margin points for the capture gate: a zero-arg callable - # returning the current grid of perturbations spanning +-1 posterior - # sigma of the identified physical params (full override dicts, - # ascending; empty when no fit with nonzero posterior width is - # deployed). A callable rather than a stored list so the points - # always track the LATEST applied fit. Installed by - # AgentSimLearningApproach; consumed by submit_plan under - # agent_plan_validation_physics_margin and by the sim.run physics - # sweep. + # Physics-margin points: a zero-arg callable returning the current grid + # of perturbations spanning +-1 posterior sigma of the identified + # physical params (full override dicts, ascending; empty when no fit + # with nonzero posterior width is deployed). A callable rather than a + # stored list so the points always track the LATEST applied fit. + # Installed by AgentSimLearningApproach; consumed by the sim.run + # physics sweep. physics_margin_provider: Optional[Callable[[], List[Dict[str, float]]]] = None - # Rule-parameter margin points for the capture gate: a zero-arg - # callable returning the calibrated rule-parameter ensemble (full - # fitted-param dicts drawn from the fit posterior; the same members - # info-seeking exploration scores with). The gate re-rolls a - # capture-eligible submission under each member so a plan that - # survives only at the point estimate of an uncertain learned - # constant is rejected as PARAM-SENSITIVE. Installed by - # AgentSimLearningApproach; consumed under - # agent_plan_validation_rule_param_margin. - # How the capture gate names one rule-param margin point and the - # set it came from in its reports. The program-world-model arm - # sweeps belief particles over the model's hidden state through - # the same gate and relabels them here. - rule_param_margin_label: str = "rule-param ensemble member" - rule_param_margin_note: str = ( - "calibrated posterior members of the learned rule parameters") - rule_param_margin_provider: Optional[Callable[[], - List[Dict[str, - float]]]] = None - # Context manager applying one rule-parameter override dict for the - # duration of a validation rollout: swap-and-restore of the live - # fitted-params mapping that the learned rules and frozen predicate - # classifiers read through (the score_atom_disagreement pattern). - # The gate enters it BEFORE the fresh-env scope so anything bound at - # env construction sees the override too. - rule_param_override_scope: Optional[Callable[[Dict[str, float]], - Any]] = None - # Capture-task keys (see ``_capture_task_key``) that have produced a - # FLAKY rejection in submit_plan. A flaky submission is direct - # evidence the agent is tuning in a marginal region where a lucky - # streak can pass the base rollout gate (run_20260717_182321: a - # 20/20-swept placement validated 3/3, then failed the real episode), - # so subsequent captures on these tasks must clear the escalated - # agent_plan_validation_rollouts_after_flaky gate instead. - flaky_capture_task_keys: Set[Any] = field(default_factory=set) - # Task-evaluator reward of the rollout that produced the current - # solved_plan capture (None when no evaluator verdict was computed). - # The restart loop ranks best-effort captures across attempts by it. - # Cleared together with solved_plan. - solved_plan_eval_reward: Optional[float] = None - # One-line record of the capture-time validation outcome (rollout - # tally, first failing step, physics-margin tally), for the journal - # auto-entry. Cleared together with solved_plan. - solved_plan_validation_summary: Optional[str] = None - # Closed-loop policy mode (CFG.agent_solve_policy_mode): the captured - # policy.py source, SNAPSHOTTED at submit_policy call time so a - # later edit of the file cannot swap unvalidated code into the - # executed artifact. Mutually exclusive with solved_plan; cleared - # together with it. - solved_policy_source: Optional[str] = None - # True while the current solve attempt's deliverable is a policy: - # submit_plan keeps its probing role but its CAPTURE gate is - # disabled, and submit_policy requires it. Set by _solve_attempt. - policy_capture_mode: bool = False - # Attempt bookkeeping, set around each attempt (a continual play - # round is one). ``attempt_start``/``attempt_deadline`` are - # time.monotonic() values; the deadline is enforced cooperatively by - # the probe (every sim call) and run_python, and surfaced in tool - # results as a budget footer. None ⇒ no attempt in flight / no wall - # clock. The deadline is cleared before the final-submission nudge so - # nothing blocks the submission itself. - attempt_index: int = 0 + # Round bookkeeping (a continual play round is one attempt): + # ``attempt_start`` is a time.monotonic() value, surfaced in tool + # results as the budget footer's elapsed time. None ⇒ no round in + # flight. attempt_start: Optional[float] = None - attempt_deadline: Optional[float] = None - # Count of full-plan belief-sim rollouts this attempt (probe runs, - # trials, capture-validation repeats). Reset per attempt; shown in - # the budget footer so sweeps carry a visible price. + # Count of full-plan belief-sim rollouts this round (probe runs, + # trials). Reset per round; shown in the budget footer so sweeps + # carry a visible price. attempt_rollout_count: int = 0 # The run's conversation as the play tools show it in the [context] # line (continual protocol): the prompt size of the latest assistant @@ -314,22 +209,14 @@ class ToolContext: context_turns: int = 0 context_compactions: int = 0 context_window_tokens: Optional[int] = None - # Best submission on the current task this attempt that - # submit_plan evaluated but refused to capture (evaluator - # scored it a non-solve, or it was flaky), ranked by evaluator - # reward. Reset per attempt; the journal auto-entry records it so a - # later attempt (or the final best-effort nudge) can resubmit it - # instead of the attempt's work vanishing with its context. - best_uncaptured_plan_lines: Optional[List[str]] = None - best_uncaptured_reward: Optional[float] = None # Per-call deadline for the run_python call currently executing - # (agent_sdk_python_call_timeout); enforced at the same - # probe checkpoints as attempt_deadline. None ⇒ no call in flight. + # (agent_sdk_python_call_timeout); enforced at the probe's + # checkpoints. None ⇒ no call in flight. python_call_deadline: Optional[float] = None # Adaptive info-seeking trigger (agent_explorer_info_seeking_adaptive): - # set True the first time submit_plan's rule-param margin gate refuses - # a plan as PARAM-SENSITIVE, cleared when a plan is captured. While - # True the proactive info-seeking apparatus (suggest_probes ranking, + # set True the first time the probe's physics sweep finds a plan's + # success straddling the belief interval. While True the proactive + # info-seeking apparatus (suggest_probes ranking, # disagreement guidance) is active; while False, and # under the adaptive flag, it stays dormant so easy levels pay no # info-seeking step tax. Ignored unless the adaptive flag is on. @@ -347,10 +234,11 @@ def info_seeking_active(self) -> bool: Off when info-seeking exploration is disabled outright. On whenever it is enabled and the adaptive flag is off (the original always-on behaviour). Under the adaptive flag it turns - on only once the capture gate has refused a plan as PARAM- - SENSITIVE this run (``param_sensitive_refusal_pending``), so the - agent spends real steps reducing uncertainty only after a - fragile plan has actually been caught. + on only once the probe's physics sweep has found a plan whose + success straddles the belief interval this run + (``param_sensitive_refusal_pending``), so the agent spends real + steps reducing uncertainty only after a fragile plan has + actually been found. """ if not CFG.agent_explorer_info_seeking: return False @@ -373,82 +261,26 @@ def note_stream_entry(self, entry: Dict[str, Any]) -> None: elif kind == "system" and entry.get("subtype") == "compact_boundary": self.context_compactions += 1 - def begin_attempt(self, index: int, wall_clock: float) -> None: - """Start restart-loop bookkeeping for solve attempt ``index``. - - Resets everything scoped to a single attempt (rollout count, - best refused submission) and arms the wall-clock deadline - (``wall_clock <= 0`` ⇒ no deadline). - """ - self.attempt_index = index + def begin_attempt(self) -> None: + """Start a round's bookkeeping: its clock and rollout count.""" self.attempt_rollout_count = 0 - self.best_uncaptured_plan_lines = None - self.best_uncaptured_reward = None self.attempt_start = time.monotonic() - self.attempt_deadline = (self.attempt_start + - wall_clock if wall_clock > 0 else None) def pause_attempt_clock(self, seconds: float) -> None: """Push every armed wall-clock mark ``seconds`` into the future. Called by the session manager after it slept out a usage limit, - so the wait is charged to neither the attempt's budget nor the - run_python call in flight, and the budget footer's elapsed time - stays honest. + so the wait is charged to neither the round nor the run_python + call in flight, and the budget footer's elapsed time stays + honest. """ if seconds <= 0: return if self.attempt_start is not None: self.attempt_start += seconds - if self.attempt_deadline is not None: - self.attempt_deadline += seconds if self.python_call_deadline is not None: self.python_call_deadline += seconds - def clear_plan_capture(self) -> None: - """Clear the four ``solved_plan*`` fields together. - - They form one value (see :class:`PlanCapture`); clearing any of - them individually would leave a stale mix. - """ - self.solved_plan = None - self.solved_sketch = None - self.solved_plan_reached_goal = None - self.solved_plan_eval_reward = None - self.solved_plan_validation_summary = None - self.solved_policy_source = None - - def take_plan_capture(self) -> PlanCapture: - """Pop the captured plan, clearing it so it cannot be reused. - - The returned capture's ``plan`` is falsy when nothing was - captured since the last clear. - """ - capture = PlanCapture( - plan=self.solved_plan, - sketch=self.solved_sketch, - reached_goal=self.solved_plan_reached_goal, - eval_reward=self.solved_plan_eval_reward, - validation_summary=self.solved_plan_validation_summary, - policy_source=self.solved_policy_source) - self.clear_plan_capture() - return capture - - -def _capture_task_key(ctx: ToolContext) -> Any: - """Stable identity of the task behind a ``task_idx="current"`` capture. - - Keys ``ctx.flaky_capture_task_keys`` so a FLAKY rejection escalates - the validation gate for later submissions on the SAME task only. - Test-time solves are keyed by the test task index (stable across the - sessions and replans of one task); exploration/synthesis captures - fall back to the learning iteration, which at worst escalates - conservatively across that cycle's tasks. - """ - if ctx.test_task_idx is not None: - return ("test", ctx.test_task_idx) - return ("iter", ctx.iteration_id) - @contextmanager def decorrelated_rollout_seed(rollout_idx: int) -> Iterator[None]: diff --git a/predicators/agent_sdk/tools/python_exec.py b/predicators/agent_sdk/tools/python_exec.py index d315a49b7c..aa79ccc7fd 100644 --- a/predicators/agent_sdk/tools/python_exec.py +++ b/predicators/agent_sdk/tools/python_exec.py @@ -151,15 +151,6 @@ async def python_exec(args: Dict[str, Any]) -> Dict[str, Any]: rollouts_before = 0 if budget_ctx is not None: rollouts_before = budget_ctx.attempt_rollout_count - attempt_dl = budget_ctx.attempt_deadline - if (attempt_dl is not None and time.monotonic() > attempt_dl - and not budget_ctx.capture_best_effort_plan): - return text_result( - "The attempt's wall-clock exploration budget is " - "exhausted - this call was not run. Submit your single " - "best plan NOW via submit_plan on the current " - "task (omit task_idx)." + - _budget_footer(budget_ctx, rollouts_before)) call_timeout = ToolSurfaceConfig.from_cfg().python_call_timeout if budget_ctx.probe_option_model_provider is not None: # Synthesis sessions probe the CANDIDATE simulator, whose @@ -184,9 +175,6 @@ def _footer() -> str: if budget_ctx is not None: if budget_ctx.python_call_deadline is not None: wd_deadlines.append(budget_ctx.python_call_deadline) - if (budget_ctx.attempt_deadline is not None - and not budget_ctx.capture_best_effort_plan): - wd_deadlines.append(budget_ctx.attempt_deadline) if call_timeout_s is not None and call_timeout_s > 0: wd_deadlines.append(time.monotonic() + call_timeout_s) if wd_deadlines: diff --git a/predicators/agent_sdk/tools/registry.py b/predicators/agent_sdk/tools/registry.py index b4ef0ff788..6932f3b75b 100644 --- a/predicators/agent_sdk/tools/registry.py +++ b/predicators/agent_sdk/tools/registry.py @@ -20,31 +20,22 @@ "TaskList", ] -TESTING_TOOL_NAMES = [ - "submit_plan", - # Closed-loop policy mode (agent_solve_policy_mode): validates and - # captures the agent-written policy.py. Only offered on solve - # rosters when the mode is on. - "submit_policy", -] -# The one code-execution tool. Solve sessions get the static instance +# The one code-execution tool. Play sessions get the static instance # built by ``create_mcp_tools`` (namespace = the BeliefProbe facade over # the deployed belief model, predicators/agent_sdk/belief_probe.py); # synthesis sessions attach their own instance under the same name # (fit data + the probe over the candidate simulator), which replaces -# the static one at assembly. Offered to every session that has a -# simulator to probe (see ``AgentModelFreeApproach._get_solve_tool_names``). +# the static one at assembly. EXPLORATION_TOOL_NAMES = [ "run_python", ] -ALL_TOOL_NAMES = TESTING_TOOL_NAMES + EXPLORATION_TOOL_NAMES +ALL_TOOL_NAMES = list(EXPLORATION_TOOL_NAMES) # Name of the tool ``create_synthesis_tools`` builds for a synthesis # session (the same ``run_python`` name as the solve-phase instance - # see EXPLORATION_TOOL_NAMES). ``tests/agent_sdk/test_tool_registry.py`` -# asserts that the factory output matches this tuple. Predicate and -# sampler drafts are loaded through the probe (``sim.predicates()`` / -# ``sim.samplers()``), not through tools. +# asserts that the factory output matches this tuple. Predicate drafts +# are loaded through the probe (``sim.predicates()``), not through tools. SYNTHESIS_TOOL_NAMES = ("run_python", ) diff --git a/predicators/agent_sdk/tools/synthesis.py b/predicators/agent_sdk/tools/synthesis.py index a4393f50ca..1cc700e973 100644 --- a/predicators/agent_sdk/tools/synthesis.py +++ b/predicators/agent_sdk/tools/synthesis.py @@ -588,9 +588,9 @@ def _evaluate_rollout_fit(rules: list, sse=post_sse, applied_physical=dict(applied), coverage=(outcome.num_survivors, len(rollouts)), - # Physics-margin points for the capture gate, restored - # when this fit is deployed as the cycle's model; the - # joint belief replaces them with its own draws. + # The physics sweep's +-1-sigma grid, restored when this + # fit is deployed as the cycle's model; under the joint + # belief the sweep reads the belief's interval ends. sigma_points=(physics_sigma_points( applied, ident_report, diff --git a/predicators/agent_sdk/tools/tasks.py b/predicators/agent_sdk/tools/tasks.py deleted file mode 100644 index 133b38f751..0000000000 --- a/predicators/agent_sdk/tools/tasks.py +++ /dev/null @@ -1,59 +0,0 @@ -"""Shared resolution of a tool's ``task_idx`` argument to a task. - -Every task-scoped tool follows the same convention: an int ``task_idx`` -indexes the train tasks (bounds-checked), and omitting it falls back to -the current solve/explore task. :func:`_resolve_task` is the single -implementation of that convention. -""" -from dataclasses import dataclass -from typing import Any, Dict, Optional, Tuple, Union - -from predicators.agent_sdk.tools.context import ToolContext -from predicators.agent_sdk.tools.results import _error_result -from predicators.structs import Task - - -@dataclass(frozen=True) -class ResolvedTask: - """A tool call's resolved task plus how to refer to it. - - ``label`` follows the tools' display convention: the int train-task - index, or the string ``"current"`` for the current solve/explore - task. Handlers interpolate it directly into report text and pass it - to ``_resolve_task_evaluator``; ``is_current`` is the boolean the - capture guards read (it replaces the old ``task_idx == "current"`` - comparisons). - """ - task: Task - label: Union[int, str] - is_current: bool - - @property - def description(self) -> str: - """Human-readable phrase: ``train task 3`` or ``current task``.""" - if self.is_current: - return "current task" - return f"train task {self.label}" - - -def _resolve_task( - ctx: ToolContext, task_idx: Optional[int] -) -> Tuple[Optional[ResolvedTask], Optional[Dict[str, Any]]]: - """Resolve a tool's ``task_idx`` argument (None ⇒ current task). - - Returns ``(resolved, error)`` with exactly one of the two set; - ``error`` is a ready-to-return tool error result. - """ - if task_idx is not None: - if task_idx < 0 or task_idx >= len(ctx.train_tasks): - return None, _error_result( - f"Invalid task_idx {task_idx}. " - f"Available: 0-{len(ctx.train_tasks)-1}") - return ResolvedTask(task=ctx.train_tasks[task_idx], - label=task_idx, - is_current=False), None - if ctx.current_task is not None: - return ResolvedTask(task=ctx.current_task, - label="current", - is_current=True), None - return None, _error_result("No task_idx provided and no current_task set.") diff --git a/predicators/agent_sdk/tools/testing.py b/predicators/agent_sdk/tools/testing.py deleted file mode 100644 index fec44e9e72..0000000000 --- a/predicators/agent_sdk/tools/testing.py +++ /dev/null @@ -1,1597 +0,0 @@ -"""Testing tools, including the submit_plan capture surface.""" -import contextlib -import functools -import logging -import os -from typing import Any, Callable, Dict, List, Optional, Sequence, Set, Tuple - -import numpy as np - -from predicators import utils -from predicators.agent_sdk import bilevel_sketch -from predicators.agent_sdk.config import RefinementConfig, ValidationConfig -from predicators.agent_sdk.parallel_rollouts import \ - prefetch_parallel as _prefetch_parallel -from predicators.agent_sdk.tools.budget import _budget_footer -from predicators.agent_sdk.tools.capture import BestEffortReason, \ - CaptureDecision, _decide_capture -from predicators.agent_sdk.tools.clearance import RobotClearanceProbe, \ - phase_skill_of -from predicators.agent_sdk.tools.context import ToolContext, \ - _capture_task_key, decorrelated_rollout_seed -from predicators.agent_sdk.tools.results import _error_result -from predicators.agent_sdk.tools.scene import format_object_poses, \ - render_pybullet_image -from predicators.agent_sdk.tools.tasks import _resolve_task -from predicators.agent_sdk.tools.verdicts import _EvalStateCollector, \ - _format_evaluator_verdict, _resolve_task_evaluator, _sandbox_base, \ - evaluate_states_with, load_ground_sampler_fns -from predicators.code_sim_learning.identifiability import straddle_summary -from predicators.settings import CFG -from predicators.structs import GroundAtom, State, Task - -# Ceiling on agent-requested validation rollouts per submission -# (validation_rollouts): the agent pays for rollouts from its budget, but -# a typo'd request should not silently torch it. -_MAX_REQUESTED_ROLLOUTS = 25 - - -def _missing_goal_atoms(task: Task, state: State) -> Set[GroundAtom]: - """Goal atoms that do NOT hold in ``state`` by their own classifiers. - - Evaluated per atom with the goal predicates' own classifiers (the - same ones ``goal_holds`` runs), never by abstracting the state with - the agent's predicate set: the env's goal predicates are not in - that set under predicate invention, so every goal atom then read - as missing whenever the goal was not reached - including the ones - that held - and one agent concluded the goal atoms could never be - made True in the belief and abandoned a working route - (2026-08-27 bridge policy seed 0, cycle 3). - """ - return {a for a in task.goal if not a.holds(state)} - - -def _policy_source_path(ctx: ToolContext) -> Optional[str]: - """Host path of the agent-editable ``policy.py`` (policy mode).""" - base = _sandbox_base(ctx) - if not base: - return None - return os.path.join(base, "policy.py") - - -def _parameter_margin_sweep( - ctx: ToolContext, validation_cfg: ValidationConfig, - fresh_scope: Callable[..., Any], rollout: Callable[[], Tuple[bool, - str]], - subject: str) -> Tuple[List[str], Optional[str], str]: - """Margin sweep over BOTH parameter-uncertainty sources of one. - - capture-eligible submission - the single code path behind the - physics-margin and rule-parameter gates of ``submit_plan`` - and ``submit_policy``. - - The execution repeats before this all run AT the fitted parameters, - so they cannot see a submission whose success band excludes the - fit's own error (run_20260723_091108: a capture validated 8/8 at - fitted lateral_friction 0.5319 failed deterministically at true - 0.5). Two sources express that error: - - * identified PHYSICAL params: the fit posterior's sigma grid, - applied as construction overrides on a fresh env (perturbing the - shared env would leak into later tool calls), at the BASE planner - seed so a failure is attributable to the perturbation alone; - * learned RULE params: the calibrated posterior ensemble (the same - members info-seeking exploration scores with), applied by - swapping the live fitted-params view - entered BEFORE the fresh - env so values bound at construction also see the member. - - ``rollout`` runs one validation rollout and returns ``(ok, why)``. - Returns ``(outcome lines, param-sensitive detail or None, suffix - for the validation note)``; any failing point sets the detail, - which rejects the submission as PARAM-SENSITIVE. - """ - outcomes: List[str] = [] - detail: Optional[str] = None - note = "" - if (validation_cfg.physics_margin - and ctx.physics_margin_provider is not None): - points = ctx.physics_margin_provider() or [] - - def _physics_rollout(point: Dict[str, float]) -> Tuple[bool, str]: - with fresh_scope(physical_overrides=point): - return rollout() - - prefetched = _prefetch_parallel( - [functools.partial(_physics_rollout, point) for point in points], - f"{subject} physics margin") - passed: List[bool] = [] - for point_idx, point in enumerate(points): - ctx.attempt_rollout_count += 1 - pre = prefetched[point_idx] - ok, why = pre if pre is not None else _physics_rollout(point) - passed.append(bool(ok)) - desc = ", ".join(f"{k}={v:.4g}" for k, v in sorted(point.items())) - if ok: - outcomes.append(f"physics point ({desc}): goal reached") - else: - outcomes.append(f"physics point ({desc}): FAILED - {why}") - if detail is None: - detail = f"at {desc}: {why}" - if (detail is not None and any(passed) - and CFG.code_sim_learning_interval_belief): - # Certification over the belief interval (interval belief): - # a mixed sweep is the interval straddling the plan's success - # boundary, and the passing/failing ranges say which way. - straddle = straddle_summary(points, passed) - detail += (f" ({sum(passed)}/{len(passed)} belief-interval " - f"points passed" + - (f"; {straddle}" if straddle else "") + ")") - if points and detail is None: - note += ( - f" Physics-margin check passed: the {subject} also reached " - f"the goal at all {len(points)} grid points spanning +-1 " - "sigma of the identified physical parameters.") - if (validation_cfg.rule_param_margin and detail is None - and ctx.rule_param_margin_provider is not None - and ctx.rule_param_override_scope is not None): - rule_points = ctx.rule_param_margin_provider() or [] - override_scope = ctx.rule_param_override_scope - - def _member_rollout(point: Dict[str, float]) -> Tuple[bool, str]: - assert override_scope is not None - with override_scope(point), fresh_scope(): - return rollout() - - # Prefetching runs every member even though the sequential loop - # below still breaks at the first failure - the extra results - # are discarded, keeping the report identical with the flag on - # or off (failures are rare enough that the prepaid tail is - # cheaper than serializing the common all-pass case). - member_prefetched = _prefetch_parallel( - [functools.partial(_member_rollout, pt) for pt in rule_points], - f"{subject} rule-param margin") - for member_idx, point in enumerate(rule_points): - ctx.attempt_rollout_count += 1 - pre = member_prefetched[member_idx] - ok, why = pre if pre is not None else _member_rollout(point) - desc = (f"{ctx.rule_param_margin_label} " - f"{member_idx + 1}/{len(rule_points)}") - if ok: - outcomes.append(f"{desc}: goal reached") - else: - shown = ", ".join(f"{k}={_fmt_point_value(v)}" - for k, v in sorted(point.items())[:8]) - if len(point) > 8: - shown += ", ..." - outcomes.append(f"{desc}: FAILED - {why}") - detail = f"under {desc} ({shown}): {why}" - break - if rule_points and detail is None: - note += ( - f" Rule-parameter margin check passed: the {subject} also " - f"reached the goal under all {len(rule_points)} " - f"{ctx.rule_param_margin_note}.") - return outcomes, detail, note - - -def _necessity_sweep( - ctx: ToolContext, fresh_scope: Callable[..., Any], - rollout_without: Callable[[int], Tuple[bool, str]], - step_names: Sequence[str]) -> Tuple[List[str], Optional[str], str]: - """Necessity gate: refuse a plan that still reaches the goal with one of - its steps removed. - - ``rollout_without(k)`` runs the plan with step ``k`` deleted and - returns ``(goal reached, why not)``. A step whose removal leaves the - goal reached is padding: it explains nothing about how the goal - comes about, spends real episode steps, and if the model is wrong - about it can break the plan for real (run_20260902_152811: a - validated capture pressed three of four buttons and released one - that was never on, for a goal its own model reached with two - presses and a Wait). - - Returns ``(outcome lines, redundant detail or None, suffix for the - validation note)``. A one-step plan has nothing to ablate. - """ - outcomes: List[str] = [] - detail: Optional[str] = None - if len(step_names) < 2: - return outcomes, detail, "" - - def _ablated(k: int) -> Tuple[bool, str]: - with fresh_scope(): - return rollout_without(k) - - prefetched = _prefetch_parallel( - [functools.partial(_ablated, k) for k in range(len(step_names))], - "plan necessity") - for k, name in enumerate(step_names): - ctx.attempt_rollout_count += 1 - pre = prefetched[k] - ok, why = pre if pre is not None else _ablated(k) - if ok: - outcomes.append(f"without step {k} ({name}): goal STILL reached") - if detail is None: - detail = f"step {k} ({name})" - else: - outcomes.append(f"without step {k} ({name}): {why}") - note = "" - if detail is None: - note = (f" Necessity check passed: removing any one of the " - f"{len(step_names)} steps loses the goal.") - return outcomes, detail, note - - -def _fmt_point_value(value: Any) -> str: - """A margin point's value for a report: numbers compactly, anything else (a - belief particle's nested latent) by its repr, truncated.""" - if isinstance(value, (int, float, np.floating, np.integer)): - return f"{float(value):.4g}" - text = repr(value) - return text if len(text) <= 40 else text[:37] + "..." - - -def _build_testing_tools(ctx: ToolContext, _text_result: Callable, - tool: Callable) -> Dict[str, Any]: - """Evaluation tools (option plans / policies against tasks).""" - - # Tool descriptions bake config values at BUILD time (session open); - # the handlers below re-read config at CALL time. - _gs_eval_doc = ( - "Runs your exact params with NO sampling (a `~` ground-sampler " - "annotation - `~ [w1, w2]` region or `~ my_sampler` - is accepted " - "but IGNORED here; only `sim.refine` uses it). " - if RefinementConfig.from_cfg().ground_samplers else - "Runs your exact params with NO sampling. ") - - @tool( - "submit_plan", - "SUBMIT a fully-specified plan as your answer for the CURRENT task. " - "`plan` is text - one option per line, same grammar as `sim.run` / " - "`sim.refine`: `Option(obj1:type1, obj2:type2)[param1, param2] -> " - "{Atom(obj:type), ...}` (typed object refs; EXACT continuous params " - "in `[]`, `[]` for none; optional `-> {atoms}` subgoals, prefix NOT " - "to require false). " + _gs_eval_doc + - "The plan is rolled out from the task's TRUE initial state through " - "the belief model and reported step by step (include_states/" - "include_atoms control the report). If it reaches the goal it is " - "captured as your answer, and the per-step subgoals make it execute " - "closed-loop (monitored, with replan-on-divergence). Capture is " - "gated: a goal-reaching plan is re-run several times (simulation " - "varies across runs; each rollout reports the motion-planner seed " - "it ran at) and a FLAKY plan is reported instead of captured. The " - "gate's rollout set is exactly what `sim.run(plan, trials=N)` runs " - "(fresh env per rollout, same planner seeds), so measure " - "reliability there BEFORE submitting, and reproduce one failed " - "rollout with `sim.run(plan, seed=S, fresh=True)`; then add " - "margin and resubmit. `validation_rollouts` " - "requests a STRICTER gate for this submission (more rollouts; never " - "fewer than configured). This is the ONLY path that captures an " - "answer: explore (other tasks, modified states, partial plans, " - "parameter sweeps, seeded reproductions) with `sim` in run_python, " - "then submit the final plan here. " - "When identified physical parameters are active, it is also re-run " - "at a grid of perturbations spanning +-1 sigma of those parameters " - "(the physics fit's own uncertainty); a PARAM-SENSITIVE plan is " - "reported instead of captured - add design margin so it succeeds " - "across the whole range. " - "When the necessity gate is on, it is also re-run once per step " - "with that step removed; a plan that still reaches the goal without " - "one of its steps is reported REDUNDANT naming the step, not " - "captured - submit the shortest plan your model needs. " - "When the task has an evaluator, a goal-reaching plan the evaluator " - "still scores as a non-solve (no success credit in its reward) is " - "NOT captured (the real env applies the same scoring, so it could " - "never count as a solve).", - { - "type": "object", - "properties": { - "plan": { - "type": - "string", - "description": - "Plan text, one option per line: " - "`Option(obj1:type1, obj2:type2)[p1, p2] -> " - "{Atom(obj:type), ...}` (exact params in `[]`; `[]` for " - "none; optional `-> {atoms}` subgoals, NOT-prefix to " - "require false).", - }, - "include_states": { - "type": - "boolean", - "description": - "Include the full low-level state feature dict after each " - "step", - "default": - True - }, - "include_atoms": { - "type": "boolean", - "description": - "Include atoms added/deleted after each step", - "default": True - }, - "validation_rollouts": { - "type": - "integer", - "description": - "Request a stricter capture gate: total validation " - "rollouts a goal-reaching submission must pass. The " - "effective count is max(configured gate, this) - it can " - "raise the gate but never lower it. Use before " - "committing a plan you suspect is marginal.", - }, - }, - "required": ["plan"], - }, - ) - async def submit_plan(args: Dict[str, Any]) -> Dict[str, Any]: - refine_cfg = RefinementConfig.from_cfg() - validation_cfg = ValidationConfig.from_cfg() - ctx.test_call_id += 1 - # Snapshot for the [budget] footer's per-call delta; this handler - # increments the counter itself (initial rollout + validation - # repeats), and without the snapshot the footer reports the - # attempt's cumulative total as "+N this call". - rollouts_before = ctx.attempt_rollout_count - - if ctx.option_model is None: - return _error_result("No option model available in ToolContext.") - - all_options = ctx.options - opt_map = {o.name: o for o in all_options} - model = ctx.option_model - model._name_to_parameterized_option = ( # type: ignore[attr-defined] # pylint: disable=protected-access - opt_map) - - plan_text = (args.get("plan") or "").strip() - include_states = args.get("include_states", False) - include_atoms = args.get("include_atoms", True) - requested_rollouts = args.get("validation_rollouts") - if requested_rollouts is not None and (not isinstance( - requested_rollouts, int) or requested_rollouts < 1): - return _error_result( - "validation_rollouts must be a positive integer.") - - # Always the CURRENT task from its true initial state: this is - # the submission path, and exploration on other tasks or from - # modified states lives on the probe (sim.run). - resolved, task_err = _resolve_task(ctx, None) - if task_err is not None: - return task_err - assert resolved is not None - task = resolved.task - task_label = resolved.label - - lines = [f"Testing option plan on task {task_label}:"] - saved_image_paths: List[str] = [] - - all_predicates = ctx.predicates - - if not plan_text: - return _error_result("`plan` is required (option plan text).") - # Parse the text plan into a sketch (options + objects + exact params + - # subgoals) using the SAME grammar/parser as sim.refine. - types = set(ctx.types) - for opt in all_options: - types.update(opt.types) - for pred in all_predicates: - types.update(pred.types) - types.update(o.type for o in task.init) - try: - # strict: the `plan` argument is pure plan text, so a line that - # fails to parse is an error the agent must see - silently - # dropping it (the freeform default) executes a different plan - # than the agent asked for. - gs_fns, gs_err = load_ground_sampler_fns(ctx) - if gs_err is not None: - return _error_result(gs_err) - parse_notices: List[str] = [] - sketch_steps = bilevel_sketch.parse_sketch_from_text( - plan_text, - task, - predicates=all_predicates, - options=all_options, - types=types, - parse_continuous_params=True, - strict=True, - parse_ground_samplers=refine_cfg.ground_samplers, - ground_sampler_fns=gs_fns or None, - notices=parse_notices) - except Exception as e: # pylint: disable=broad-except - return _error_result(f"Could not parse plan: {e}") - lines.extend(f"NOTE: {n}" for n in parse_notices) - if not sketch_steps: - return _error_result( - "Parsed empty plan. Each line must be " - "`Option(obj:type, ...)[params] -> {subgoals}` with a known " - "option, typed object refs, and exact params in `[]`.") - # Ground each step with its parsed exact params, via the same - # helper the refine path uses: an annotated Wait gets its - # wait_target_atoms installed, so it waits for the annotated - # atoms here exactly as in refine and in real execution - - # grounding directly made the same Wait terminate on the first - # incidental atom change in this rollout but wait for its - # targets in refine, two different durations for one plan. - grounded_plan: List[Any] = [] - for step_idx, st in enumerate(sketch_steps): - params = (st.initial_params if st.initial_params is not None else - np.array([], dtype=np.float32)) - try: - grounded_plan.append( - bilevel_sketch.ground_step( - st, np.asarray(params, dtype=np.float32))) - except Exception as e: # pylint: disable=broad-except - return _error_result(f"Failed to ground step {step_idx} " - f"({st.option.name}): {e}") - - # Per-low-level-step states + option labels for the task-evaluator - # verdict below (see _EvalStateCollector for why per-step states). - eval_collector = _EvalStateCollector(model, task.init) - - # Robot-clearance probe (see tools/clearance.py): rollout 1 and - # every validation repeat feed it their low-level trajectories, - # and its verdict joins the margin gates below. None when the - # plan's skills carry no planning simulator to measure on. - clearance_probe: Optional[RobotClearanceProbe] = None - probe_skill = phase_skill_of(grounded_plan) - if probe_skill is not None: - clearance_probe = RobotClearanceProbe(probe_skill) - rollout_counter = [1] - - def _probe_clearance(label: str, outcome: Any) -> None: - if clearance_probe is None: - return - traj = getattr(model, "last_trajectory", None) - states = getattr(traj, "states", None) - if states: - clearance_probe.observe(label, outcome.option, states) - - # Per-step report callback, driven by the shared forward executor. - def _report_step(i: int, outcome: Any) -> None: - eval_collector.collect(outcome) - _probe_clearance("rollout 1", outcome) - opt = outcome.option - sig = f"{opt.name}({[o.name for o in opt.objects]})" - if not outcome.initiable: - atoms = utils.abstract(outcome.pre_state, ctx.predicates) - atoms_str = ", ".join(str(a) for a in sorted(atoms)) - lines.append(f"Step {i}: {sig} - NOT INITIABLE\n" - f" Current atoms: {{{atoms_str}}}\n" - f" Object poses at failure:\n" - f"{format_object_poses(outcome.pre_state)}") - return - step_line = f"Step {i}: {sig} ({outcome.num_actions} actions)" - if (opt.name == "Wait" and outcome.failure_reason is None - and outcome.num_actions >= utils.wait_rollout_step_cap()): - step_line += ( - "\n NOTE: this Wait ran to its step cap - " - "its wait-target atoms never became true in the " - "belief (and no other atom changed). Check whether " - "the awaited change is modeled, or drop the Wait.") - if outcome.failure_reason is not None: - step_line += (f"\n FAILURE REASON: {outcome.failure_reason}" - "\n Object poses at failure:\n" - f"{format_object_poses(outcome.pre_state)}") - post = outcome.post_state - if post is not None and include_atoms: - before = utils.abstract(outcome.pre_state, ctx.predicates) - after = utils.abstract(post, ctx.predicates) - added_s = ", ".join(str(a) for a in sorted(after - before)) - del_s = ", ".join(str(a) for a in sorted(before - after)) - step_line += (f"\n Added: {{{added_s}}}" - f"\n Deleted: {{{del_s}}}") - if post is not None and include_states: - step_line += ("\n State:\n" + - post.dict_str(indent=4, num_decimal_points=4)) - lines.append(step_line) - # Render from the outcome's state explicitly: the rollout - # runs on the gate's fresh env (see fresh_scope below) while - # the renderer draws the shared session env, so without the - # state the images would show a stale scene. - img_block = render_pybullet_image( - ctx, - f"step_{i}_{opt.name}", - state=post if post is not None else outcome.pre_state) - if img_block and img_block.get("saved_path"): - saved_image_paths.append(img_block["saved_path"]) - - # One substrate for the WHOLE gate: rollout 1 (the capture - # rollout) runs on the same freshly constructed env as the - # validation repeats, at the base planner seed. This makes the - # gate reproducible from inside the session - sim.run(plan, - # trials=N) runs the identical rollout set (fresh env per trial, - # planner seeds base..base+N-1) - and stops a submission from - # advancing the shared session env. Rollout 1 on the warm shared - # env was a different physics substrate from every repeat: the - # 2026-08-30 bridge runs tuned plans to 27/27 on one substrate - # that then scored 1/10 on the other, with no way to reproduce - # the gate's rollouts. - fresh_scope = (ctx.validation_env_scope - if validation_cfg.fresh_env else None) - # Execute exactly like the real closed-loop executor: abort at the - # first failing option (0-action collision / not-initiable / env - # failure) instead of pressing on. Otherwise forward simulation can - # continue past a collision and report a goal that the real rollout — - # which ends the episode at that failed option — never reaches. - ctx.attempt_rollout_count += 1 - with (fresh_scope() - if fresh_scope is not None else contextlib.nullcontext()): - result = bilevel_sketch.execute_plan_forward( - task, - grounded_plan, - ctx.option_model, - predicates=all_predicates, - sketch=sketch_steps, - on_step=_report_step, - stop_on_failure=True) - final_atoms = utils.abstract(result.final_state, ctx.predicates) - # Task-evaluator verdict on this belief-sim rollout, computed - # BEFORE capture and INSIDE the scope (certificate probes must - # judge on the env the rollout ran on): the real evaluator - # applies the same certificate, so a goal-reaching but - # illegitimate plan can never count as a solve and must not be - # captured as the answer (run_20260712_173955 tasks 1-2: - # flagged-illegitimate captures stood all session and were - # executed only to be rejected). Failure-tolerant: verdict - # stays None when the task has no evaluator or nothing - # executed. A coarse verdict (option-boundary states only) can - # falsely reject a legitimate cascade, so it never blocks - # capture. - evaluator = _resolve_task_evaluator(ctx, task_label) - verdict: Optional[Dict[str, Any]] = None - if evaluator is not None and len(eval_collector.states) > 1: - try: - verdict = evaluate_states_with(evaluator, - eval_collector.states, - eval_collector.labels, - sim_env=getattr( - ctx.option_model, - "sim_env", None)) - except Exception as e: # pylint: disable=broad-except - logging.debug("Task-evaluator verdict failed: %s", e) - # Use the env's goal-check (its own classifiers); robust to invented - # predicates that don't reuse env names. - goal_reached = result.goal_reached - # One more real-executor constraint the option model doesn't enforce: - # the episode is capped at the phase's step budget (the horizon, or - # the interaction-request cap for explore episodes). A plan whose - # goal is reached only after more steps than that will time out in - # real rollout, so don't count it as achieved/captured. - horizon = ctx.execution_step_budget() - within_horizon = (result.actions_to_goal is not None - and result.actions_to_goal <= horizon) - goal_achieved = (goal_reached and result.clean_to_goal - and within_horizon) - evaluator_rejected = (verdict is not None and not verdict["legitimate"] - and not eval_collector.coarse) - # An evaluator rejection only disqualifies a capture when the goal - # atoms actually hold via an illegitimate route - a reward hack (e.g. - # the agent knocked the target over directly). An honest shortfall, - # where the rollout simply fails to reach the goal, is ALSO - # legitimate=False (there is no genuine cascade to certify), but that - # is exactly what a best-effort submission is meant to capture, so it - # must not be conflated with a reward hack. - reward_hack = (evaluator_rejected and verdict is not None - and verdict["terminated"]) - - # Multi-rollout validation of a capture candidate. The shared sim - # env is nondeterministic across repeats (motion-planner sampling, - # physics-solver state), which is the same variability the real - # rollout will sample - a plan that only sometimes succeeds here is - # a margin-free plan that will likely fail on the real env - # (run_20260712_192457 task 1: a sim-validated 2-hop relay died on a - # ~9mm placement drift). So a goal-reaching plan is captured only - # after every one of validation_cfg.rollouts total - # rollouts succeeds; a flaky repeat is reported to the agent, who - # still has the session to add margin and resubmit. - def _validation_rollout() -> Tuple[bool, str, List[Optional[State]]]: - """One extra rollout of the exact plan. - - Returns ``(ok, failure detail, per-step post-states)``; the - post-state list is padded with ``None`` to the plan length - so a truncated (failed) rollout still indexes safely. - Passing rollouts' post-states feed the captured-annotation - intersection filter. - """ - v_collector = _EvalStateCollector(model, task.init) - rollout_counter[0] += 1 - rollout_label = f"rollout {rollout_counter[0]}" - - def _on_validation_step(i: int, outcome: Any) -> None: - v_collector.on_step(i, outcome) - _probe_clearance(rollout_label, outcome) - - r = bilevel_sketch.execute_plan_forward( - task, - grounded_plan, - model, - predicates=all_predicates, - sketch=sketch_steps, - on_step=_on_validation_step, - stop_on_failure=True) - posts: List[Optional[State]] = [s.post_state for s in r.steps] - posts += [None] * (len(grounded_plan) - len(posts)) - if r.first_failure_idx is not None: - fr = r.steps[r.first_failure_idx].failure_reason - opt = r.steps[r.first_failure_idx].option - return False, (f"step {r.first_failure_idx} " - f"({opt.name}) failed: {fr}"), posts - if not r.goal_reached: - missing = _missing_goal_atoms(task, r.final_state) - missing_str = ", ".join(str(a) for a in sorted(missing)) - detail = f" (missing: {{{missing_str}}})" if missing else "" - return False, f"goal not reached{detail}", posts - if not (r.actions_to_goal is not None - and r.actions_to_goal <= horizon): - return False, (f"goal reached only after " - f"{r.actions_to_goal} low-level steps, past " - f"the episode horizon ({horizon})"), posts - # Same legitimacy rule as the first rollout: a non-coarse - # illegitimate verdict fails the validation. - if (evaluator is not None and len(v_collector.states) > 1 - and not v_collector.coarse): - try: - v = evaluate_states_with(evaluator, - v_collector.states, - v_collector.labels, - sim_env=getattr( - ctx.option_model, "sim_env", - None)) - if not v["legitimate"]: - return False, ( - "this rollout reached the goal atoms but the " - "task evaluator scored it as a non-solve " - f"(solved=False, reward={v['reward']:.2f})"), posts - except Exception as e: # pylint: disable=broad-except - logging.debug("Validation-rollout verdict failed: %s", e) - return True, "", posts - - flaky_detail: Optional[str] = None - validation_note = "" - n_rollouts = max(1, validation_cfg.rollouts) - # Escalated gate once this task has produced a FLAKY rejection: the - # agent is provably tuning in a marginal region, where a lucky - # streak passes the base gate and dies on the single real episode - # (run_20260717_182321: a 20/20-swept relay placement validated 3/3, - # then missed the target for real). - capture_task_key = _capture_task_key(ctx) - if capture_task_key in ctx.flaky_capture_task_keys: - n_rollouts = max(n_rollouts, validation_cfg.rollouts_after_flaky) - # The agent may request a STRICTER gate for this submission (a - # plan it suspects is marginal); it can never lower the - # configured gate - that would let a lucky draw bypass it. - capped_request: Optional[int] = None - if requested_rollouts is not None: - capped_request = min(requested_rollouts, _MAX_REQUESTED_ROLLOUTS) - if capped_request < requested_rollouts: - lines.append( - f"NOTE: validation_rollouts={requested_rollouts} capped " - f"at {_MAX_REQUESTED_ROLLOUTS}.") - n_rollouts = max(n_rollouts, capped_request) - # fresh_scope (computed above, shared with rollout 1): repeats on - # the shared env are correlated (its reset cannot reconstruct - # state exactly), so only fresh envs sample the same distribution - # the real episode will. - rollout_outcomes: List[str] = [] - # Per-step post-states of PASSING validation rollouts, for the - # captured-annotation intersection filter. Failing rollouts are - # excluded on purpose: they are off-track by definition, so their - # post-states are not evidence about what holds on a successful - # execution (using them would prune annotations that hold in - # every on-track run). Physics-margin rollouts are likewise - # excluded: they run under deliberately perturbed physics. - passing_validation_posts: List[List[Optional[State]]] = [] - base_planner_seed = CFG.seed - if (ctx.capture_goal_reaching_plans and goal_achieved - and not evaluator_rejected and grounded_plan - and n_rollouts > 1): - # Run ALL validation rollouts even after a failure: the - # per-rollout outcome list distinguishes failure modes (a - # physics-tail fizzle vs. an IK stall vs. a certificate - # rejection) and yields a reliability estimate - a bare - # "rollout k FAILED" left agents guessing which - # (run_20260717_182040 seed0 turn 214). - # decorrelated_rollout_seed: a fresh env alone gives - # bit-identical repeats (motion planning reads the - # constant CFG.seed at call time), so without it the - # validation repeats re-run the capture rollout verbatim - # and detect nothing. The capture rollout itself keeps - # the base seed; repeats sample execution variability. - def _repeat_rollout( - repeat_idx: int - ) -> Tuple[bool, str, List[Optional[State]]]: - with (fresh_scope() if fresh_scope is not None else - contextlib.nullcontext()), \ - decorrelated_rollout_seed(repeat_idx - 1): - return _validation_rollout() - - repeat_indices = list(range(2, n_rollouts + 1)) - repeat_prefetched = _prefetch_parallel([ - functools.partial(_repeat_rollout, k) for k in repeat_indices - ], "capture repeat rollouts") - for pos, repeat_idx in enumerate(repeat_indices): - ctx.attempt_rollout_count += 1 - pre = repeat_prefetched[pos] - ok, why, repeat_posts = (pre if pre is not None else - _repeat_rollout(repeat_idx)) - repeat_seed = base_planner_seed + repeat_idx - 1 - if ok: - passing_validation_posts.append(repeat_posts) - rollout_outcomes.append( - f"rollout {repeat_idx} (planner seed " - f"{repeat_seed}): goal reached") - else: - rollout_outcomes.append( - f"rollout {repeat_idx} (planner seed " - f"{repeat_seed}): FAILED - {why}") - if flaky_detail is None: - flaky_detail = (f"rollout {repeat_idx}/{n_rollouts} " - f"(planner seed {repeat_seed}) " - f"FAILED: {why}") - if flaky_detail is None: - fresh_note = (", each on a freshly constructed simulator " - "instance" if fresh_scope is not None else "") - validation_note = ( - f" Validated {n_rollouts}/{n_rollouts} rollouts " - f"(planner seeds {base_planner_seed}-" - f"{base_planner_seed + n_rollouts - 1}; the " - "simulator's motion planning and physics stepping vary " - "across runs; repeats sample that execution " - f"variability{fresh_note}; sim.run(plan, " - f"trials={n_rollouts}) reruns this exact rollout set).") - - # Parameter-margin gates (see _parameter_margin_sweep): the - # execution repeats above all run AT the fitted parameters, so - # they cannot see a plan whose success band excludes the fit's - # own error - in the identified physical params or the learned - # rule constants. - param_sensitive_detail: Optional[str] = None - margin_outcomes: List[str] = [] - if (fresh_scope is not None and ctx.capture_goal_reaching_plans - and goal_achieved and not evaluator_rejected and grounded_plan - and flaky_detail is None): - margin_outcomes, param_sensitive_detail, margin_note = \ - _parameter_margin_sweep( - ctx, validation_cfg, fresh_scope, - lambda: _validation_rollout()[:2], "plan") - validation_note += margin_note - - # Clearance gate (see tools/clearance.py): the rollouts above - # certify the plan against the belief's own execution - # variability, not against the real executor's realization slop; - # a robot link passing inside that slop of a bystander is a - # margin-free plan whether or not every rollout cleared it. - clearance_detail: Optional[str] = None - clearance_summary = "" - if (clearance_probe is not None and goal_achieved - and not evaluator_rejected and grounded_plan - and flaky_detail is None): - clearance_ok, clearance_summary, clearance_why = \ - clearance_probe.verdict() - if not clearance_ok: - clearance_detail = clearance_why - any_margin_detail = param_sensitive_detail or clearance_detail - - # Necessity gate (see _necessity_sweep): only a plan that cleared - # every gate above is worth ablating, and only a plan with more - # than one step can be. - redundant_detail: Optional[str] = None - necessity_outcomes: List[str] = [] - if (validation_cfg.necessity and fresh_scope is not None - and ctx.capture_goal_reaching_plans and goal_achieved - and not evaluator_rejected and grounded_plan - and flaky_detail is None and any_margin_detail is None): - - def _rollout_without(k: int) -> Tuple[bool, str]: - ablated_plan = grounded_plan[:k] + grounded_plan[k + 1:] - ablated_sketch = (sketch_steps[:k] + sketch_steps[k + 1:] - if sketch_steps is not None else None) - r = bilevel_sketch.execute_plan_forward( - task, - ablated_plan, - model, - predicates=all_predicates, - sketch=ablated_sketch, - stop_on_failure=True) - if r.first_failure_idx is not None: - fr = r.steps[r.first_failure_idx].failure_reason - return False, (f"step {r.first_failure_idx} of the " - f"shortened plan failed: {fr}") - if not r.goal_reached: - missing = _missing_goal_atoms(task, r.final_state) - missing_str = ", ".join(str(a) for a in sorted(missing)) - return False, ("goal not reached " - f"(missing: {{{missing_str}}})") - return True, "" - - necessity_outcomes, redundant_detail, necessity_note = \ - _necessity_sweep( - ctx, fresh_scope, _rollout_without, - [f"{g.name}({', '.join(o.name for o in g.objects)})" - for g in grounded_plan]) - validation_note += necessity_note - - def _stash_uncaptured_submission() -> None: - """Remember the best refused submission of this attempt. - - The journal auto-entry records it at attempt end, so the - plan (and its honest evaluator reward) survives the fresh- - context restart and the final best-effort nudge can resubmit - it instead of the attempt's work vanishing with its context. - """ - reward = float(verdict["reward"]) if verdict is not None else None - prev = ctx.best_uncaptured_reward - if ctx.best_uncaptured_plan_lines is not None and ( - reward is None or (prev is not None and reward <= prev)): - return - ctx.best_uncaptured_reward = reward - ctx.best_uncaptured_plan_lines = list( - bilevel_sketch.format_plan_lines(grounded_plan)) - - # The capture decision itself is pure (see _decide_capture, which - # also documents the best-effort-mode semantics); the branches - # below apply its ctx mutations and format its messages. - capture_outcome = _decide_capture( - # In policy mode the deliverable is policy.py (via - # submit_policy); this tool remains a probe but can no - # longer capture the answer. - capture_enabled=(ctx.capture_goal_reaching_plans - and not ctx.policy_capture_mode), - is_current_task=True, - have_plan=bool(grounded_plan), - goal_achieved=goal_achieved, - evaluator_rejected=evaluator_rejected, - reward_hack=reward_hack, - flaky=flaky_detail is not None, - best_effort_mode=ctx.capture_best_effort_plan, - have_validated_capture=bool(ctx.solved_plan_reached_goal), - param_sensitive=any_margin_detail is not None, - redundant=redundant_detail is not None) - decision = capture_outcome.decision - captured = capture_outcome.captured - # Adaptive info-seeking trigger (ctx.info_seeking_active): a plan - # refused because a learned/physical parameter's uncertainty - # threatens it is the signal that the agent now needs to spend - # real steps reducing that uncertainty; a captured plan clears it. - # Clearance (bystander slop) is a separate concern and does not - # arm info-seeking, so key on the parameter-margin detail only. - if captured: - ctx.param_sensitive_refusal_pending = False - elif param_sensitive_detail is not None: - ctx.param_sensitive_refusal_pending = True - if captured: - # Capture the plan with a sketch that keeps only the subgoals - # that actually held (so the closed-loop monitor won't flag a - # spurious divergence on a wrong annotation). An annotation - # must hold in rollout 1 AND in every PASSING validation - # rollout: an atom that held once by luck under the sim's own - # nondeterminism would otherwise survive into the executed - # sketch and kill the real episode on a spurious divergence. - # (With zero passing repeats this reduces to the rollout-1 - # filter.) - validated_solve = decision is CaptureDecision.VALIDATED_CAPTURE - captured_sketch = [] - - def _held_in_passing_repeats(atom: Any, i: int, - want_held: bool) -> bool: - for posts in passing_validation_posts: - post_i = posts[i] if i < len(posts) else None - # bool(): classifiers may return numpy bools, which - # fail identity checks against Python bools. - if post_i is None or bool(atom.holds(post_i)) != want_held: - return False - return True - - # Execution-verifiability probe. The closed-loop monitor - # evaluates the captured annotations on REAL observations, - # which carry no latent (``State.latent`` is None outside - # belief rollouts). An atom whose truth in the certifying - # post-state depends on the belief latent therefore reads - # false at execution no matter what physically happens, and - # a single such annotation aborts a healthy episode (a - # latent-only SeamBonded killed two runs whose bonds had in - # fact formed). Certify each surviving positive atom on the - # same post-state with the latent stripped and drop the - # ones that fail; a classifier that RAISES without a latent - # would crash the monitor, so it is dropped from either - # polarity the same way (a stripped negative atom that - # merely evaluates is kept - it cannot fire spuriously). - unverifiable_dropped: List[str] = [] - - def _probe_without_latent(atom: Any, - post: State) -> Optional[bool]: - stripped = State(post.data) - try: - return bool(atom.holds(stripped)) - except Exception: # pylint: disable=broad-except - return None - - for i, st in enumerate(sketch_steps): - post = (result.steps[i].post_state - if i < len(result.steps) else None) - if post is not None: - after = utils.abstract(post, all_predicates) - pos_held = { - a - for a in (st.subgoal_atoms or set()) - if a in after and _held_in_passing_repeats(a, i, True) - } - neg_held = { - a - for a in (st.subgoal_neg_atoms or set()) - if a not in after - and _held_in_passing_repeats(a, i, False) - } - pos_drop = { - a - for a in pos_held - if _probe_without_latent(a, post) is not True - } - neg_drop = { - a - for a in neg_held - if _probe_without_latent(a, post) is None - } - unverifiable_dropped.extend( - f"step {i} ({st.option.name}): {a}" - for a in sorted(pos_drop | neg_drop, key=str)) - pos_held -= pos_drop - neg_held -= neg_drop - else: - pos_held, neg_held = set(), set() - captured_sketch.append( - bilevel_sketch.SketchStep(option=st.option, - objects=st.objects, - subgoal_atoms=pos_held or None, - subgoal_neg_atoms=neg_held - or None)) - # Re-align each Wait's target atoms with the FILTERED - # sketch: the real executor waits on exactly the monitored - # (execution-verifiable) atoms. A latent-only target (e.g. a - # belief Bonded) reads false on every real observation, so - # leaving it in the grounded option's memory would stall the - # real Wait to its step-cap backstop no matter what happens. - # With every target filtered away the Wait falls back to - # any-atom-change, the same rule the belief rollout then - # shares. - for g_opt, cap_step in zip(grounded_plan, captured_sketch): - if g_opt.name != "Wait": - continue - g_opt.memory.pop("wait_target_atoms", None) - g_opt.memory.pop("wait_target_neg_atoms", None) - if cap_step.subgoal_atoms: - g_opt.memory["wait_target_atoms"] = cap_step.subgoal_atoms - if cap_step.subgoal_neg_atoms: - g_opt.memory["wait_target_neg_atoms"] = \ - cap_step.subgoal_neg_atoms - ctx.solved_plan = grounded_plan - ctx.solved_sketch = captured_sketch - ctx.solved_plan_reached_goal = validated_solve - ctx.solved_plan_eval_reward = (float(verdict["reward"]) - if verdict is not None else None) - summary_bits = [ - f"validation: " - f"{1 + sum(1 for o in rollout_outcomes if 'FAILED' not in o)}" - f"/{1 + len(rollout_outcomes)} rollouts ok" - ] - if flaky_detail is not None: - summary_bits.append(f"first failure: {flaky_detail}") - if margin_outcomes: - n_margin_ok = sum(1 for o in margin_outcomes - if "FAILED" not in o) - summary_bits.append(f"physics margin: {n_margin_ok}/" - f"{len(margin_outcomes)} points ok") - if clearance_summary: - summary_bits.append(clearance_summary) - ctx.solved_plan_validation_summary = "; ".join(summary_bits) - n_annot = sum(1 for s in captured_sketch - if s.subgoal_atoms or s.subgoal_neg_atoms) - reason = capture_outcome.best_effort_reason - if reason is None: - best_effort_note = "" - elif reason is BestEffortReason.GOAL_NOT_REACHED: - best_effort_note = (" (best-effort: goal NOT reached, " - "accepted because the attempt budget is " - "exhausted; it executes for its honest " - "reward but will not count as a solve)") - elif reason is BestEffortReason.REWARD_HACK: - best_effort_note = (" (best-effort: the rollout reaches the " - "goal atoms but the task evaluator " - "scores it as a non-solve, and the real " - "env applies the same scoring; accepted " - "because the attempt budget is exhausted " - "- it executes for its honest reward but " - "will not count as a solve)") - elif reason is BestEffortReason.FLAKY: - best_effort_note = (f" (best-effort: {flaky_detail}; " - "accepted because the attempt budget is " - "exhausted - it executes for its honest " - "reward but may not reproduce its " - "solve)") - elif reason is BestEffortReason.PARAM_SENSITIVE: - best_effort_note = (" (best-effort: failed " - f"{any_margin_detail}; accepted " - "because the attempt budget is exhausted " - "- it executes for its honest reward but " - "may fail under the true physics)") - else: - assert reason is BestEffortReason.REDUNDANT - best_effort_note = (" (best-effort: the plan also reaches " - f"the goal without {redundant_detail}; " - "accepted because the attempt budget is " - "exhausted - it executes for its honest " - "reward, padding included)") - unverifiable_note = "" - if unverifiable_dropped: - dropped_lines = "\n".join(f" {d}" - for d in unverifiable_dropped) - unverifiable_note = ( - f"\nNOTE: {len(unverifiable_dropped)} annotation(s) " - "cannot be verified from a real observation (their " - "truth here depends on the belief latent, which real " - "env states do not carry) and were excluded from " - "closed-loop monitoring - the plan still executes " - "them, they just cannot trigger a replan:\n" - f"{dropped_lines}\n" - "Prefer annotating with predicates whose classifiers " - "read observable features.") - logging.info( - "Capture: excluded %d execution-unverifiable " - "annotation(s) from the monitored sketch:\n%s", - len(unverifiable_dropped), dropped_lines) - lines.append(f"Captured as the current answer{best_effort_note}: " - f"{len(grounded_plan)} steps, " - f"{n_annot} with subgoal annotations for closed-loop " - f"monitoring.{validation_note}{unverifiable_note}") - elif decision is CaptureDecision.FLAKY_NO_CAPTURE: - # Record the task so later submissions face the escalated - # gate - flakiness here is evidence the whole parameter - # region is marginal, not just this point. - ctx.flaky_capture_task_keys.add(capture_task_key) - _stash_uncaptured_submission() - escalated_n = max(max(1, validation_cfg.rollouts), - validation_cfg.rollouts_after_flaky) - n_ok = 1 + sum(1 for o in rollout_outcomes if "FAILED" not in o) - per_rollout = "\n".join(f" {o}" for o in rollout_outcomes) - lines.append( - f"FLAKY (plan NOT captured): the plan reached the goal on " - f"rollout 1 but {flaky_detail}. Per-rollout outcomes " - f"(estimated reliability {n_ok}/{n_rollouts}):\n" - f" rollout 1 (planner seed {base_planner_seed}): " - f"goal reached\n{per_rollout}\n" - "The simulator's motion " - "planning and physics stepping vary across runs, and the " - "real environment samples the same variability - a plan " - "that only sometimes succeeds in simulation will likely " - "fail for real. This gate is reproducible in run_python: " - f"sim.run(plan, trials={n_rollouts}) runs the identical " - "rollout set (fresh env per trial, same planner seeds), " - "and sim.run(plan, seed=, fresh=True) re-runs one failed rollout exactly " - "with full per-step reporting (without fresh=True the " - "warm session env is a different, optimistic substrate). " - "Then add margin (e.g. tighter spacing, aim " - "impacts closer to the middle of the fall path) and " - "resubmit. Because this task has now produced a flaky " - f"submission, captures require {escalated_n}/{escalated_n} " - "successful rollouts: fix the margin rather than " - "resubmitting near-identical parameters.") - elif decision is CaptureDecision.REDUNDANT_NO_CAPTURE: - _stash_uncaptured_submission() - per_step = "\n".join(f" {o}" for o in necessity_outcomes) - lines.append( - "REDUNDANT (plan NOT captured): every validation rollout " - f"reached the goal, but so does the plan without " - f"{redundant_detail}. A captured plan is an explanation of " - "how the goal comes about, and a step whose absence changes " - "nothing explains nothing: it only spends real episode " - "steps, and if your model is wrong about it, it can break " - "the plan for real. Per-step ablation:\n" - f"{per_step}\n" - "Drop the unnecessary step(s) and resubmit the shortest " - "plan your model needs. If you believed that step was " - "necessary, your model disagrees: check the belief before " - "resubmitting.") - elif (decision is CaptureDecision.PARAM_SENSITIVE_NO_CAPTURE - and param_sensitive_detail is None): - _stash_uncaptured_submission() - lines.append( - "CLEARANCE-SENSITIVE (plan NOT captured): every validation " - f"rollout reached the goal, but {clearance_detail}. The " - "real executor realizes each move anywhere within that " - "slop (pose tolerance, grasp height, landing scatter), so " - "a plan this tight succeeds in the belief by the luck of " - "the draw and fails for real on the next one. Widen the " - "margin at that step: stage and place objects farther " - "apart than the gripper's footprint, hover higher above " - "faces, or grasp farther from the neighbor - then " - "resubmit.") - elif decision is CaptureDecision.PARAM_SENSITIVE_NO_CAPTURE: - _stash_uncaptured_submission() - per_point = "\n".join(f" {o}" for o in margin_outcomes) - lines.append( - "PARAM-SENSITIVE (plan NOT captured): the plan passed " - "execution validation at the fitted physical parameters " - f"but FAILED {param_sensitive_detail}.\n" - "Physics-margin rollouts (a grid spanning +-1 sigma of " - "the identified physical parameters, the sysID fit's " - f"own uncertainty):\n{per_point}\n" - "The fitted values are uncertain at this scale and the " - "real environment may sit anywhere in that range - " - "including BETWEEN passing points: success can be " - "non-monotonic in a physical parameter, so a design must " - "hold across the whole range, not just at the values you " - "tuned at. Add margin to the DESIGN (not the execution) - " - "e.g. tighter spacing or impacts nearer the middle of the " - "fall path - then resubmit.") - if CFG.code_sim_learning_interval_belief and any( - "goal reached" in o for o in margin_outcomes): - lines.append( - "The belief interval straddles this plan's success " - "boundary (the passing and failing ranges are named " - "above). One real experiment that narrows that " - "parameter is worth more than more planning here: call " - "sim.suggest_probes(plan_text) to rank probes on your " - "sketch, run the best one for real, refit, and " - "resubmit - or find a design that holds across the " - "whole interval.") - elif decision is CaptureDecision.REWARD_HACK_NO_CAPTURE: - assert verdict is not None - _stash_uncaptured_submission() - lines.append( - "NOT CAPTURED: the rollout reaches the goal atoms but the " - "task evaluator scores it as a non-solve (solved=False, " - f"reward={verdict['reward']:.2f}). The real env applies the " - "same scoring, so executing this plan cannot count as a " - "solve. Find a plan whose rollout the evaluator scores " - "solved=True.") - if result.first_failure_idx is not None: - fr = result.steps[result.first_failure_idx].failure_reason - lines.append( - f"\nPlan FAILED at step {result.first_failure_idx}: {fr}") - final_atoms_str = ", ".join(str(a) for a in sorted(final_atoms)) - lines.append(f"\nFinal atoms: {{{final_atoms_str}}}") - if task.goal_nl: - lines.append(f"Goal (natural language): {task.goal_nl}") - else: - goal_str = ", ".join(str(g) for g in sorted(task.goal)) - lines.append(f"Goal: {{{goal_str}}}") - lines.append(f"Goal achieved: {goal_achieved}") - # Task-evaluator verdict line (verdict computed above, before the - # capture decision it gates). On a FLAKY rejection this verdict is - # rollout 1's only - printing it unlabeled next to a failing - # rollout's non-solve read as two contradictory verdicts in one - # message (run_20260717_182040 seed1 turn 96). - if verdict is not None: - vline = _format_evaluator_verdict(verdict, - coarse=eval_collector.coarse) - if flaky_detail is not None and not captured: - vline += (" [rollout 1 only - NOT the operative outcome; " - "this submission was rejected as FLAKY above]") - lines.append(vline) - # Goal atoms hold but the plan needs more low-level steps than the - # episode horizon allows: say so and that it was NOT captured, so the - # agent shortens the plan instead of stopping on a false positive. - # (A best-effort capture still happens above; then only warn.) - if goal_reached and not within_horizon and not captured: - lines.append( - f"NOT EXECUTABLE (plan was NOT captured): reaching the goal " - f"takes {result.actions_to_goal} low-level steps but the " - f"episode horizon is {horizon}. The real executor will run " - f"out of steps — shorten the plan (fewer or quicker steps) " - f"before resubmitting.") - elif goal_reached and not within_horizon: - lines.append( - f"WARNING: reaching the goal takes {result.actions_to_goal} " - f"low-level steps but the episode horizon is {horizon}, so " - f"the real executor will run out of steps before the goal.") - # Print the missing goal atoms even when the goal is stated in - # natural language: "Goal achieved: False" with no per-atom - # diagnosis left agents unable to tell a near-miss from a - # non-starter, and validation-rollout failures already name the - # missing atoms - this just makes rollout 1 report the same way. - if not goal_reached: - missing = _missing_goal_atoms(task, result.final_state) - missing_str = ", ".join(str(a) for a in sorted(missing)) - lines.append(f"Missing goal atoms: {{{missing_str}}}") - - # Append image save paths to text output - if saved_image_paths: - lines.append("\nSaved images:") - for p in saved_image_paths: - lines.append(f" {p}") - - # Build result with text only (images are saved to disk) - return _text_result("\n".join(lines) + - _budget_footer(ctx, rollouts_before)) - - @tool( - "submit_policy", - "Validate ./policy.py - your closed-loop `get_option(state, memory)` " - "program - on the CURRENT task and capture it as your answer. The " - "policy source is SNAPSHOTTED at call time (later edits need a new " - "call). Each rollout runs the policy closed-loop through the belief " - "model: get_option is called at every option boundary with the " - "actual current state; option failures (not initiable, motion-" - "planning refusal, 0 actions) do NOT end the episode - the failure " - "text arrives in memory['last_failure'] and get_option is asked " - "again, so RECOVERY is your policy's job; exceptions in get_option, " - "unparsable/ungroundable lines, re-issuing one identical " - "failing line repeatedly (the stuck-loop guard), and re-issuing " - "one identical line that keeps completing with no state change " - "(its no-op livelock twin) DO end it. Capture " - "is gated like " - "submit_plan: the goal-reaching rollout is repeated " - "several times (fresh simulator env + varied planner seed per " - "repeat, fresh memory per episode) and a FLAKY policy is reported " - "instead of captured; physics-margin perturbations apply too. " - "`validation_rollouts` requests a stricter gate. Test recovery " - "behavior first with sim.run_policy() in " - "run_python, which runs ./policy.py from the CURRENT probe " - "state (including perturbed or mid-plan states).", - { - "type": "object", - "properties": { - "validation_rollouts": { - "type": - "integer", - "description": - "Request a stricter capture gate: total validation " - "rollouts a goal-reaching policy must pass (effective " - "count is max(configured, this); never fewer).", - }, - "include_atoms": { - "type": "boolean", - "description": - "Include atoms added/deleted after each step", - "default": True - }, - }, - }, - ) - async def submit_policy(args: Dict[str, Any]) -> Dict[str, Any]: - # pylint: disable-next=import-outside-toplevel - from predicators.agent_sdk.policy_execution import \ - build_policy_option_fn, execute_policy_forward - validation_cfg = ValidationConfig.from_cfg() - ctx.test_call_id += 1 - rollouts_before = ctx.attempt_rollout_count - if not ctx.policy_capture_mode: - return _error_result( - "submit_policy is only available in policy mode " - "(agent_solve_policy_mode); submit plans via " - "submit_plan instead.") - if ctx.option_model is None: - return _error_result("No option model available in ToolContext.") - all_options = ctx.options - model = ctx.option_model - model._name_to_parameterized_option = ( # type: ignore[attr-defined] # pylint: disable=protected-access - {o.name: o - for o in all_options}) - requested_rollouts = args.get("validation_rollouts") - include_atoms = args.get("include_atoms", True) - if requested_rollouts is not None and (not isinstance( - requested_rollouts, int) or requested_rollouts < 1): - return _error_result( - "validation_rollouts must be a positive integer.") - - resolved, task_err = _resolve_task(ctx, None) - if task_err is not None: - return task_err - assert resolved is not None - task = resolved.task - task_label = resolved.label - - policy_path = _policy_source_path(ctx) - if policy_path is None or not os.path.isfile(policy_path): - return _error_result( - "No ./policy.py found. Write your closed-loop policy there " - "first: `def get_option(state, memory): ...` returning one " - "plan line (sketch grammar) or None for DONE.") - with open(policy_path, "r", encoding="utf-8") as f: - policy_source = f.read() - - all_predicates = ctx.predicates - types = set(ctx.types) - for opt in all_options: - types.update(opt.types) - for pred in all_predicates: - types.update(pred.types) - types.update(o.type for o in task.init) - - def _fresh_option_fn() -> Tuple[Optional[Any], Optional[str]]: - # Fresh instance per episode: memory must reset per rollout. - return build_policy_option_fn(policy_source, - task, - predicates=all_predicates, - options=all_options, - types=types) - - option_fn, load_err = _fresh_option_fn() - if load_err is not None or option_fn is None: - return _error_result(load_err or "policy.py failed to load.") - max_opts = CFG.agent_policy_max_options - horizon = ctx.execution_step_budget() - - lines = [f"Testing policy.py on task {task_label}:"] - saved_image_paths: List[str] = [] - eval_collector = _EvalStateCollector(model, task.init) - - def _report_step(i: int, outcome: Any) -> None: - eval_collector.collect(outcome) - opt = outcome.option - sig = f"{opt.name}({[o.name for o in opt.objects]})" - step_line = f"Step {i}: {sig} ({outcome.num_actions} actions)" - if outcome.failure_reason is not None: - step_line += ( - f"\n OPTION FAILURE (surfaced to the policy as " - f"memory['last_failure']): {outcome.failure_reason}") - post = outcome.post_state - if post is not None and include_atoms: - before = utils.abstract(outcome.pre_state, ctx.predicates) - after = utils.abstract(post, ctx.predicates) - added_s = ", ".join(str(a) for a in sorted(after - before)) - del_s = ", ".join(str(a) for a in sorted(before - after)) - step_line += (f"\n Added: {{{added_s}}}" - f"\n Deleted: {{{del_s}}}") - lines.append(step_line) - # Explicit state: the rollout runs on the gate's fresh env - # while the renderer draws the shared session env (see - # submit_plan's _report_step). - img_block = render_pybullet_image( - ctx, - f"policy_step_{i}_{opt.name}", - state=post if post is not None else outcome.pre_state) - if img_block and img_block.get("saved_path"): - saved_image_paths.append(img_block["saved_path"]) - - # One substrate for the whole gate, as in submit_plan: rollout 1 - # runs on the same fresh env as the validation repeats, at the - # base planner seed, so the gate is reproducible in-session. - fresh_scope = (ctx.validation_env_scope - if validation_cfg.fresh_env else None) - ctx.attempt_rollout_count += 1 - with (fresh_scope() - if fresh_scope is not None else contextlib.nullcontext()): - result = execute_policy_forward(task, - option_fn, - model, - predicates=all_predicates, - max_policy_options=max_opts, - on_step=_report_step) - # Verdict INSIDE the scope: certificate probes must judge on - # the env the rollout ran on. - evaluator = _resolve_task_evaluator(ctx, task_label) - verdict: Optional[Dict[str, Any]] = None - if evaluator is not None and len(eval_collector.states) > 1: - try: - verdict = evaluate_states_with(evaluator, - eval_collector.states, - eval_collector.labels, - sim_env=getattr( - ctx.option_model, - "sim_env", None)) - except Exception as e: # pylint: disable=broad-except - logging.debug("Task-evaluator verdict failed: %s", e) - - goal_reached = result.goal_reached - within_horizon = (result.actions_to_goal is not None - and result.actions_to_goal <= horizon) - if result.policy_error is not None: - lines.append(f"POLICY ERROR (ended the episode): " - f"{result.policy_error}") - n_surfaced = sum(1 for s in result.steps - if s.failure_reason is not None) - if n_surfaced: - lines.append( - f"{n_surfaced} option failure(s) were surfaced to the " - "policy during this rollout (recovery attempts included " - "above).") - # Closed-loop: recovered option failures do NOT disqualify - the - # policy handling them is the point. Only the goal, the horizon, - # and policy-code errors gate. - goal_achieved = (goal_reached and within_horizon - and result.policy_error is None) - evaluator_rejected = (verdict is not None and not verdict["legitimate"] - and not eval_collector.coarse) - reward_hack = (evaluator_rejected and verdict is not None - and verdict["terminated"]) - - def _policy_validation_rollout() -> Tuple[bool, str]: - fn, err = _fresh_option_fn() - if err is not None or fn is None: - return False, f"policy failed to load: {err}" - v_collector = _EvalStateCollector(model, task.init) - r = execute_policy_forward(task, - fn, - model, - predicates=all_predicates, - max_policy_options=max_opts, - on_step=v_collector.on_step) - if r.policy_error is not None: - return False, f"policy error: {r.policy_error}" - if not r.goal_reached: - missing = _missing_goal_atoms(task, r.final_state) - missing_str = ", ".join(str(a) for a in sorted(missing)) - detail = f" (missing: {{{missing_str}}})" if missing else "" - return False, f"goal not reached{detail}" - if not (r.actions_to_goal is not None - and r.actions_to_goal <= horizon): - return False, (f"goal reached only after {r.actions_to_goal} " - f"low-level steps, past the episode horizon " - f"({horizon})") - if (evaluator is not None and len(v_collector.states) > 1 - and not v_collector.coarse): - try: - v = evaluate_states_with(evaluator, - v_collector.states, - v_collector.labels, - sim_env=getattr( - ctx.option_model, "sim_env", - None)) - if not v["legitimate"]: - return False, ( - "this rollout reached the goal atoms but the " - "task evaluator scored it as a non-solve " - f"(solved=False, reward={v['reward']:.2f})") - except Exception as e: # pylint: disable=broad-except - logging.debug("Validation-rollout verdict failed: %s", e) - return True, "" - - flaky_detail: Optional[str] = None - validation_note = "" - n_rollouts = max(1, validation_cfg.rollouts) - capture_task_key = _capture_task_key(ctx) - if capture_task_key in ctx.flaky_capture_task_keys: - n_rollouts = max(n_rollouts, validation_cfg.rollouts_after_flaky) - if requested_rollouts is not None: - n_rollouts = max(n_rollouts, - min(requested_rollouts, _MAX_REQUESTED_ROLLOUTS)) - # fresh_scope computed above, shared with rollout 1. - rollout_outcomes: List[str] = [] - base_planner_seed = CFG.seed - if (ctx.capture_goal_reaching_plans and goal_achieved - and not evaluator_rejected and n_rollouts > 1): - - def _policy_repeat_rollout(repeat_idx: int) -> Tuple[bool, str]: - with (fresh_scope() if fresh_scope is not None else - contextlib.nullcontext()), \ - decorrelated_rollout_seed(repeat_idx - 1): - return _policy_validation_rollout() - - repeat_indices = list(range(2, n_rollouts + 1)) - repeat_prefetched = _prefetch_parallel([ - functools.partial(_policy_repeat_rollout, k) - for k in repeat_indices - ], "policy repeat rollouts") - for pos, repeat_idx in enumerate(repeat_indices): - ctx.attempt_rollout_count += 1 - pre = repeat_prefetched[pos] - ok, why = (pre if pre is not None else - _policy_repeat_rollout(repeat_idx)) - repeat_seed = base_planner_seed + repeat_idx - 1 - if ok: - rollout_outcomes.append( - f"rollout {repeat_idx} (planner seed " - f"{repeat_seed}): goal reached") - else: - rollout_outcomes.append( - f"rollout {repeat_idx} (planner seed " - f"{repeat_seed}): FAILED - {why}") - if flaky_detail is None: - flaky_detail = (f"rollout {repeat_idx}/{n_rollouts} " - f"(planner seed {repeat_seed}) " - f"FAILED: {why}") - if flaky_detail is None: - validation_note = ( - f" Validated {n_rollouts}/{n_rollouts} rollouts " - f"(planner seeds {base_planner_seed}-" - f"{base_planner_seed + n_rollouts - 1}; fresh env and " - "fresh policy memory per rollout).") - - # Parameter-margin gates, mirroring submit_plan (one - # shared code path: see _parameter_margin_sweep). - param_sensitive_detail: Optional[str] = None - margin_outcomes: List[str] = [] - if (fresh_scope is not None and ctx.capture_goal_reaching_plans - and goal_achieved and not evaluator_rejected - and flaky_detail is None): - margin_outcomes, param_sensitive_detail, margin_note = \ - _parameter_margin_sweep(ctx, validation_cfg, fresh_scope, - _policy_validation_rollout, "policy") - validation_note += margin_note - - capture_outcome = _decide_capture( - capture_enabled=(ctx.capture_goal_reaching_plans - and ctx.policy_capture_mode), - is_current_task=True, - have_plan=True, - goal_achieved=goal_achieved, - evaluator_rejected=evaluator_rejected, - reward_hack=reward_hack, - flaky=flaky_detail is not None, - best_effort_mode=ctx.capture_best_effort_plan, - have_validated_capture=bool(ctx.solved_plan_reached_goal), - param_sensitive=param_sensitive_detail is not None) - decision = capture_outcome.decision - captured = capture_outcome.captured - if captured: - validated_solve = decision is CaptureDecision.VALIDATED_CAPTURE - ctx.solved_plan = None - ctx.solved_sketch = None - ctx.solved_policy_source = policy_source - ctx.solved_plan_reached_goal = validated_solve - ctx.solved_plan_eval_reward = (float(verdict["reward"]) - if verdict is not None else None) - summary_bits = [ - f"validation: " - f"{1 + sum(1 for o in rollout_outcomes if 'FAILED' not in o)}" - f"/{1 + len(rollout_outcomes)} rollouts ok" - ] - if flaky_detail is not None: - summary_bits.append(f"first failure: {flaky_detail}") - if margin_outcomes: - n_margin_ok = sum(1 for o in margin_outcomes - if "FAILED" not in o) - summary_bits.append(f"physics margin: {n_margin_ok}/" - f"{len(margin_outcomes)} points ok") - ctx.solved_plan_validation_summary = "; ".join(summary_bits) - reason = capture_outcome.best_effort_reason - best_effort_note = "" - if reason is not None: - best_effort_note = ( - " (best-effort: accepted because the attempt budget is " - "exhausted; it executes for its honest reward but may " - "not count as a solve)") - lines.append( - f"Captured policy.py as the current answer{best_effort_note}" - f": {len(result.steps)} option(s) in the capture rollout." - f"{validation_note}") - elif decision is CaptureDecision.FLAKY_NO_CAPTURE: - ctx.flaky_capture_task_keys.add(capture_task_key) - per_rollout = "\n".join(f" {o}" for o in rollout_outcomes) - n_ok = 1 + sum(1 for o in rollout_outcomes if "FAILED" not in o) - lines.append( - f"FLAKY (policy NOT captured): rollout 1 reached the goal " - f"but {flaky_detail}. Per-rollout outcomes (estimated " - f"reliability {n_ok}/{n_rollouts}):\n{per_rollout}\n" - "A closed-loop policy that cannot recover in some rollouts " - "needs better feedback handling - inspect the failing " - "seeds with rollout_seed and strengthen the recovery " - "branches.") - elif decision is CaptureDecision.PARAM_SENSITIVE_NO_CAPTURE: - per_point = "\n".join(f" {o}" for o in margin_outcomes) - lines.append(f"PARAM-SENSITIVE (policy NOT captured): failed " - f"{param_sensitive_detail}. Per-point outcomes:\n" - f"{per_point}") - elif decision is CaptureDecision.REWARD_HACK_NO_CAPTURE: - lines.append( - "NOT captured: the rollout reaches the goal atoms but the " - "task evaluator scores it as a non-solve, and the real env " - "applies the same scoring.") - - lines.append(f"Goal achieved: {goal_reached}") - if verdict is not None: - lines.append( - _format_evaluator_verdict(verdict, - coarse=eval_collector.coarse)) - if goal_reached and not within_horizon: - lines.append( - f"NOT EXECUTABLE: reaching the goal takes " - f"{result.actions_to_goal} low-level steps but the episode " - f"horizon is {horizon}.") - if not goal_reached: - missing = _missing_goal_atoms(task, result.final_state) - missing_str = ", ".join(str(a) for a in sorted(missing)) - lines.append(f"Missing goal atoms: {{{missing_str}}}") - if saved_image_paths: - lines.append("\nSaved images:") - lines.extend(f" {p}" for p in saved_image_paths) - return _text_result("\n".join(lines) + - _budget_footer(ctx, rollouts_before)) - - return { - "submit_plan": submit_plan, - "submit_policy": submit_policy, - } diff --git a/predicators/agent_sdk/tools/verdicts.py b/predicators/agent_sdk/tools/verdicts.py index 771283d477..30e8e72b9d 100644 --- a/predicators/agent_sdk/tools/verdicts.py +++ b/predicators/agent_sdk/tools/verdicts.py @@ -16,11 +16,11 @@ class _EvalStateCollector: """Per-step states + option labels of one rollout, for evaluator verdicts. The single collector behind every surface that scores a belief-sim - rollout (``submit_plan``'s first and validation rollouts, - ``_belief_rollout_verdict``). The cascade certificate needs per-step - states (topple-onset analysis); option-boundary states give garbage - verdicts, so prefer the option model's ``last_trajectory`` and flag - the verdict as coarse (``self.coarse``) when it is unavailable. + rollout (``_belief_rollout_verdict``). The cascade certificate needs + per-step states (topple-onset analysis); option-boundary states give + garbage verdicts, so prefer the option model's ``last_trajectory`` + and flag the verdict as coarse (``self.coarse``) when it is + unavailable. """ def __init__(self, option_model: Any, init_state: State) -> None: @@ -174,6 +174,14 @@ def _sandbox_base(ctx: ToolContext) -> Optional[str]: return ctx.log_dir or None +def _policy_source_path(ctx: ToolContext) -> Optional[str]: + """Host path of the agent-editable ``policy.py``.""" + base = _sandbox_base(ctx) + if not base: + return None + return os.path.join(base, "policy.py") + + def _ground_samplers_path(ctx: ToolContext) -> Optional[str]: """Host path of the agent-editable ``ground_samplers.py``.""" base = _sandbox_base(ctx) diff --git a/predicators/approaches/agent_continual_ablation_approach.py b/predicators/approaches/agent_continual_ablation_approach.py index 88c3f924e2..e55ed6f2e6 100644 --- a/predicators/approaches/agent_continual_ablation_approach.py +++ b/predicators/approaches/agent_continual_ablation_approach.py @@ -50,8 +50,6 @@ def __init__(self, *args: Any, **kwargs: Any) -> None: disabled = ( "continual_uncertainty_decisions", "agent_sim_learn_param_uncertainty", - "agent_plan_validation_rule_param_margin", - "agent_plan_validation_physics_margin", "agent_explorer_info_seeking", "agent_explorer_info_seeking_adaptive", "agent_explorer_info_seeking_noise_aware", diff --git a/predicators/approaches/agent_continual_real_to_sim_approach.py b/predicators/approaches/agent_continual_real_to_sim_approach.py index 4e8dd3e97e..202f6863cb 100644 --- a/predicators/approaches/agent_continual_real_to_sim_approach.py +++ b/predicators/approaches/agent_continual_real_to_sim_approach.py @@ -27,8 +27,6 @@ _UNCERTAINTY_FLAGS = ( "continual_uncertainty_decisions", "agent_sim_learn_param_uncertainty", - "agent_plan_validation_rule_param_margin", - "agent_plan_validation_physics_margin", "agent_explorer_info_seeking", "agent_explorer_info_seeking_adaptive", "agent_explorer_info_seeking_noise_aware", diff --git a/predicators/approaches/agent_program_world_model_approach.py b/predicators/approaches/agent_program_world_model_approach.py index 1f7e500969..4c22ae77fd 100644 --- a/predicators/approaches/agent_program_world_model_approach.py +++ b/predicators/approaches/agent_program_world_model_approach.py @@ -2,31 +2,25 @@ invented predicates (paper arm C4: a code world model with no engine underneath, in the form of Pinductor / POMDP Coder). -Everything about the loop is the residual arm's - the explorer, the -sketch / refine / run tools, the capture gate, the solve pipeline, -predicate invention - except the model artifact: instead of residual -rules over a physics engine the agent writes ``world_model.py``, an -option-level transition program with its own hidden state (see +Everything about the loop is the residual arm's - the sketch / refine / +run tools and predicate invention - except the model artifact: instead of +residual rules over a physics engine the agent writes ``world_model.py``, +an option-level transition program with its own hidden state (see :mod:`code_sim_learning.program_world_model`). There is no parameter -fit; the learn session scores the program with the Pinductor +fit; the synthesis tools score the program with the Pinductor particle-filter kernel pseudo-likelihood (``sim.score``) and the agent -edits it. The belief over the hidden state is a particle set drawn from -the program's ``initial_latent``; the capture gate re-rolls every -submission under every particle through the residual arm's -rule-parameter margin channel. +edits it. """ from __future__ import annotations -import copy import logging import os from contextlib import contextmanager from typing import Any, Callable, Dict, FrozenSet, Iterator, List, Optional, \ - Tuple, cast + Tuple import numpy as np -from predicators import utils from predicators.agent_sdk.tools import _SnapshotTarget from predicators.agent_sdk.tools.program_synthesis import CandidateLoader, \ create_program_synthesis_tools @@ -58,15 +52,6 @@ def __init__(self, *args: Any, **kwargs: Any) -> None: super().__init__(*args, **kwargs) self._program: Optional[ProgramWorldModel] = None self._program_model: Optional[ProgramOptionModel] = None - # The belief particles ride the capture gate's rule-parameter - # margin channel: every submission is re-rolled from each - # particle's hidden state. - ctx = self._tool_context - ctx.rule_param_margin_provider = self._belief_particles - ctx.rule_param_override_scope = self._particle_override_scope - ctx.rule_param_margin_label = "belief particle" - ctx.rule_param_margin_note = ( - "particles of the belief over the world model's hidden state") @classmethod def get_name(cls) -> str: @@ -252,46 +237,6 @@ def _load_program_file( # ── Belief over the hidden state ───────────────────────────── - def _belief_particles(self) -> List[Dict[str, float]]: - """Distinct draws from ``initial_latent`` for the current task: the - capture gate's margin points (empty until a model exists).""" - model = self._program_model - if model is None: - return [] - task = self._tool_context.current_task - if task is None: - if not self._train_tasks: - return [] - task = self._train_tasks[0] - rng = np.random.default_rng(CFG.seed + 7919) - particles: List[Dict[str, Any]] = [] - seen = set() - for _ in range(CFG.agent_program_belief_particles): - try: - latent = model.initial_latent(task.init, rng=rng) - except utils.OptionExecutionFailure as e: - logger.warning("Belief particles unavailable: %s", e) - break - key = repr(sorted(latent.items(), key=lambda kv: str(kv[0]))) - if key in seen: - continue - seen.add(key) - particles.append(latent) - return cast(List[Dict[str, float]], particles) - - @contextmanager - def _particle_override_scope(self, - particle: Dict[str, float]) -> Iterator[None]: - """Roll every latent-less start from ``particle`` while entered.""" - model = self._program_model - assert model is not None - prev = model.initial_latent_override - model.initial_latent_override = copy.deepcopy(particle) - try: - yield - finally: - model.initial_latent_override = prev - def materialise_latent( self, traj: LowLevelTrajectory) -> List[Optional[Dict[str, Any]]]: if self._program is None: diff --git a/predicators/approaches/agent_sim_learning_approach.py b/predicators/approaches/agent_sim_learning_approach.py index 1f910b482c..50c24422c9 100644 --- a/predicators/approaches/agent_sim_learning_approach.py +++ b/predicators/approaches/agent_sim_learning_approach.py @@ -25,8 +25,8 @@ from gym.spaces import Box from predicators import utils -from predicators.agent_sdk import learn_prompts from predicators.agent_sdk.fit_status import format_fit_status +from predicators.agent_sdk.play_prompts import render_physical_params_section from predicators.agent_sdk.session_base import max_session_log_number from predicators.agent_sdk.tools import SYNTHESIS_TOOL_NAMES, \ _SnapshotTarget, evaluate_states_with, finalize_versioned_snapshot, \ @@ -225,23 +225,12 @@ def __init__(self, # sample the distribution the real episode will. self._tool_context.validation_env_scope = \ self._fresh_validation_env_scope - # Physics-margin points for the capture gate (+-1 posterior sigma - # of the latest applied fit): a callable so the tool always sees - # the current fit, not the one deployed when the session opened. - # Under agent_sim_learn_param_uncertainty False the points are - # never built (see _physics_margin_points), so this returns []. + # The sim.run physics sweep's stress points (see _stress_points): + # a callable so the probe always sees the current fit, not the + # one deployed when the session opened. Under + # agent_sim_learn_param_uncertainty False the legacy grid is never + # built (see _physics_margin_points). self._tool_context.physics_margin_provider = self._stress_points - # Rule-parameter margin points for the capture gate: the - # calibrated ensemble the info-seeking explorer scores with - # (posterior subsample / Laplace / jitter, see - # _select_param_ensemble) doubles as the uncertainty sweep over - # LEARNED rule constants - a submission must survive every - # member, not just the fitted point estimate. Callables so the - # gate always sees the latest fit's ensemble. - self._tool_context.rule_param_margin_provider = \ - lambda: [dict(m) for m in self._param_ensemble] - self._tool_context.rule_param_override_scope = \ - self._rule_param_override_scope # The joint belief's rehearsals split a draw's parameters into the # ones the env applies and the rule parameters read through the # draw scope. @@ -342,7 +331,7 @@ def __init__(self, # delta against the previous version reads from here. self._fit_evidence_history: Dict[str, Dict[str, float]] = {} # +-1-posterior-sigma perturbations of the applied params (the - # capture gate's physics-margin points). Set only by the joint + # legacy physics sweep's grid). Set only by the joint # rollout fit, which has the identifiability report; cleared by # every _apply_identified_physical_params call so points can # never outlive the fit they were derived from. @@ -416,26 +405,18 @@ def _get_synthesis_tool_names(self) -> Optional[List[str]]: """Complete tool surface for the synthesis agent. The names of the dynamic synthesis callables (just - ``run_python``) attached to ``ctx.extra_mcp_tools`` inside - :meth:`_synthesize_with_agent`. The mixin asserts the attached - instances and this list agree. Fitting, residual reports, and - plan validation are NOT tools: they live on the ``sim`` probe - (``sim.fit`` / ``sim.residuals`` / ``sim.refine`` / - ``sim.run``) inside ``run_python``. - - No inspect tools: the type/option digests are injected into the - learn message (see :meth:`_build_synthesis_learn_message`) and - trajectory access lives in ``run_python`` (``trajectories`` + - ``describe_trajectory``). The probe rides inside that same - ``run_python`` namespace as ``sim`` (one exec namespace per - session - a helper defined next to the data is visible to probe - sweeps; the solve-phase instance of the tool is not built when - this one is attached). In the - agent-synthesis session the probe runs against the CANDIDATE - simulator.py via ctx.probe_option_model_provider (installed in - _synthesize_with_agent); in the oracle-sim-program sampler - session no provider is installed and the probe falls back to - ctx.option_model, which there IS the deployed belief model. + ``run_python``); a continual round keeps only the toolkit tools + named here. Fitting, residual reports, and plan validation are + NOT tools: they live on the ``sim`` probe (``sim.fit`` / + ``sim.residuals`` / ``sim.refine`` / ``sim.run``) inside + ``run_python``. + + No inspect tools: trajectory access lives in ``run_python`` + (``trajectories`` + ``describe_trajectory``). The probe rides + inside that same ``run_python`` namespace as ``sim`` (one exec + namespace per session - a helper defined next to the data is + visible to probe sweeps) and runs against the CANDIDATE + simulator.py via ctx.probe_option_model_provider. """ names: List[str] = list(SYNTHESIS_TOOL_NAMES) return names @@ -1136,13 +1117,12 @@ def _provider() -> _OracleOptionModel: def _rebuild_param_ensemble(self) -> None: """Rebuild the learned model's rule-parameter ensemble. - Two consumers share it: info-seeking exploration (off under - ablation A6) and the capture gate's rule-param margin (off under - ablation A7). Built when either is on, a fit has populated - ``_fitted_params`` and parameter uncertainty is in use; cleared - otherwise. The ensemble can use an exploration-only posterior - even when solver params remain at the global-budget point - estimate. + Its consumer is info-seeking exploration + (:meth:`score_atom_disagreement`). Built when that is on, a fit + has populated ``_fitted_params`` and parameter uncertainty is in + use; cleared otherwise. The ensemble can use an exploration-only + posterior even when solver params remain at the global-budget + point estimate. Picks the most *calibrated* ensemble the fit affords, preferring spreads that reflect real posterior uncertainty over uniform @@ -1162,9 +1142,7 @@ def _rebuild_param_ensemble(self) -> None: "Rule-parameter ensemble: %d draws of the joint belief.", len(self._param_ensemble)) return - wanted = (CFG.agent_explorer_info_seeking - or CFG.agent_plan_validation_rule_param_margin) - if (not wanted or not self._fitted_params + if (not CFG.agent_explorer_info_seeking or not self._fitted_params or not CFG.agent_sim_learn_param_uncertainty): self._param_ensemble = [] return @@ -1304,9 +1282,7 @@ def _rule_param_override_scope( swap changes their gates for the wrapped validation rollout and the restore returns the deployed fit untouched - the same pattern :meth:`score_atom_disagreement` uses for ensemble - scoring. Installed on the tool context as - ``rule_param_override_scope`` for the capture gate's - rule-parameter margin sweep. + scoring. """ saved = dict(self._fitted_params) self._fitted_params.clear() @@ -1572,7 +1548,7 @@ def _physics_margin_points( report: Dict[str, Dict[str, Any]], physical_specs: List[ParamSpec], ) -> List[Dict[str, float]]: - """The capture gate's physics-margin grid for ``applied``. + """The physics sweep's +-1-sigma grid for ``applied``. Empty under ``agent_sim_learn_param_uncertainty`` False (ablations A6+A7 combined: point estimates only, so there is no width to @@ -2581,17 +2557,16 @@ def _fresh_model_env_scope( float]] = None) -> Iterator[None]: """Run the option model on a freshly constructed base env. - ``physical_overrides`` (the capture gate's physics-margin - rollouts) is applied to the fresh env ON TOP of the identified - params, so the rollout runs at a perturbed physics; the shared - session env is never touched. - - Installed as ``ToolContext.validation_env_scope`` so - ``submit_plan``'s capture-validation rollouts each sample - a fresh physics world. The shared ``_base_env``'s reset cannot - reconstruct state exactly (solver warm-start state, velocity - residuals, near-matching bodies skipped by the reconstruction diff - - the same mechanism measured in :func:`rollout_states`), so + ``physical_overrides`` (a physics-sweep point) is applied to the + fresh env ON TOP of the identified params, so the rollout runs at + a perturbed physics; the shared session env is never touched. + + Installed as ``ToolContext.validation_env_scope`` so the probe's + trials and sweep rollouts each sample a fresh physics world. The + shared ``_base_env``'s reset cannot reconstruct state exactly + (solver warm-start state, velocity residuals, near-matching bodies + skipped by the reconstruction diff - the same mechanism measured + in :func:`rollout_states`), so repeats on it are correlated with each other and systematically offset from the fresh env the real episode runs in (run_20260717_182321: a placement swept 20/20 on the shared env @@ -2814,4 +2789,4 @@ def _physical_params_prompt_section(self) -> str: getter = getattr(base_env, "get_physical_param_info", None) if callable(getter): info = getter() or {} - return learn_prompts.render_physical_params_section(info) + return render_physical_params_section(info) diff --git a/predicators/approaches/agent_sim_predicate_invention_approach.py b/predicators/approaches/agent_sim_predicate_invention_approach.py index b1b6a3cde7..4a242e1676 100644 --- a/predicators/approaches/agent_sim_predicate_invention_approach.py +++ b/predicators/approaches/agent_sim_predicate_invention_approach.py @@ -9,32 +9,23 @@ option model's abstraction function, and every other caller asking the approach for its current predicates. -Predicates persist across online learning cycles: ``predicates.py`` is -preserved at the sandbox root, and every version evaluated during -synthesis (plus a final snapshot of post-eval edits) is saved to -``predicates_versions/`` as ``cycle_XXX_vers_YYY_predicates.py``. - -Partial observability is not a separate approach: like every -sim-learning arm, the synthesis prompt follows -``CFG.partially_observable`` (see ``AgentSimLearningApproach``) - under -the flag the agent is taught the recurrent 5-arg rule signature -``rule(observation, latent, history, updates, params)`` and -``LATENT_INIT``, and this module appends the predicate-side latent -guidance (classifiers may take an optional ``latent`` kwarg, -auto-routed by ``Predicate.holds``). The latent *mechanics* (recurrent -LM fitting, the latent-threaded combined simulator riding +Predicates persist across rounds: ``predicates.py`` is preserved at the +sandbox root, and every version evaluated during synthesis (plus a final +snapshot of post-eval edits) is saved to ``predicates_versions/`` as +``cycle_XXX_vers_YYY_predicates.py``. + +Partial observability is not a separate approach. The latent mechanics +(recurrent LM fitting, the latent-threaded combined simulator riding ``State.latent`` so backtracking restores it per search node, ``LATENT_INIT`` loading and initial-latent seeding) live in ``AgentSimLearningApproach`` and activate automatically whenever the -loaded rules use the 5-arg signature, independent of the flag. - -Example command (partially observable):: +loaded rules use the 5-arg signature ``rule(observation, latent, +history, updates, params)``; classifiers may take an optional +``latent`` kwarg, auto-routed by ``Predicate.holds``. - python predicators/main.py --env pybullet_boil \ - --approach agent_sim_predicate_invention --seed 0 \ - --num_train_tasks 10 --num_test_tasks 5 \ - --partially_observable True \ - --num_online_learning_cycles 2 --explorer agent_model_free +The continual arms (``AgentContinualApproach`` and its variants) build +on this class; launch them through +``scripts/configs/empiric/benchmark.yaml``. """ import logging diff --git a/predicators/approaches/continual_play_mixin.py b/predicators/approaches/continual_play_mixin.py index ecff81e57f..0706ca1610 100644 --- a/predicators/approaches/continual_play_mixin.py +++ b/predicators/approaches/continual_play_mixin.py @@ -321,8 +321,8 @@ def _run_conversation_round(self, session: ProtocolSession) -> PlayState: kind = "continue" query = self._build_query(session, kind) round_number = self._rounds_played + 1 - # No per-round clock: the run's wall-clock cap is the only clock. - ctx.begin_attempt(round_number, 0.0) + # No per-round budget: the run's wall-clock cap is the only one. + ctx.begin_attempt() self._round_in_flight = True self.save(session.level_index) entries_before = len(session.index_entries()) @@ -331,7 +331,6 @@ def _run_conversation_round(self, session: ProtocolSession) -> PlayState: responses = self._query_agent_sync(query, kind=SESSION_KIND) finally: ctx.attempt_start = None - ctx.attempt_deadline = None self._round_in_flight = False self._agent_session.resume_session_id = None self._after_round(session, state) diff --git a/predicators/code_sim_learning/identifiability.py b/predicators/code_sim_learning/identifiability.py index 24908be060..fa40238b43 100644 --- a/predicators/code_sim_learning/identifiability.py +++ b/predicators/code_sim_learning/identifiability.py @@ -378,12 +378,10 @@ def physics_sigma_points(applied: Dict[str, float], num_points: int = 2) -> List[Dict[str, float]]: """A grid of perturbations spanning +-1 posterior sigma of the fit. - Consumed by the capture gate's physics-margin check - (``agent_plan_validation_physics_margin``) and by the ``sim.run`` - physics sweep: validation rollouts AT the fitted values sample - execution variability only, so a plan can pass them all while - having zero margin to the fit's parameter error - (run_20260723_091108: a capture validated 8/8 at fitted + Consumed by the ``sim.run`` physics sweep: rollouts AT the fitted + values sample execution variability only, so a plan can pass them + all while having zero margin to the fit's parameter error + (run_20260723_091108: a plan validated 8/8 at fitted lateral_friction 0.5319 failed deterministically at true 0.5). Each param whose FITTED value was deployed (``Verdict.applies_fitted``, which under the interval belief includes ``Verdict.WIDE``: a moved diff --git a/predicators/code_sim_learning/program_world_model.py b/predicators/code_sim_learning/program_world_model.py index 80f064df34..38ef65d139 100644 --- a/predicators/code_sim_learning/program_world_model.py +++ b/predicators/code_sim_learning/program_world_model.py @@ -136,9 +136,8 @@ class ProgramOptionModel(_OptionModelBase): The latent rides on ``State.latent``; a state without one (a task's initial state, a probe reset) is seeded from the program's - ``initial_latent`` - or from ``initial_latent_override`` while the - capture gate sweeps the belief particles. Program exceptions surface - as :class:`utils.OptionExecutionFailure` with the traceback tail in + ``initial_latent``. Program exceptions surface as + :class:`utils.OptionExecutionFailure` with the traceback tail in ``last_execution_failure``, the same channel the engine-backed model uses, so every consumer reports them the same way. """ @@ -147,7 +146,6 @@ def __init__(self, program: ProgramWorldModel, seed: int = 0) -> None: super().__init__() self._program = program self._rng = np.random.default_rng(seed) - self.initial_latent_override: Optional[Dict[str, Any]] = None self.last_execution_failure: Optional[str] = None self.last_trajectory: Optional[LowLevelTrajectory] = None self.sim_env: Optional[Any] = None @@ -161,10 +159,8 @@ def initial_latent( self, state: State, rng: Optional[np.random.Generator] = None) -> Dict[str, Any]: - """A latent for ``state``: the override if one is set, else a draw from - the program's ``initial_latent`` (with ``rng`` or the model's own).""" - if self.initial_latent_override is not None: - return copy.deepcopy(self.initial_latent_override) + """A latent for ``state``: a draw from the program's ``initial_latent`` + (with ``rng`` or the model's own).""" latent = self._program.initial_latent( _observation(state), rng if rng is not None else self._rng) if not isinstance(latent, dict): diff --git a/predicators/ground_truth_models/skill_factories/base.py b/predicators/ground_truth_models/skill_factories/base.py index 601a98ca3e..828caca57d 100644 --- a/predicators/ground_truth_models/skill_factories/base.py +++ b/predicators/ground_truth_models/skill_factories/base.py @@ -218,9 +218,6 @@ def _fmt_option_params(params: Array) -> str: # (state, objects, params, config) -> (x, y, z, yaw) TargetPoseFn = Callable[[State, Sequence[Object], Array, SkillConfig], Tuple[float, float, float, float]] -# Objects a skill contacts by design beyond its arguments, for a -# grounding; see ``PhaseSkill.contact_objects``. -ContactObjectsFn = Callable[[State, Sequence[Object]], Set[Object]] # --------------------------------------------------------------------------- # Internal type aliases for Phase target functions @@ -492,16 +489,14 @@ class PhaseSkill: option = PhaseSkill("Pick", types, params_space, config, phases).build() """ - def __init__( - self, - name: str, - types: Sequence[Type], - params_space: Box, - config: SkillConfig, - phases: List[Phase], - params_description: Optional[Tuple[str, ...]] = None, - base_mode: Optional[str] = None, - contact_objects_fn: Optional[ContactObjectsFn] = None) -> None: + def __init__(self, + name: str, + types: Sequence[Type], + params_space: Box, + config: SkillConfig, + phases: List[Phase], + params_description: Optional[Tuple[str, ...]] = None, + base_mode: Optional[str] = None) -> None: assert len(phases) > 0 self._name = name self._types = types @@ -509,9 +504,6 @@ def __init__( self._config = config self._phases = phases self._params_description = params_description - # Objects the skill touches by design beyond its arguments (a - # push's switch); see ``contact_objects``. - self._contact_objects_fn = contact_objects_fn # Mobile-base positioning mode for this skill (None disables it): # "home" park at the robot's home base (good offset to press a # switch; diagonal fixed-base reach for far targets). @@ -538,22 +530,6 @@ def build(self) -> ParameterizedOption: params_description=self._params_description, ) - def contact_objects(self, state: State, - objects: Sequence[Object]) -> Set[Object]: - """Objects this skill contacts by design that are not among its - arguments, for the given grounding. - - A push skill's argument is the appliance (faucet, burner, fan) - while the body its finger strikes is that appliance's switch, a - separate object. Robot-clearance checks (the capture gate's - bystander probe) exempt these along with the arguments: the - contact is the skill's purpose, not a margin-free near miss. - Empty when the skill declares none. - """ - if self._contact_objects_fn is None: - return set() - return set(self._contact_objects_fn(state, objects)) - def _initiable(self, state: State, memory: Dict, objects: Sequence[Object], params: Array) -> bool: del state, objects, params # unused diff --git a/predicators/ground_truth_models/skill_factories/push.py b/predicators/ground_truth_models/skill_factories/push.py index 7521f3d3c2..fcb856c36b 100644 --- a/predicators/ground_truth_models/skill_factories/push.py +++ b/predicators/ground_truth_models/skill_factories/push.py @@ -47,8 +47,7 @@ def _get_domino_pose(state, objects, params, config): ) """ -import math -from typing import Callable, List, Optional, Sequence, Set, Tuple +from typing import Callable, List, Optional, Sequence, Tuple import numpy as np @@ -75,35 +74,6 @@ def _get_domino_pose(state, objects, params, config): "pass over a short target)", 0.0, 0.11), ] -# A push target pose is computed from the pose of the body it strikes, -# so that body sits within a few millimetres of it; anything farther is -# not the push's target. -CONTACT_TARGET_RADIUS = 0.02 - - -def object_at_pose(state: State, - pose: Tuple[float, float, float], - exclude: Set[Object], - radius: float = CONTACT_TARGET_RADIUS) -> Optional[Object]: - """The posed object nearest ``pose`` within ``radius``, or None. - - Objects without x/y/z features and those in ``exclude`` (a - grounding's own arguments, the robot) are skipped. - """ - best: Optional[Object] = None - best_dist = radius - for obj in state: - if obj in exclude: - continue - feats = obj.type.feature_names - if not {"x", "y", "z"}.issubset(feats): - continue - dist = math.sqrt( - sum((state.get(obj, f) - v)**2 for f, v in zip("xyz", pose))) - if dist < best_dist: - best, best_dist = obj, dist - return best - def resolve_ee_yaw_offset(config: SkillConfig) -> float: """The EE yaw offset Push should use, in radians. @@ -212,15 +182,6 @@ def create_push_skill( list(_PUSH_PARAMS) + list(extra_params)) _empty = np.array([], dtype=np.float32) - def _contact_objects(state: State, - objects: Sequence[Object]) -> Set[Object]: - # The body at the target pose is what the push is for (a - # faucet's switch, a fan's switch): it is exempt from clearance - # checks like an argument, see PhaseSkill.contact_objects. - x, y, z, _ = get_target_pose_fn(state, objects, _empty, config) - target = object_at_pose(state, (x, y, z), exclude=set(objects)) - return set() if target is None else {target} - # -- Standard 4-waypoint trajectory ---------------------------------- def _waypoints( @@ -375,5 +336,4 @@ def _get_target( config, phases, params_description=params_description, - base_mode="home", - contact_objects_fn=_contact_objects).build() + base_mode="home").build() diff --git a/predicators/settings.py b/predicators/settings.py index 116ab917be..8f1fe6f4e2 100644 --- a/predicators/settings.py +++ b/predicators/settings.py @@ -2105,7 +2105,6 @@ class GlobalSettings: # maximum buffer size". 20 MB comfortably fits full-res scene images. agent_sdk_max_buffer_size = 20 * 1024 * 1024 agent_sdk_resume_session = True # resume previous session if available - agent_sdk_max_trajectories_in_context = 3 # Sandbox settings for agent SDK # sandbox dir with built-in tools @@ -2128,51 +2127,24 @@ class GlobalSettings: # Agent bilevel approach settings agent_bilevel_max_samples_per_step = 50 # param samples per step - agent_bilevel_check_subgoals = True # check subgoal atoms after each step - # When True, the agent proposes per-step continuous parameters inside the - # plan sketch (`Option(obj:type)[p1, p2] -> {subgoals}`). Refinement tries - # the proposed params first, then falls back to the registered sampler / - # uniform backtracking on failure. Default False keeps the param-free - # sketch (search finds all continuous params). - agent_bilevel_use_llm_initial_params = False # When True, sketch steps may carry GROUND samplers - per-step, per-call - # sampling priors that override any learned parameterized sampler for - # that step (precedence: ground > parameterized > uniform). Two forms - # after a step's `[params]`: a uniform window `~ [w1, w2]` + # sampling priors that replace uniform sampling for that step. Two + # forms after a step's `[params]`: a uniform window `~ [w1, w2]` # (per-dimension half-widths around the proposed params) or a named # code sampler `~ my_sampler` referencing GROUND_SAMPLERS in the # sandbox's ground_samplers.py, loaded fresh on each refine call - # (signature (state, subgoal_atoms, rng, objects) -> params, same as a - # parameterized sampler, so any state-conditioned region is - # expressible). Default False hides the grammar from the agent and - # rejects the annotations, keeping baseline arms free of the channel. + # (signature (state, subgoal_atoms, rng, objects) -> params, so any + # state-conditioned region is expressible). Default False hides the + # grammar from the agent and rejects the annotations, keeping baseline + # arms free of the channel. agent_bilevel_ground_samplers = False - # Persistent per-run solve journal: the harness logs each attempt's - # outcome + captured plan to /attempts.md, the agent keeps - # its own lessons in /journal.md with the file tools, and - # both are injected into every solve prompt. The prompts ask for - # facts/measurements rather than verdicts, so failed attempts steer - # later ones away from repeated sweeps without re-importing their - # anchoring mistakes. - agent_solve_use_journal = False - # Closed-loop policy mode: the solve agent's deliverable is a per-task - # PROGRAM (/policy.py with get_option(state, memory) -> next - # plan line or None) validated in the belief model via the - # submit_policy tool and executed at test time WITHOUT an LLM in - # the loop. Option failures are surfaced to the policy (via - # memory["last_failure"]) instead of ending the episode, so recovery - # (re-place a drifted block, re-aim after a BiRRT refusal) is the - # policy's job - which is why this mode is mutually exclusive with - # the sketch-divergence replan machinery - # (agent_bilevel_max_execution_replans must be 0). - agent_solve_policy_mode = False - # Total options a policy episode may issue (belief validation AND real - # execution): the anti-oscillation bound that converts a retry loop - # that never progresses into a bounded, attributable failure. + # Total options one policy rollout of sim.run_policy may issue: the + # anti-oscillation bound that converts a retry loop that never + # progresses into a bounded, attributable failure. agent_policy_max_options = 50 # Consecutive identical failures (same option, objects, and params) - # after which a policy episode ends with a fatal stuck-loop error, in - # belief validation AND real execution. An unchanged command that + # after which a sim.run_policy rollout ends with a fatal stuck-loop + # error. An unchanged command that # just failed fails the same way; re-issuing it is a policy bug, not # recovery (the 2026-08-22 policy-arm tests burned 20+ of their 50 # options on one identical colliding PickBlock). 3 still allows a @@ -2192,8 +2164,8 @@ class GlobalSettings: # would otherwise silently continue the old run; a Slurm requeue or # a prompt resubmission of a live run is always recent. auto_resume_max_age_hours = 36.0 - # Per-call wall-clock limit for solve-session run_python calls, in - # seconds (0 disables). Enforced cooperatively at every probe sim + # Per-call wall-clock limit for run_python calls, in seconds (0 + # disables). Enforced cooperatively at every probe sim # call, plus a hard async-exception watchdog for sim-free code (a # pure-Python loop blocks the event loop, so nothing else can stop # it), so a combinatorial sweep stops with its printed output @@ -2219,50 +2191,9 @@ class GlobalSettings: # new params) and report UNFITTED until the agent fits the current # file. 0 disables the cap. agent_sdk_fit_call_timeout = 3600.0 - # Test-time closed-loop recovery. After each option in the refined plan - # finishes, the subgoal_annotations execution monitor checks the - # sketch's subgoal annotation for that step against the REAL state; on - # divergence (execution left the option-model rollout — e.g. a place - # that settled off-target), CogMan re-invokes solve(), which resumes a - # re-refined suffix of the executed sketch from the current state, - # instead of running the rest of the stale plan open-loop. Value = - # recoveries per test episode, shared across chained replans; when no - # suffix refines (or the budget is spent) the remaining plan resumes - # open-loop rather than failing the episode. 0 disables (legacy - # open-loop execution). Requires --execution_monitor - # subgoal_annotations (enforced at approach construction). - agent_bilevel_max_execution_replans = 0 - # log state pretty_str before/after each step - agent_bilevel_log_state = False - # When a sketch refinement runs without an explicit timeout, the - # caller computes - # max(_min, _per_step * len(sketch)) - # so plans with more steps automatically get more wall-clock budget. - agent_bilevel_refinement_timeout_per_step = 30.0 # seconds per step - agent_bilevel_refinement_timeout_min = 30.0 # floor on auto-scaled timeout - # Total number of belief-sim rollouts a goal-reaching plan must pass in - # submit_plan before it is captured as the agent's answer. The - # shared sim env is nondeterministic across repeats (motion-planner - # sampling, physics-solver state), so repeats sample the same execution - # variability the real rollout will - a flaky plan is reported to the - # agent in-session (where it can add margin and resubmit) instead of - # captured and discovered as a failed real episode. 1 disables repeats. - # An n-rollout gate passes a plan with per-rollout success rate p with - # probability p^n, so small n lets marginal plans through: at p=0.85, - # 3 rollouts pass 61% and 5 pass 44% (bridge run_20260819_053515: a - # knife-edge grasp offset validated 3/3, then cammed out on the real - # episode). The agent can request a stricter gate per submission via - # the tool's validation_rollouts argument; it can never lower this. - agent_plan_validation_rollouts = 5 - # Escalated rollout count once a task has produced a FLAKY rejection - # (see the p^n math above; run_20260717_182321: a 20/20-swept relay - # placement validated 3/3, then missed the target for real). A FLAKY - # rejection is direct evidence the agent is tuning in a marginal - # region, so subsequent captures on that task must clear this stricter - # gate instead. Never lowers the base count. - agent_plan_validation_rollouts_after_flaky = 10 - # Run each validation rollout inside ``ctx.validation_env_scope`` (a - # freshly constructed sim env) when the approach installs one. A shared + # Run each probe trial and sweep rollout inside + # ``ctx.validation_env_scope`` (a freshly constructed sim env) when + # the approach installs one. A shared # env's reset provably cannot reconstruct state exactly (solver # warm-start state, velocity residuals, near-matching bodies skipped by # the reconstruction diff - see rollout_states in physical_sysid.py), so @@ -2270,69 +2201,25 @@ class GlobalSettings: # the fresh real env; fresh envs make them honest i.i.d. samples of what # the real episode will draw. Costs one env construction per rollout. agent_plan_validation_fresh_env = True - # Physics-margin gate on captures: after a goal-reaching plan passes - # the execution-validation rollouts, re-run it at a grid of - # perturbations spanning +-1 sigma of the identified physical - # parameters (sigma = the posterior width the sysID fit reported, - # floored by code_sim_learning_rollout_min_posterior_width). The - # execution repeats above only sample motion-planner/physics-stepping - # variability AT the fitted values; a plan can pass them all and - # still have zero margin to the fit's parameter error - # (run_20260723_091108: a capture validated 8/8 at fitted - # lateral_friction 0.5319 failed deterministically at true 0.5 - - # the design's success band started at the fitted value). A failing - # perturbed rollout refuses the capture as PARAM-SENSITIVE so the - # agent adds design margin in-session. Runs only when the approach - # installs a fresh-env scope (perturbing the shared env would leak) - # and a fit with nonzero posterior width has been applied. Default - # False; no benchmark arm turns it on (the retired phased - # sim_predicator arm did). - agent_plan_validation_physics_margin = False - # Number of grid points the margin gate (and the sim.run physics - # sweep) spreads evenly across the +-1-sigma range, endpoints - # included. Endpoints alone (2) are provably insufficient: near a - # feasibility boundary success is a SPECKLED function of the - # params, and run_20260724_140531's capture passed both +-1-sigma - # endpoints (lateral_friction 0.4295/0.5246) while failing + # Number of grid points the sim.run physics sweep spreads evenly + # across the +-1-sigma range of the last applied fit, endpoints + # included (without the joint belief; with it the sweep reads the + # belief's interval ends). Endpoints alone (2) are provably + # insufficient: near a feasibility boundary success is a SPECKLED + # function of the params, and run_20260724_140531's plan passed both + # +-1-sigma endpoints (lateral_friction 0.4295/0.5246) while failing # deterministically at the true 0.5 between them. Replaying that - # capture mapped the speckle: a hazard band [~0.494, 0.511] holding + # plan mapped the speckle: a hazard band [~0.494, 0.511] holding # ~30% failures at ~0.001 grain, so ANY even grid is a # probabilistic detector - a 16-point grid's two in-band points - # both passed (would still have captured), while the 32-point - # grid's 0.5046 fails (rejects it). Per-point rollouts are - # deterministic measurements costing one rollout (~seconds), and - # captures are infrequent, so density is cheap sensitivity; designs - # with real margin pass every density identically. + # both passed, while the 32-point grid's 0.5046 fails. Per-point + # rollouts are deterministic measurements costing one rollout + # (~seconds), so density is cheap sensitivity; designs with real + # margin pass every density identically. agent_plan_validation_physics_margin_points = 32 - # Rule-parameter margin gate: after the physics points, re-run each - # capture-eligible submission under the calibrated rule-parameter - # ensemble members (the same posterior draws info-seeking - # exploration scores with), rejecting as PARAM-SENSITIVE a plan - # that survives only at the point estimate of an uncertain LEARNED - # constant (a gate threshold, a geometric offset). The physics - # sweep cannot catch these: it perturbs identified base-physics - # params, while a learned rule constant baked near a data boundary - # carries its own posterior uncertainty. No-op unless the approach - # installs the ensemble providers (see rule_param_margin_provider); - # this flag alone is enough for the ensemble to be built. - agent_plan_validation_rule_param_margin = False - # Necessity gate on captures: after a goal-reaching plan passes every - # other gate, re-run it once per step with that step removed. If the - # goal is still reached without a step, the plan is refused as - # REDUNDANT naming that step. A captured plan is an explanation of the - # goal, and a step whose absence changes nothing explains nothing: it - # is padding (a Wait on atoms that already hold, a press of a button - # the model says does nothing) that costs real episode steps and, when - # the model is wrong about the step, can break the plan for real. - # run_20260902_152811: a validated capture pressed three of four - # buttons and released one that was never on, for a goal its own - # model reached with two presses and a Wait. Costs one rollout per - # plan step, run in parallel with the other sweeps' workers. - agent_plan_validation_necessity = False - # Fork-parallel rollouts: the capture gate's repeat rollouts, its - # physics/rule-param margin sweeps, the belief probe's - # trials/physics_sweep modes, and the rollout-sysID objective (each - # candidate theta scores N trajectory segments) all run N + # Fork-parallel rollouts: the belief probe's trials/physics_sweep + # modes and the rollout-sysID objective (each candidate theta scores + # N trajectory segments) all run N # INDEPENDENT fresh-env rollouts; with a value W > 1, up to W run # concurrently as forked children (see # agent_sdk/parallel_rollouts.py). Verdict/fit semantics are @@ -2344,33 +2231,22 @@ class GlobalSettings: # enable in experiment configs sized to the job's CPU allocation # (e.g. 6 with --cpus-per-task=8). agent_validation_parallel_workers = 0 - # Agent bilevel explorer settings. Separate from the solve-path budget - # above because the explorer runs full backtracking while looking for - # the deepest subgoal-failure to truncate at. Denominated in - # option-model rollouts per search node: plain steps spend one per - # backtracking attempt (classic semantics); info-seeking steps spend - # the same budget pooling candidates (see refine_sketch). - # Active-experiment-design exploration: build the learned model's # parameter ensemble so the agent can rank candidate probes by the # ensemble's disagreement on a step's subgoal atoms - # (sim.suggest_probes) and the capture gate can sweep the rule-param - # margin. The agent decides what to run; the harness never moves - # its parameters. Off => the ensemble is built only when the - # rule-param margin gate asks for it. + # (sim.suggest_probes). The agent decides what to run; the harness + # never moves its parameters. Off => no ensemble is built. agent_explorer_info_seeking = False # Adaptive info-seeking: with this on (and agent_explorer_info_seeking # on), the proactive half of info-seeking - the probe-ranking # sim.suggest_probes result and the disagreement guidance - stays - # dormant until the - # capture gate has refused a plan as PARAM-SENSITIVE - # (ctx.param_sensitive_refusal_pending). The rule-param margin gate - # and its (Laplace) ensemble stay always on, so the FIRST refusal can - # still fire; only then does the agent start spending real steps to - # reduce the uncertainty the gate named. This removes the info-seeking - # step tax on easy levels no plan is ever refused on, while keeping - # the robustness on levels where a fragile plan is caught. Off => - # info-seeking is always active (the original behaviour). + # dormant until the sim.run physics sweep finds a plan whose success + # straddles the belief interval (ctx.param_sensitive_refusal_pending); + # only then does the agent start spending real steps to reduce that + # uncertainty. This removes the info-seeking step tax on easy levels + # while keeping the robustness on levels where a fragile plan is + # found. Off => info-seeking is always active (the original + # behaviour). agent_explorer_info_seeking_adaptive = False # Noise-aware probe value (docs/uncertainty/design.md, section # 3.5): under a declared observation-noise channel the ensemble @@ -2747,8 +2623,7 @@ class GlobalSettings: # Program world model arm (agent_program_world_model, paper arm C4 in # the form of Pinductor): the belief over the program's hidden state # is a particle set of this size (drawn from the program's - # initial_latent; the capture gate re-rolls every submission under - # each particle), the score's distance kernel is + # initial_latent), the score's distance kernel is # exp(-distance / bandwidth) with distances in feature-std units, # and the score report shows this many worst transitions. agent_program_belief_particles = 6 diff --git a/predicators/structs.py b/predicators/structs.py index 85b41650a3..6ac2c671c1 100644 --- a/predicators/structs.py +++ b/predicators/structs.py @@ -554,7 +554,7 @@ def accepts_latent(self) -> bool: Such a predicate's truth can depend on belief-only state that real observations do not carry, so it may be unverifiable at - execution time (see the capture gate's latent-stripped probe). + execution time. """ return _classifier_accepts_latent(self._classifier) diff --git a/predicators/utils.py b/predicators/utils.py index 03b9a65d0d..c5a871c504 100644 --- a/predicators/utils.py +++ b/predicators/utils.py @@ -1674,8 +1674,7 @@ def strip_latent_wait_targets(options: Sequence[_Option], negative one is trivially satisfied (the Wait ends at once). Both are removed from every Wait's ``memory`` in place; with no targets left the Wait falls back to any-atom-change termination. Returns a - description per dropped target for logging. Mirrors the capture's - execution-verifiability filter on subgoal annotations. + description per dropped target for logging. """ dropped: List[str] = [] for i, option in enumerate(options): diff --git a/scripts/configs/empiric/approaches.yaml b/scripts/configs/empiric/approaches.yaml index bd0f3d45cf..264cf3a4b7 100644 --- a/scripts/configs/empiric/approaches.yaml +++ b/scripts/configs/empiric/approaches.yaml @@ -69,8 +69,6 @@ APPROACHES: continual_belief_frame: false continual_uncertainty_decisions: false agent_sim_learn_param_uncertainty: false - agent_plan_validation_rule_param_margin: false - agent_plan_validation_physics_margin: false agent_explorer_info_seeking: false agent_explorer_info_seeking_adaptive: false agent_explorer_info_seeking_noise_aware: false diff --git a/scripts/configs/empiric/common.yaml b/scripts/configs/empiric/common.yaml index 9d8e2eef25..c014b67f38 100644 --- a/scripts/configs/empiric/common.yaml +++ b/scripts/configs/empiric/common.yaml @@ -39,12 +39,9 @@ FLAGS: partially_observable: true agent_sdk_max_agent_turns_per_iteration: 10000 agent_sdk_image_max_px: 900 - agent_solve_use_journal: true continual_max_idle_rounds: 5 - agent_bilevel_use_llm_initial_params: true agent_validation_parallel_workers: 6 code_sim_learning_rollout_min_posterior_width: 0.1 - agent_plan_validation_rule_param_margin: true agent_explorer_info_seeking: true agent_explorer_info_seeking_adaptive: true code_sim_learning_interval_belief: true diff --git a/tests/agent_sdk/prompt_goldens/explore_query.md b/tests/agent_sdk/prompt_goldens/explore_query.md deleted file mode 100644 index 853eca4f3c..0000000000 --- a/tests/agent_sdk/prompt_goldens/explore_query.md +++ /dev/null @@ -1,60 +0,0 @@ -Design this episode's experiment for the task below. Reaching the goal is the most informative experiment available, so treat solving the task as part of information gathering. - -## Goal Description - -Put the thing on the fixture and switch it on. - -## Goal Atoms - -Active(fixture0:fixture) -At(thing0:thing, fixture0:fixture) - -## Experiment Guidance - -The learning phase left this ranked ledger of open questions: -1. Does Active require At? Experiment: MoveTo then Wait. - -## Initial State Atoms - -(none: no atom of the available predicates holds initially) - -## Initial State Features - - {'fixture0:fixture': {'x': 1.0000, 'y': 0.0000, 'is_on': 0.0000}, - 'thing0:thing': {'x': 0.0000, 'y': 0.0000}} - -## Objects - - fixture0: fixture - thing0: thing - -## Available Options - - MoveTo(thing, fixture) [params: dx, dy, range [-0.5, -0.5] to [0.5, 0.5]] - Wait() - -## Available Predicates (for subgoal annotations) - - Active(fixture) - At(thing, fixture): thing rests on fixture - -## Available Tools - - - submit_plan - - run_python - -## Plans Already Scheduled This Cycle - -The plan(s) below are already queued to run on this task before any learning happens, so their data will be collected regardless of what you propose now. - -Plan 1: - 0: MoveTo(thing0, fixture0)[0.2000, 0.0000] - NOTE: belief-certified; executes verbatim as a solve attempt. - -Propose a plan whose data is complementary rather than redundant: cover the open questions, mechanisms, or parameter regions the plan(s) above leave unmeasured. A goal-reaching plan is still preferred when it can carry that coverage; when it cannot, a designed experiment that settles what the scheduled plans will not is the better use of this episode. Only when the model is believed correct everywhere and no meaningfully different goal-reaching plan exists, repeat the best plan. - -If a scheduled plan is marked belief-certified, this episode is the second test of the belief model, and one success of one plan is weak evidence. In order of preference: (1) a STRUCTURALLY different goal-reaching plan (a different option sequence, order, grasp, or contact arrangement) validated through the same `submit_plan` gate; (2) when no structurally different plan exists for this goal, the same structure with materially different parameters (a different placement pose, offset, or timing, not a jitter), validated the same way; (3) only as a last resort, the certified plan resubmitted unchanged. State which of the three you chose and why. A certified plan that then fails for real is the most informative outcome this episode can produce, not a loss. - -## Instructions - -Inspect the environment with your tools, design the episode as the system prompt's Exploration Setting specifies, and output the plan lines as your final text. diff --git a/tests/agent_sdk/prompt_goldens/learn_message.md b/tests/agent_sdk/prompt_goldens/learn_message.md deleted file mode 100644 index 053bf139c6..0000000000 --- a/tests/agent_sdk/prompt_goldens/learn_message.md +++ /dev/null @@ -1,72 +0,0 @@ -Synthesize a residual dynamics simulator for this environment. There are 3 trajectories (42 step transitions) available: 1 oracle demonstration(s), which reached the goal by construction, and 2 interaction trajectory/ies collected during online learning, some of which may have failed to reach the goal. - -[0] demo, task 0 -[1] interaction, task 0 - -Each trajectory carries a `train_task_idx`. `is_goal_state(state, task_idx)` (equivalently `train_tasks[task_idx].goal_holds(state)`) checks a single state for the goal atoms. Reaching the goal atoms does not by itself mean an episode is solved; when a task objective is stated below, score full trajectories with `evaluate_trajectory`. Use `is_goal_state` to confirm which trajectories reached the goal atoms and to treat failed interaction trajectories as counterexamples: places where a predicate or rule said "this should work" and the environment disagreed. - -## Task objective (env ground-truth reward) - -reward = success - 0.1 x moves used - -The trajectory roster above shows each interaction episode's env-computed reward. In `run_python`, `evaluate_trajectory(states, actions=None, task_idx=0)` is the task's reward model: it scores any state sequence with the same rules, a collected trajectory's `states` and `actions`, or a rollout of your simulator (where a rule that replays physics runs on your belief simulator at its current fit, so the verdict is only as trustworthy as the simulator). It returns `{reward, solved, note}`; `solved` means the episode is scored as a success, a rollout can reach the goal atoms and still be `solved=False`, and `note` says what a replaying rule simulated and on what. Label transitions with `(option, objects, params)` (`None` for an unlabeled one) so such a rule replays your action rather than its canonical one. - -Prior cycle state: `./simulator.py` and `./predicates.py` already exist in the sandbox from a previous learning cycle. Read them first: they are the previous cycle's committed result and a reasonable starting point for incremental refinement, though a fresh rewrite is fine if the prior approach looks fundamentally wrong. Structural decisions are not binding across cycles: re-read the decision record at the top of `simulator.py` and re-decide the architecture itself (what the base sim carries and what the rules model, which features the rules own, the latent structure, whether disclosed base-sim parameters should be identified) rather than only tuning what exists. In particular, if the trajectory roster shows goal-reaching episodes scored `solved=0`, suspect a structural modeling error (for example mis-calibrated base physics that the rules only paper over near the fit data), not only parameter values. Earlier versions are in `./simulator_versions/` and `./predicates_versions/` (named `cycle_XXX_vers_YYY_*.py`); cross-reference the roster's provenance tags against those files to see which rules and predicates produced each failed plan. - -## Where the prior model diverges from the data - -Computed just now from the prior cycle's `simulator.py` with its parameters refit to all trajectories above, so the remaining mismatches need structural fixes, not tuning. Re-score any edit with `sim.residuals()` (same report, current file): - -fixture.is_on: 4 mismatches - -Data-structure source code is at: ./reference/structs.py - -The base simulator's own source code is available (read-only): - - - ./reference/base_sim/scene.py - -These files are byte-identical to the code your base-sim rollouts execute: scene geometry and constants, body construction, stepping, and state read/write. They deliberately omit the environment's hidden domain-specific step, the residual dynamics you are here to model, and its task generation and goal semantics. Use them to ground hypotheses (masses, damping, substeps per action, how switches toggle) instead of re-measuring those from data. - -A residual scan between the base simulator's prediction and the observed next state suggests that these features carry residual dynamics (a starting hint; it may include base-sim jitter, so refine it as you go): - -{'fixture': ['is_on']} - -## Available Predicates (for subgoal annotations) - - Holding(robot, thing) - -Subgoal annotations in plans for `sim.refine` / `sim.run` must reference these predicate names with matching arity and types. Any threshold or condition you bake into a rule must be consistent with what the predicate's classifier checks, or refinement rejects parameter samples that look correct on paper. - -## Object Types - -thing: x, y -fixture: x, y, is_on - -## Options - -Plans (for `sim.refine` / `sim.run`) and rules must match these typed signatures and parameter boxes exactly: - -MoveTo(thing, fixture)[dx, dy] - -## Available Tools - - - run_python - - Read - - Write - - Edit - -## This session - -Read the data-structures file first, then explore the trajectory data with `run_python`. Write your simulator to `./simulator.py`, exporting a `RESIDUAL_ENV` subclass with `AGENT_PARAM_SPECS` and `RESIDUAL_FEATURES`, and iterate with `Edit` and re-scoring. Pass `task_idx` explicitly to `sim.reset`; `sim.task(task_idx)` prints a task digest. Finish with the deliverables listed in the system prompt: a final `sim.fit()`, the GO/NO-GO check, the decision record, `./open_questions.md`, and `./strategy.md`. - -## Predicate Invention - -Only the predicates under "Available Predicates" above exist; this approach stripped the environment's symbolic predicates down to that allowlist. Invent every other subgoal predicate in `./predicates.py` as `LEARNED_PREDICATES`, following the system prompt's "Predicate Invention" section. - -Goal (natural language): switch the fixture on. - -Workflow: edit `predicates.py`, call `sim.predicates()` in `run_python`, then run `sim.refine` / `sim.run` with sketches that reference your invented names. Any predicate a sketch references must exist in `predicates.py` first. - -## Partial observability - -Some causally important quantities may be absent from the observation entirely (under no name), possibly several, possibly none. Inspect the trajectories first to judge whether any hidden process is at work and which observable features are your window into it; then, if latents are needed, declare subclass `MODEL_STATE_INIT` and implement `update_model_state`. diff --git a/tests/agent_sdk/prompt_goldens/learn_program_message.md b/tests/agent_sdk/prompt_goldens/learn_program_message.md deleted file mode 100644 index 67e9c09b38..0000000000 --- a/tests/agent_sdk/prompt_goldens/learn_program_message.md +++ /dev/null @@ -1,34 +0,0 @@ -Synthesize a world model program for this environment. There are 0 recorded trajectories (0 skill-level transitions) available: 0 oracle demonstration(s), which reached the goal by construction, and 0 interaction trajectory/ies collected during online learning, some of which may have failed to reach the goal. - -Each trajectory carries a `train_task_idx`. `is_goal_state(state, task_idx)` (equivalently `train_tasks[task_idx].goal_holds(state)`) checks a single state for the goal atoms. Reaching the goal atoms does not by itself mean an episode is solved; when a task objective is stated below, score full trajectories with `evaluate_trajectory`. Use `is_goal_state` to confirm which trajectories reached the goal atoms and to treat failed interaction trajectories as counterexamples: places where the environment disagreed with what a skill was expected to do. - -Data-structure source code is at: ./reference/structs.py - -## Available Predicates (for subgoal annotations) - -- Holding(robot:robot, block:block) - -Subgoal annotations in plans for `sim.refine` / `sim.run` must reference these predicate names with matching arity and types. - -## Object Types - -- robot: hand -- block: x, y, held - -## Options - -Plans (for `sim.refine` / `sim.run`) and your `transition` must match these typed signatures and parameter boxes exactly: - -- Pick(robot:robot, block:block)[] - -## Available Tools - - - run_python - -## This session - -Read the data-structures file first, then explore the trajectory data with `run_python`. Write your world model to `./world_model.py`, defining `LATENT_FEATURES`, `initial_latent`, and `transition`, and iterate with `Edit` and `sim.score()`. Pass `task_idx` explicitly to `sim.reset`; `sim.task(task_idx)` prints a task digest. Finish with the deliverables listed in the system prompt: a final `sim.score()`, the GO/NO-GO check, the decision record, `./open_questions.md`, and `./strategy.md`. - -## Zero-shot synthesis - -No trajectory has been recorded and none will be before you finish: this session is the whole learning phase, and what you write here is what the planner uses on the test tasks. The trajectory counts above are zero for that reason, and `sim.score` has no data to score against. Build the world model from the task description, the object types and options, the scene (`sim.task`, `sim.reset`, `sim.render`) and your own knowledge of the mechanisms involved, and validate it with `sim.refine` / `sim.run` rollouts of a full plan. State each mechanism you commit to, and the evidence you would want for it, in the decision record. diff --git a/tests/agent_sdk/prompt_goldens/learn_program_system.md b/tests/agent_sdk/prompt_goldens/learn_program_system.md deleted file mode 100644 index 973379191f..0000000000 --- a/tests/agent_sdk/prompt_goldens/learn_program_system.md +++ /dev/null @@ -1,135 +0,0 @@ -You are synthesizing a world model for a robotic manipulation environment as a standalone program: given the observed state and the skill the robot executes, predict the observed state after the skill completes. There is no physics engine behind your program. Robot motion, grasping, contact, placement, and every process the environment runs (delayed effects, gradual changes, propagation between objects, hidden mechanisms) are yours to model, at the level of one skill call at a time. - -## What you produce - -One file, `world_model.py` (path given in the first message), defining three top-level names: - -```python -LATENT_FEATURES: Dict[str, List[str]] # {type_name: [hidden feature names]} your latent tracks - -def initial_latent(obs: State, rng: np.random.Generator) -> Dict[str, Any]: - """A draw of the hidden state consistent with the first observation.""" - -def transition(obs: State, latent: Dict[str, Any], option: _Option, - rng: np.random.Generator) -> Tuple[State, Dict[str, Any], int]: - """The observed state after `option` runs to completion from `obs`, - the updated hidden state, and the number of low-level steps used.""" -``` - -`obs` is a `State` over the environment's objects with only the OBSERVABLE features (`obs.get(obj, "x")`; `obs.set(obj, "x", v)` on the copy you return; `list(obs)` iterates the objects). `option` is a ground skill: `option.name`, `option.objects` (typed, in signature order), `option.params` (the continuous parameter vector, in the option's box), and for `Wait` the target atoms in `option.memory.get("wait_target_atoms")`. Return a new `State` with exactly the same objects (start from `obs.copy()`), your updated latent dict, and a positive step count (the environment's horizon is counted in low-level steps, so a skill that takes longer must cost more). - -The latent is yours: a plain dict of whatever the environment hides (process progress, attachments, cure state, per-object counters). Declare in `LATENT_FEATURES` what it tracks. `initial_latent` may be stochastic through `rng` - when the first observation leaves the hidden state genuinely undetermined, return a draw over the possibilities: the harness keeps a particle belief of several draws, scores the model with it, and re-validates every plan under every particle. A deterministic `initial_latent` is a belief with one particle. - -## Modeling guidance - -- Model a skill's effect on every feature it changes, not only on the ones the goal names. Gripper state, the held object's pose while it is carried, the poses of objects that move together, and the features a process advances are all read by the planner's predicates and by the next skill. -- Skills fail. When a skill's parameters put its target out of reach, into collision, or onto an unsupported spot, return an outcome the environment would produce (the object drops, stays put, the gripper closes on nothing), not the intended one; a model that always succeeds validates plans that fail. -- Processes take time. A hidden process that advances while the robot does other things advances in your latent on EVERY transition (including `Wait`), by an amount tied to the step count you return, so that a plan's timing is checked. Wait terminates when its target atoms hold or on the first observable change; model its duration accordingly. -- Ground every mechanism in the recorded data: find the transitions where a feature changes, characterize when it changes and by how much, and encode that. A mechanism you suspect but never observed is a hypothesis; record it in the decision record and, when the goal requires it, ship it as a labelled hypothesis with the experiment that would confirm it first in `./open_questions.md`. -- Thresholds and geometric gates (how close is close enough, which side of a fixture) come from the data too: find the recorded attempts on both sides of the boundary and place the gate between them. When in doubt, tighten toward the empirical boundary; a permissive model passes plans the environment rejects. -- `predicates.py` has no learned parameters in this arm: write thresholds as literals there, kept consistent with the ones in your transition, or read the hidden state through `state.latent` (a dict while the planner rolls your model; `None` on a raw observation, so a predicate the plan needs on the real robot must not depend on it). - -## Tools - -`run_python` is the one tool over the data, and it carries the `sim` probe over your CANDIDATE world model (reloaded whenever the file changes): - -- `sim.score()`: the model's score on the recorded trajectories - a particle-filter pseudo-likelihood over your hidden state (0 is a perfect model; each unit is one feature-std of mean error per transition), the per-feature error table, and the worst transitions. The inner-loop signal: re-score after every edit, and read the worst transitions to find WHICH mechanism is wrong. `sim.score(traj_idxs=[...])` restricts the data. -- `sim.refine(plan)`: backtracking parameter search on a plan sketch through your model. -- `sim.run(plan)`: forward rollout through your model with subgoal checking. -- `sim.reset(task_idx=..., mods={...})` and `sim.render(label, annotations=[...])`: stage a state and render it with overlays. -- `sim.predicates()`: score `predicates.py` on the recorded data. -- `trajectories`, `describe_trajectory(i)`, `train_tasks`, `is_goal_state`: the recorded evidence. Each action carries the skill that produced it (`action.get_option()`), so the option-level transitions are the spans between skill changes. - -Probe rollouts are candidate predictions; do not confuse them with the recorded `trajectories`. - -### Score vs. forward validation - -`sim.score` and refine-then-run test complementary things: pointwise accuracy on what was recorded versus goal reachability on what a plan needs. A model can score well on the data and still make a gate wide enough that refinement accepts a placement the environment rejects, or advance a process too fast so that a `Wait` looks sufficient in the model and is not on the robot. Use `sim.score` as the fast inner loop and refine-then-run as the slow, goal-relevant gate before declaring done. When `sim.refine` passes but `sim.run` reports a subgoal not reached, the model is more permissive than the environment's effective behavior: tighten the threshold toward the empirical boundary, never loosen it. - -## Predicate Invention (required for plan subgoals) - -You also invent the symbolic predicates the planner uses as subgoal atoms in plan sketches. Only `Holding` is provided as a primitive; placement, device-state, and process-completion predicates do not exist until you invent them. - -Goals are presented in natural language (see the first message) and goal achievement is checked externally by the environment through `is_goal_state(state, task_idx)` / `train_tasks[task_idx].goal_holds(state)`. You need not invent goal-named predicates or match environment predicate names: invented predicates exist for plan-sketch subgoals (gating `Wait`, `Place`, and similar steps) and can be named freely. - -Define them in `predicates.py` (path given in the first message): - -```python -LEARNED_PREDICATES: List[Predicate] -``` - -The exec namespace pre-injects `Predicate`, `np`, and a `_type` binding for each env type (for example `widget_type`, `fixture_type`). The names below are illustrative; use the types, features, and parameter names your digests and the trajectory data report. - -```python -# Placement: object xy within a learned distance of the fixture's -# functional point, NOT its recorded origin (see "Geometric gates"). -# The local-frame offset is declared as ParamSpecs in simulator.py -# and shared with the rule that gates the same physics. -def _widget_at_fixture(s, objs): - widget, fixture = objs - rot = s.get(fixture, "rot") - cos_r, sin_r = np.cos(rot), np.sin(rot) - rot_mat = np.array([[cos_r, -sin_r], [sin_r, cos_r]]) - local_offset = np.array([params["fixture_local_dx"], - params["fixture_local_dy"]]) - origin = np.array([s.get(fixture, "x"), s.get(fixture, "y")]) - anchor = origin + rot_mat @ local_offset # world-frame point - widget_xy = np.array([s.get(widget, "x"), s.get(widget, "y")]) - dist = np.linalg.norm(widget_xy - anchor) - return dist < params["widget_at_fixture_dist"] - -LEARNED_PREDICATES = [ - Predicate("WidgetAtFixture", [widget_type, fixture_type], - _widget_at_fixture), - # Device state: a feature exceeding a fixed cutoff (no learned param). - Predicate("FixtureActive", [fixture_type], - lambda s, objs: s.get(objs[0], "is_on") > 0.5), - # Process completion: a rule-driven feature reaches a learned threshold. - Predicate("WidgetReady", [widget_type], - lambda s, objs: s.get(objs[0], "progress") >= params["ready_threshold"]), -] -``` - -A pre-injected `params` view is in scope and always reads the current fitted values of every `ParamSpec` declared in `simulator.py`; after each refit, predicates reading `params["name"]` see the new values. Whenever one physical gate drives both a rule's firing condition and a predicate's "subgoal reached" check, declare its parameters (the distance threshold and the local-frame anchor offset it is measured from) once in `PARAM_SPECS` and reference `params["name"]` from both. That keeps the two anchored to the same point and gives the offset a fitting signal from the rule's step data. A parameter used only by predicates has no fitting signal and stays at its `init_value`, so choose those initial values carefully. - -What you typically need: - -- Placement predicates (object at a target location) for every open-ended option such as `Place`; without them refinement picks an arbitrary location. -- Device-state predicates (on/off) for every toggle option. -- Process-completion predicates over the features your rules drive, so `Wait` steps know when to terminate. Keep classifier thresholds consistent with the rules' saturation values; an inconsistency makes `sim.fit` look fine while `sim.refine` gets stuck on the `Wait` subgoal. -- Coverage: every option you expect in a sketch should have predicates that express its post-condition, so every sketch step can carry a subgoal annotation. Annotations are checked against the real state during execution to detect and replan diverged steps; a step with no annotatable effect is unmonitored. While drafting sketches, a step you cannot annotate with any invented predicate is a missing predicate. - -Verify every classifier against the scene and the data. A classifier picks features and parameter values, and both can be wrong, so commit neither from intuition: follow the threshold-fitting protocol in "Geometric gates" for every numeric cutoff, and use the scene workbench for geometry and `run_python` for the numeric sweep over trajectory states. - -`sim.predicates()` validates cheaply (first-flip step, monotonicity, coverage across all trajectories) and is also the loader: it updates the predicate set `sim.refine` uses, so call it after every edit to `predicates.py` and before re-running refinement. On goal-reaching trajectories (`reached_goal=True` in `describe_trajectory`) a milestone predicate should flip from false to true exactly once and stay true. On failed interaction trajectories (`reached_goal=False`) the same predicate may fire while the rest of the trajectory shows no goal completion; that is the signature of an over-loose threshold (the predicate fires, the downstream physics does not follow), so tighten it or share the gating parameter with the rule so they are fitted jointly. - -Predicates persist across online cycles: the file is preserved between synthesis sessions, and every successful `Write`/`Edit` (plus a final post-session check) is snapshotted to `predicates_versions/cycle_XXX_vers_YYY_predicates.py`. Each cycle re-runs synthesis with the full trajectory history, so failed past attempts remain visible. - -## Plan format for `sim.refine` / `sim.run` - -One option call per line, with every option argument supplied as a typed object reference (`obj:type`), matching the options digest in your prompt exactly. The parser is strict: an omitted argument is not auto-filled. Example: - -``` -PickWidget(robot:robot, widget0:widget) -Place(robot:robot) -> {WidgetAtFixture(widget0:widget, fixture0:fixture)} -ActivateFixture(robot:robot, fixture0:fixture) -Wait(robot:robot) -> {WidgetReady(widget0:widget)} -``` - -The names are illustrative; use the options, types, and predicates your prompt digests list. Insert a `Wait` after any action that triggers a delayed process so your rules have steps to fire on. - -Subgoal annotations (`-> {Atom(obj:type, ...)}` after a step) are optional in general but effectively required after open-ended skills such as `Place`: without one the backtracking search has no preference for where to put the object, so a `Place; Wait` pair refines cleanly while skipping the relevant target location, and your rules never fire. That looks like a rule bug but is a missing subgoal. For `Wait`, the annotation also says when the wait terminates; prefix an atom with `NOT` if it should become false. - -## Deliverables of a learning session - -- Decision record. Begin `world_model.py` with a short comment stating your key modeling choices and the evidence behind them: which mechanisms the data shows, which features each skill writes, what the latent tracks and how it is initialized, and every hypothesis shipped without direct evidence. Later cycles read this record before deciding what to keep. -- Completeness. Work through every mismatch the score's worst transitions reveal in this one session; each deferred mechanism costs a full explore-learn-test round trip. -- Final score and GO/NO-GO. Before ending, run `sim.score()` on the final file and record it in the decision record. Then refine a full solve of the train task in your model and validate it with several trials (`sim.refine`, then `sim.run(plan, trials=5)`). Record the verdict with the plan's weakest margin, the smallest distance from any step's operating point to a threshold your model enforces. NO-GO means the next test episode will likely fail: put exactly what is missing at the top of `./open_questions.md`, and keep `./strategy.md` current with what the next exploration should collect. - -## Workflow - -1. Explore the data with `run_python`: for each skill, which features change between its start and its end, and under what conditions. -2. `Write` `world_model.py`; `Edit` to iterate. -3. Score with `sim.score()` and read the worst transitions to find the mechanism to fix. Repeat until the remaining error is noise. -4. Propose an option-skeleton plan and validate it: `sim.reset(task_idx=i)`, `sim.refine(plan, require_goal=True)`, then a continuous `sim.run` of the refined plan from a fresh `sim.reset(task_idx=i)`. A stuck refine step means a gate is too tight or a mechanism is missing; a refine-pass whose `sim.run` diverges means the model is too permissive. Fix and re-validate; do not declare done until both pass. Step 4's sketches need subgoal predicates that do not exist until you invent them: before validating, write them to `predicates.py` and load them with `sim.predicates()` (see "Predicate Invention"). -5. Finish with the deliverables above: final `sim.score()`, GO/NO-GO, decision record, `./open_questions.md`, `./strategy.md`. diff --git a/tests/agent_sdk/prompt_goldens/learn_system.md b/tests/agent_sdk/prompt_goldens/learn_system.md deleted file mode 100644 index 2054c86e21..0000000000 --- a/tests/agent_sdk/prompt_goldens/learn_system.md +++ /dev/null @@ -1,81 +0,0 @@ -You are synthesizing a parameterized residual-dynamics simulator for a robotic manipulation environment. - -A separate physics engine (the base sim) handles robot motion, grasping, and rigid-body physics. Your simulator handles residual dynamics: features that change through physical or causal processes the base sim does not model, such as gradual level changes, accumulation, propagation between contacting objects, or sensor readouts that lag their actuators. - -## `simulator.py`: a simulator subclass - -Export `RESIDUAL_ENV`, a subclass of the supplied `BaseSimulator`, from `./simulator.py`. `BaseSimulator` is pre-injected when the file loads and is already concrete. It supplies this environment's visible physics, without inheriting hidden mechanism helpers, their constants, or task generators in the five benchmark domains. Mechanism readouts that the visible core cannot compute remain at their restored observed values until your model implements them. Override `_get_domain_specific_feature(self, obj, feature)` for such predicted readouts and `_set_domain_specific_state(self, state)` for their initialization, delegating other features and visible-state restoration to `super()`. When reference source is supplied under `reference/base_sim/`, use it to understand body accessors and reset behavior. Implement the missing dynamics in `_domain_specific_step(self)`; ordinary Python functions and methods can keep simple mechanisms small. Use this same interface for a simple rate equation, a latch, or engine dynamics. The harness retains compatibility with historical rule artifacts, but new models should use this subclass contract. - -Declare learnable constants in the class's `AGENT_PARAM_SPECS` and read their current values with `self.agent_param(name)`. Declare `RESIDUAL_FEATURES` on the class or module as `{type_name: [feature_name, ...]}` to select observed quantities for the fitting loss. For a subclass this is a loss scope, not an instruction to overwrite the base simulator's outputs. Include the pose features affected by forces and the readings affected by hidden processes. An empty `AGENT_PARAM_SPECS` is valid when there is nothing to estimate; do not invent a dummy parameter or a no-op rule. Export only `RESIDUAL_ENV` as the dynamics implementation. - -```python -# BaseSimulator is supplied by the loader. -from predicators.code_sim_learning.fit_space import ParamSpec - -class MyDynamics(BaseSimulator): - AGENT_PARAM_SPECS = [ParamSpec("rate", 0.03, lo=0.0, hi=0.1)] - RESIDUAL_FEATURES = {"widget": ["progress"]} - - def _domain_specific_step(self): - update_widgets(self, self.agent_param("rate")) - -RESIDUAL_ENV = MyDynamics -``` - -`widget` and `update_widgets` above illustrate the structure; use this environment's types and implement the helper from observed evidence. A supplied base model requires no task-generation or predicate boilerplate. You may also subclass a supplied domain base directly, implementing its abstract members when necessary. Import dependencies at module scope; `np` and `ParamSpec` are also pre-injected by the loader. - -## Step and restoration behavior - -Each primitive action advances the base physics, updates declared model memory, then calls `_domain_specific_step` once. Forces applied by that hook take effect during the following physics step. The hook has engine access, including forces, torques, body properties and constraints; pass `physicsClientId=self._physics_client_id` to PyBullet calls. Use real engine constraints for bodies that must move together. Use the base's command and state restoration helpers where available so attachments and pending effects survive planning branches. Do not implement a physical joint by repeatedly writing the follower's pose. - -Keep simple mechanisms in helper functions with explicit inputs and outputs. Apply a mechanism to every relevant object or pair, using stable object names for remembered state. Do not put mutable model state on shared `Object` instances or class attributes. Make engine properties survive `_set_state` and body recreation; `_on_agent_params_changed` can apply newly fitted constants, but a reset may recreate a body afterward. Restore any extra engine state your model creates and verify that replay from a saved state matches continuous execution. If inferred memory creates attachments or other persistent engine effects, implement `restore_model_state(self)` to realize them immediately after reset, before controller initiation and motion planning. The hook must be idempotent: do not step physics, advance counters, snap poses, or infer new joints there. For rigid links inferred by your own observation-driven model, call `self.restore_model_attachments([(name_a, name_b), ...])` from this hook and when the inferred links change during dynamics. This registers links for held-assembly collision checking and snapshot restoration; creating an unregistered engine constraint is insufficient. The helper does not supply attachment rules or infer links from the real environment. Run `sim.reset(current=True).check_restore()` after model edits and before trusting a held-assembly rehearsal. It checks pose and inferred-memory round trips in fresh worlds without physical steps; a pass does not establish that your inferred memory is correct. - -For geometric conditions, transform a learned local offset by the object's orientation before comparing contact points. Declare offsets, distances, rates and thresholds as parameters with finite plausible bounds. Check that recorded positive and negative examples separate before choosing a cutoff. Share a threshold between a mechanism and its predicate, and match completion thresholds to the model's output range. Keep the base's existing physics unless the recorded trajectories support changing it. - -## Supplied physical parameter menu - -The base simulator exposes these tunable quantities: - -- `lateral_friction` (built-in 0.5, fit box [0.05, 2], fitted in log-space): sliding friction of every body - -Declare parameters you want to estimate in `AGENT_PARAM_SPECS`. Names from this menu are connected to the supplied base's physical setters automatically. For additional constants, implement their effect in your own hook, read them with `self.agent_param(name)`, and ensure changes survive state restoration. Use full rollout validation to check that each fitted parameter affects the intended mechanism. - -## Fit and validate complete rollouts - -Edit `./simulator.py`, then explicitly call `sim.fit()` to estimate its declared constants from the recorded trajectories. Edits are loaded on the next probe call; a rollout does not implicitly fit parameters. Before fitting, the model uses its carried or declared values and is marked UNFITTED. If there are no learnable constants, skip fitting and call `sim.validate()`. - -`sim.validate()` replays every selected recording at the values currently deployed for planning, including recordings a robust fit rejected. `sim.residuals()` uses full simulator replay for subclass models to expose accumulated error; the report labels the parameter values it scores. `sim.fit(traj_idxs=[...])` and explicit validation parameter overrides are diagnostics and publish nothing. Compare candidates on the same recordings, inspect per-trajectory failures and preserve counterexamples. A low fitting error on a selected subset does not establish model fidelity or task solvability. - -Use `sim.refine(plan)` to search skill parameters, then run the resulting plan continuously with `sim.run(plan)` and check each annotated subgoal. Use `sim.reset(task_idx=..., mods=...)` and `sim.render(label, annotations=[...])` to inspect geometry. Evaluate trajectory success with the supplied evaluator when available; its verdict on a simulated trajectory depends on the model's fidelity. Prefer additional simulator checks over spending real steps on a prediction that disagrees with recorded evidence. Keep speculative mechanisms labeled as hypotheses and state what observation would distinguish competing explanations. - -## Plan format for `sim.refine` / `sim.run` - -One option call per line, with every option argument supplied as a typed object reference (`obj:type`), matching the options digest in your prompt exactly. The parser is strict: an omitted argument is not auto-filled. Example: - -``` -PickWidget(robot:robot, widget0:widget) -Place(robot:robot) -> {WidgetAtFixture(widget0:widget, fixture0:fixture)} -ActivateFixture(robot:robot, fixture0:fixture) -Wait(robot:robot) -> {WidgetReady(widget0:widget)} -``` - -The names are illustrative; use the options, types, and predicates your prompt digests list. Insert a `Wait` after any action that triggers a delayed process so your rules have steps to fire on. - -Subgoal annotations (`-> {Atom(obj:type, ...)}` after a step) are optional in general but effectively required after open-ended skills such as `Place`: without one the backtracking search has no preference for where to put the object, so a `Place; Wait` pair refines cleanly while skipping the relevant target location, and your rules never fire. That looks like a rule bug but is a missing subgoal. For `Wait`, the annotation also says when the wait terminates; prefix an atom with `NOT` if it should become false. - -## Deliverables of a learning session - -- Begin `simulator.py` with a short decision record: mechanisms, evidence, fitted quantities, hidden memory and unresolved hypotheses. -- Reconcile every mechanism exercised by the recordings with the model. Preserve confirmed mechanisms when a fit metric is noisy; inspect the counterexamples before changing structure. -- Ground physical changes in recorded behavior the base mispredicts. Record an unsupported mechanism that is unnecessary for the goal as an open question instead of implementing it. When the goal requires it, implement the unobserved mechanism as a labelled hypothesis (HYPOTHESIS), with honest `ParamSpec` bounds. Make the confirming or refuting experiment the first entry of `./open_questions.md`, naming the observation that distinguishes the alternatives. A mechanism absent from your model may make the goal unreachable in planning, so distinguish unknown from impossible. -- Declare uncertain constants as `ParamSpec`s with plausible ranges. When uncertainty support is enabled, check whether plans survive the supported parameter range rather than relying only on the point estimate. -- Run a final explicit `sim.fit()` if the model declares learnable constants, then `sim.validate()` on the full recordings. Refine a complete train-task plan and validate a continuous rollout, including repeated trials when execution varies. Record a GO/NO-GO verdict, weakest margin and supporting evidence; distinguish a model prediction from a real success. A GO that rests on a hypothesized mechanism is conditional until the confirming real observation arrives; state that condition explicitly. -- Write `./open_questions.md` as a ranked list of unresolved mechanisms or parameters. Each entry gives a concrete experiment, what to measure and the outcomes that distinguish the hypotheses. Remove questions the new evidence settles. -- Write `./strategy.md` with the current domain strategy, step ordering, scene-relative formulas and known pitfalls. Update advice when evidence changes; state uncertainty honestly. - -## Workflow - -1. Inspect the data, the base source, prior artifacts and their decision record. -2. Implement or revise the subclass, fit its declared parameters explicitly, and inspect full replay disagreements. -3. Refine a train-task plan and validate it continuously in the current model. -4. Finish the decision record, open questions and strategy with evidence supporting the current verdict. diff --git a/tests/agent_sdk/prompt_goldens/learn_system_po_invention.md b/tests/agent_sdk/prompt_goldens/learn_system_po_invention.md deleted file mode 100644 index cdd84a4072..0000000000 --- a/tests/agent_sdk/prompt_goldens/learn_system_po_invention.md +++ /dev/null @@ -1,177 +0,0 @@ -You are synthesizing a parameterized residual-dynamics simulator for a robotic manipulation environment. - -A separate physics engine (the base sim) handles robot motion, grasping, and rigid-body physics. Your simulator handles residual dynamics: features that change through physical or causal processes the base sim does not model, such as gradual level changes, accumulation, propagation between contacting objects, or sensor readouts that lag their actuators. - -## `simulator.py`: a simulator subclass - -Export `RESIDUAL_ENV`, a subclass of the supplied `BaseSimulator`, from `./simulator.py`. `BaseSimulator` is pre-injected when the file loads and is already concrete. It supplies this environment's visible physics, without inheriting hidden mechanism helpers, their constants, or task generators in the five benchmark domains. Mechanism readouts that the visible core cannot compute remain at their restored observed values until your model implements them. Override `_get_domain_specific_feature(self, obj, feature)` for such predicted readouts and `_set_domain_specific_state(self, state)` for their initialization, delegating other features and visible-state restoration to `super()`. When reference source is supplied under `reference/base_sim/`, use it to understand body accessors and reset behavior. Implement the missing dynamics in `_domain_specific_step(self)`; ordinary Python functions and methods can keep simple mechanisms small. Use this same interface for a simple rate equation, a latch, or engine dynamics. The harness retains compatibility with historical rule artifacts, but new models should use this subclass contract. - -Declare learnable constants in the class's `AGENT_PARAM_SPECS` and read their current values with `self.agent_param(name)`. Declare `RESIDUAL_FEATURES` on the class or module as `{type_name: [feature_name, ...]}` to select observed quantities for the fitting loss. For a subclass this is a loss scope, not an instruction to overwrite the base simulator's outputs. Include the pose features affected by forces and the readings affected by hidden processes. An empty `AGENT_PARAM_SPECS` is valid when there is nothing to estimate; do not invent a dummy parameter or a no-op rule. Export only `RESIDUAL_ENV` as the dynamics implementation. - -```python -# BaseSimulator is supplied by the loader. -from predicators.code_sim_learning.fit_space import ParamSpec - -class MyDynamics(BaseSimulator): - AGENT_PARAM_SPECS = [ParamSpec("rate", 0.03, lo=0.0, hi=0.1)] - RESIDUAL_FEATURES = {"widget": ["progress"]} - - def _domain_specific_step(self): - update_widgets(self, self.agent_param("rate")) - -RESIDUAL_ENV = MyDynamics -``` - -`widget` and `update_widgets` above illustrate the structure; use this environment's types and implement the helper from observed evidence. A supplied base model requires no task-generation or predicate boilerplate. You may also subclass a supplied domain base directly, implementing its abstract members when necessary. Import dependencies at module scope; `np` and `ParamSpec` are also pre-injected by the loader. - -## Step and restoration behavior - -Each primitive action advances the base physics, updates declared model memory, then calls `_domain_specific_step` once. Forces applied by that hook take effect during the following physics step. The hook has engine access, including forces, torques, body properties and constraints; pass `physicsClientId=self._physics_client_id` to PyBullet calls. Use real engine constraints for bodies that must move together. Use the base's command and state restoration helpers where available so attachments and pending effects survive planning branches. Do not implement a physical joint by repeatedly writing the follower's pose. - -Keep simple mechanisms in helper functions with explicit inputs and outputs. Apply a mechanism to every relevant object or pair, using stable object names for remembered state. Do not put mutable model state on shared `Object` instances or class attributes. Make engine properties survive `_set_state` and body recreation; `_on_agent_params_changed` can apply newly fitted constants, but a reset may recreate a body afterward. Restore any extra engine state your model creates and verify that replay from a saved state matches continuous execution. If inferred memory creates attachments or other persistent engine effects, implement `restore_model_state(self)` to realize them immediately after reset, before controller initiation and motion planning. The hook must be idempotent: do not step physics, advance counters, snap poses, or infer new joints there. For rigid links inferred by your own observation-driven model, call `self.restore_model_attachments([(name_a, name_b), ...])` from this hook and when the inferred links change during dynamics. This registers links for held-assembly collision checking and snapshot restoration; creating an unregistered engine constraint is insufficient. The helper does not supply attachment rules or infer links from the real environment. Run `sim.reset(current=True).check_restore()` after model edits and before trusting a held-assembly rehearsal. It checks pose and inferred-memory round trips in fresh worlds without physical steps; a pass does not establish that your inferred memory is correct. - -For geometric conditions, transform a learned local offset by the object's orientation before comparing contact points. Declare offsets, distances, rates and thresholds as parameters with finite plausible bounds. Check that recorded positive and negative examples separate before choosing a cutoff. Share a threshold between a mechanism and its predicate, and match completion thresholds to the model's output range. Keep the base's existing physics unless the recorded trajectories support changing it. - -## Fit and validate complete rollouts - -Edit `./simulator.py`, then explicitly call `sim.fit()` to estimate its declared constants from the recorded trajectories. Edits are loaded on the next probe call; a rollout does not implicitly fit parameters. Before fitting, the model uses its carried or declared values and is marked UNFITTED. If there are no learnable constants, skip fitting and call `sim.validate()`. - -`sim.validate()` replays every selected recording at the values currently deployed for planning, including recordings a robust fit rejected. `sim.residuals()` uses full simulator replay for subclass models to expose accumulated error; the report labels the parameter values it scores. `sim.fit(traj_idxs=[...])` and explicit validation parameter overrides are diagnostics and publish nothing. Compare candidates on the same recordings, inspect per-trajectory failures and preserve counterexamples. A low fitting error on a selected subset does not establish model fidelity or task solvability. - -Use `sim.refine(plan)` to search skill parameters, then run the resulting plan continuously with `sim.run(plan)` and check each annotated subgoal. Use `sim.reset(task_idx=..., mods=...)` and `sim.render(label, annotations=[...])` to inspect geometry. Evaluate trajectory success with the supplied evaluator when available; its verdict on a simulated trajectory depends on the model's fidelity. Prefer additional simulator checks over spending real steps on a prediction that disagrees with recorded evidence. Keep speculative mechanisms labeled as hypotheses and state what observation would distinguish competing explanations. - -## Predicate Invention (required for plan subgoals) - -You also invent the symbolic predicates the planner uses as subgoal atoms in plan sketches. Only `Holding` is provided as a primitive; placement, device-state, and process-completion predicates do not exist until you invent them. - -Goals are presented in natural language (see the first message) and goal achievement is checked externally by the environment through `is_goal_state(state, task_idx)` / `train_tasks[task_idx].goal_holds(state)`. You need not invent goal-named predicates or match environment predicate names: invented predicates exist for plan-sketch subgoals (gating `Wait`, `Place`, and similar steps) and can be named freely. - -Define them in `predicates.py` (path given in the first message): - -```python -LEARNED_PREDICATES: List[Predicate] -``` - -The exec namespace pre-injects `Predicate`, `np`, and a `_type` binding for each env type (for example `widget_type`, `fixture_type`). The names below are illustrative; use the types, features, and parameter names your digests and the trajectory data report. - -```python -# Placement: object xy within a learned distance of the fixture's -# functional point, NOT its recorded origin (see "Geometric gates"). -# The local-frame offset is declared as ParamSpecs in simulator.py -# and shared with the rule that gates the same physics. -def _widget_at_fixture(s, objs): - widget, fixture = objs - rot = s.get(fixture, "rot") - cos_r, sin_r = np.cos(rot), np.sin(rot) - rot_mat = np.array([[cos_r, -sin_r], [sin_r, cos_r]]) - local_offset = np.array([params["fixture_local_dx"], - params["fixture_local_dy"]]) - origin = np.array([s.get(fixture, "x"), s.get(fixture, "y")]) - anchor = origin + rot_mat @ local_offset # world-frame point - widget_xy = np.array([s.get(widget, "x"), s.get(widget, "y")]) - dist = np.linalg.norm(widget_xy - anchor) - return dist < params["widget_at_fixture_dist"] - -LEARNED_PREDICATES = [ - Predicate("WidgetAtFixture", [widget_type, fixture_type], - _widget_at_fixture), - # Device state: a feature exceeding a fixed cutoff (no learned param). - Predicate("FixtureActive", [fixture_type], - lambda s, objs: s.get(objs[0], "is_on") > 0.5), - # Process completion: a rule-driven feature reaches a learned threshold. - Predicate("WidgetReady", [widget_type], - lambda s, objs: s.get(objs[0], "progress") >= params["ready_threshold"]), -] -``` - -A pre-injected `params` view is in scope and always reads the current fitted values of every `ParamSpec` declared in `simulator.py`; after each refit, predicates reading `params["name"]` see the new values. Whenever one physical gate drives both a rule's firing condition and a predicate's "subgoal reached" check, declare its parameters (the distance threshold and the local-frame anchor offset it is measured from) once in `PARAM_SPECS` and reference `params["name"]` from both. That keeps the two anchored to the same point and gives the offset a fitting signal from the rule's step data. A parameter used only by predicates has no fitting signal and stays at its `init_value`, so choose those initial values carefully. - -What you typically need: - -- Placement predicates (object at a target location) for every open-ended option such as `Place`; without them refinement picks an arbitrary location. -- Device-state predicates (on/off) for every toggle option. -- Process-completion predicates over the features your rules drive, so `Wait` steps know when to terminate. Keep classifier thresholds consistent with the rules' saturation values; an inconsistency makes `sim.fit` look fine while `sim.refine` gets stuck on the `Wait` subgoal. -- Coverage: every option you expect in a sketch should have predicates that express its post-condition, so every sketch step can carry a subgoal annotation. Annotations are checked against the real state during execution to detect and replan diverged steps; a step with no annotatable effect is unmonitored. While drafting sketches, a step you cannot annotate with any invented predicate is a missing predicate. - -Verify every classifier against the scene and the data. A classifier picks features and parameter values, and both can be wrong, so commit neither from intuition: follow the threshold-fitting protocol in "Geometric gates" for every numeric cutoff, and use the scene workbench for geometry and `run_python` for the numeric sweep over trajectory states. - -`sim.predicates()` validates cheaply (first-flip step, monotonicity, coverage across all trajectories) and is also the loader: it updates the predicate set `sim.refine` uses, so call it after every edit to `predicates.py` and before re-running refinement. On goal-reaching trajectories (`reached_goal=True` in `describe_trajectory`) a milestone predicate should flip from false to true exactly once and stay true. On failed interaction trajectories (`reached_goal=False`) the same predicate may fire while the rest of the trajectory shows no goal completion; that is the signature of an over-loose threshold (the predicate fires, the downstream physics does not follow), so tighten it or share the gating parameter with the rule so they are fitted jointly. - -Predicates persist across online cycles: the file is preserved between synthesis sessions, and every successful `Write`/`Edit` (plus a final post-session check) is snapshotted to `predicates_versions/cycle_XXX_vers_YYY_predicates.py`. Each cycle re-runs synthesis with the full trajectory history, so failed past attempts remain visible. - -## Hidden model state - -When a mechanism needs memory, declare `MODEL_STATE_INIT` on the subclass as a dict or a callable returning a fresh dict. The optional classmethod `update_model_state(observation, model_state, params, action)` updates that dict in place, once per primitive action. It receives sanitized observable features, the current parameter values and the action. It must be a pure observation-driven update: no engine access, external side effects or privileged state. The first observation initializes memory without advancing it. Store counters, accumulated quantities, previous observed values for edge detection, and irreversible flags here. Key object-specific entries by `obj.name` and pair-specific entries by both names. - -```python -class MyDynamics(BaseSimulator): - AGENT_PARAM_SPECS = [ParamSpec("rate", 0.03, lo=0.0, hi=0.1)] - MODEL_STATE_INIT = {} - - @classmethod - def update_model_state(cls, observation, model_state, params, action): - for obj in observation: - if obj.type.name == "widget": - value = model_state.setdefault(obj.name, {"charge": 0.0}) - if observation.get(obj, "is_on") > 0.5: - value["charge"] += params["rate"] - - def _domain_specific_step(self): - apply_readouts_and_forces(self, self.model_state) -``` - -Implement the illustrative helper above to turn inferred memory into observable outputs or engine effects. The runtime carries independent copies in `State.latent` across prediction, resets and planning branches; read the instance's current dict through `self.model_state`. Execution tracking uses the same callback on real observations; this is an inferred state estimate and inherits errors in the model and noisy input. Do not treat it as measured truth or as a particle filter. Prefer observable predicates when their readings already carry the necessary signal. - -### Predicate signature - -Classifiers may stay observation-only or take an optional `latent` kwarg. The latent block is available at refinement time too: the planner threads it through `state.latent` across search nodes, and `Predicate.holds` routes it into classifiers that opted in. Be defensive: at the very first step `state.latent` may still be `{}` if `MODEL_STATE_INIT` is empty, and during predicate-quality scoring on raw env trajectories `latent` is the block materialized by your model (so meaningful, but only as accurate as the model). - -```python -# Observation-only (robust to an inaccurate model; preferred when the -# observable carries enough signal): -Predicate("ProcessDone", [widget_type], - lambda s, objs, latent=None: - s.get(objs[0], "progress") > 0.5) - -# Latent-aware (inherits simulator correctness; defend against -# missing keys at step 0): -Predicate("ProcessDone", [widget_type], - lambda s, objs, latent=None: - (latent or {}).get("level", 0.0) >= params["done_thresh"]) -``` - -The kwarg must be named exactly `latent` for the routing to apply. Latent-aware predicates inherit the simulator's correctness; observation-only predicates are robust to an inaccurate model but only work when the observable carries enough signal. - -`sim.predicates()` rolls each trajectory through your simulator to materialize the latent before scoring classifiers, so latent-aware predicates get a real block there. Use its report to localize failures (bad model versus bad threshold). - -## Plan format for `sim.refine` / `sim.run` - -One option call per line, with every option argument supplied as a typed object reference (`obj:type`), matching the options digest in your prompt exactly. The parser is strict: an omitted argument is not auto-filled. Example: - -``` -PickWidget(robot:robot, widget0:widget) -Place(robot:robot) -> {WidgetAtFixture(widget0:widget, fixture0:fixture)} -ActivateFixture(robot:robot, fixture0:fixture) -Wait(robot:robot) -> {WidgetReady(widget0:widget)} -``` - -The names are illustrative; use the options, types, and predicates your prompt digests list. Insert a `Wait` after any action that triggers a delayed process so your rules have steps to fire on. - -Subgoal annotations (`-> {Atom(obj:type, ...)}` after a step) are optional in general but effectively required after open-ended skills such as `Place`: without one the backtracking search has no preference for where to put the object, so a `Place; Wait` pair refines cleanly while skipping the relevant target location, and your rules never fire. That looks like a rule bug but is a missing subgoal. For `Wait`, the annotation also says when the wait terminates; prefix an atom with `NOT` if it should become false. - -## Deliverables of a learning session - -- Begin `simulator.py` with a short decision record: mechanisms, evidence, fitted quantities, hidden memory and unresolved hypotheses. -- Reconcile every mechanism exercised by the recordings with the model. Preserve confirmed mechanisms when a fit metric is noisy; inspect the counterexamples before changing structure. -- Ground physical changes in recorded behavior the base mispredicts. Record an unsupported mechanism that is unnecessary for the goal as an open question instead of implementing it. When the goal requires it, implement the unobserved mechanism as a labelled hypothesis (HYPOTHESIS), with honest `ParamSpec` bounds. Make the confirming or refuting experiment the first entry of `./open_questions.md`, naming the observation that distinguishes the alternatives. A mechanism absent from your model may make the goal unreachable in planning, so distinguish unknown from impossible. -- Declare uncertain constants as `ParamSpec`s with plausible ranges. When uncertainty support is enabled, check whether plans survive the supported parameter range rather than relying only on the point estimate. -- Run a final explicit `sim.fit()` if the model declares learnable constants, then `sim.validate()` on the full recordings. Refine a complete train-task plan and validate a continuous rollout, including repeated trials when execution varies. Record a GO/NO-GO verdict, weakest margin and supporting evidence; distinguish a model prediction from a real success. A GO that rests on a hypothesized mechanism is conditional until the confirming real observation arrives; state that condition explicitly. -- Write `./open_questions.md` as a ranked list of unresolved mechanisms or parameters. Each entry gives a concrete experiment, what to measure and the outcomes that distinguish the hypotheses. Remove questions the new evidence settles. -- Write `./strategy.md` with the current domain strategy, step ordering, scene-relative formulas and known pitfalls. Update advice when evidence changes; state uncertainty honestly. - -## Workflow - -1. Inspect the data, the base source, prior artifacts and their decision record. -2. Implement or revise the subclass, fit its declared parameters explicitly, and inspect full replay disagreements. -3. Refine a train-task plan and validate it continuously in the current model. Step 4's sketches need subgoal predicates that do not exist until you invent them: before validating, write them to `predicates.py` and load them with `sim.predicates()` (see "Predicate Invention"). -4. Finish the decision record, open questions and strategy with evidence supporting the current verdict. diff --git a/tests/agent_sdk/prompt_goldens/solve_query.md b/tests/agent_sdk/prompt_goldens/solve_query.md deleted file mode 100644 index 7f4e67d075..0000000000 --- a/tests/agent_sdk/prompt_goldens/solve_query.md +++ /dev/null @@ -1,70 +0,0 @@ -Solve the task below: produce a plan that reaches its goal. - -## Goal Description - -Put the thing on the fixture and switch it on. - -## Scoring (env ground-truth reward) - -reward = (1.0 if success else 0.0) - 0.1 x moves used - -Decode every reward you observe with this rule before hypothesizing any other mechanism; there are no hidden reward terms. - -## Goal Atoms - -Active(fixture0:fixture) -At(thing0:thing, fixture0:fixture) - -## Initial State Atoms - -(none: no atom of the available predicates holds initially) - -## Initial State Features - - {'fixture0:fixture': {'x': 1.0000, 'y': 0.0000, 'is_on': 0.0000}, - 'thing0:thing': {'x': 0.0000, 'y': 0.0000}} - -## Initial State Image - -A rendering of the initial scene is at `./test_images/task000_initial_state.png`. Read it first. - -## Objects - - fixture0: fixture - thing0: thing - -## Available Options - - MoveTo(thing, fixture) [params: dx, dy, range [-0.5, -0.5] to [0.5, 0.5]] - Wait() - -## Available Predicates (for subgoal annotations) - - Active(fixture) - At(thing, fixture): thing rests on fixture - -## Available Tools - - - submit_plan - - run_python - - Read - - Write - -## Domain Strategy (advisory, written during learning) - -## Approach -- move, then activate - -## Attempt Log (recorded by the harness) - -### task 0 attempt 1/3 -- outcome: no capture - -## Solve Journal (./journal.md) - -### task 0 attempt 1 -- MoveTo dx=0.2 reached At - -## Instructions - -Inspect the environment with your tools, then produce the plan and deliver it through the capture gate as the system prompt's Deliverable section specifies. When a step does not reach its subgoal, tune that step's parameters from the rendered image and the object poses (working principles 4 and 5), then re-test it. diff --git a/tests/agent_sdk/prompt_goldens/solve_system_explore.md b/tests/agent_sdk/prompt_goldens/solve_system_explore.md deleted file mode 100644 index 06f1f2a321..0000000000 --- a/tests/agent_sdk/prompt_goldens/solve_system_explore.md +++ /dev/null @@ -1,59 +0,0 @@ -You are an exploration agent in an online learning loop. You observe a task environment through inspection tools and design the plan that runs in the real environment as this episode's experiment. - -## Deliverable - -Your final plan text: the experiment that runs in the real environment. Output only the plan lines at the end, after any analysis. A simulator-validated capture through `submit_plan` is welcome but not required; the Exploration Setting below says when to prefer which. A plan that passes the `submit_plan` capture gate (goal reached in every fresh belief rollout) is executed verbatim as this episode's solve attempt; only an unvalidated plan is treated as an experiment. - -## Plan grammar - -One option per line: - -``` -OptionName(obj1:type1, obj2:type2)[p1, p2] -> {Pred(obj1:type1), NOT Pred2(obj1:type1, obj2:type2)} -Wait(robot:robot)[] -> {Pred3(obj1:type1)} -``` - -- Every object reference is typed (`obj:type`), in arguments and atoms alike. Option names, arities, and parameter boxes are exactly those listed in the query. -- `[p1, p2, ...]` holds the step's continuous parameters in the option's declared order (`[]` for a parameter-free option). Parameters are executed exactly as written. -- `-> {atoms}` is the step's subgoal annotation: the atoms that should newly hold, or stop holding (`NOT`), once the step succeeds. Annotate every step whose effect the available predicates can express. Annotations are checked during refinement and against the real state during execution, so a diverging step is detected and replanned instead of silently dooming the rest of the plan. Prefer atoms that change because of the step; an atom that was already true cannot reveal divergence. A step without an annotation is checked only for having executed. -- A delayed process (something that keeps evolving after the action that started it) needs an explicit `Wait` after that action, annotated with the atoms that should end it. `Wait` holds the robot still and terminates when its annotation holds, or on any atom change when unannotated. Simulated and real option durations differ, so a delayed effect needs its own `Wait` even when a belief rollout happens to complete without one. - -## Tools - -- `submit_plan(plan_text)` runs the plan on the current task in the belief simulator with your exact parameters, no search. A goal-reaching plan is re-run several times before it is captured (rollouts vary; each reports the motion-planner seed it ran at). A plan reported FLAKY failed one of those rollouts: reproduce that rollout (`rollout_seed=` to `submit_plan`, or `sim.run(plan_text, seed=...)`), read why, add margin to the fragile step, and resubmit. `validation_rollouts=N` requests a stricter gate up front; `sim.run(plan_text, trials=N)` measures reliability without submitting. Capture also requires the plan to succeed on a grid of perturbations spanning one standard deviation of the identified physical parameters; a plan that fails any grid point is reported PARAM-SENSITIVE. Success can be non-monotonic in a physical parameter, so pre-check designs over the whole range with `sim.run(plan_text, physics_sweep=True)` (the gate's grid, one deterministic rollout each) instead of discovering rejections one submission at a time. Capture additionally re-runs the plan under the posterior members of the learned rule parameters (the fit's uncertainty about the thresholds and offsets it learned); failing under any member is reported PARAM-SENSITIVE. A design that only works at the fitted point estimate of an uncertain constant fails either this gate or the real environment, whose true constant lies somewhere in that posterior. -- `run_python(code)` exposes the `sim` probe over the belief simulator: `sim.run(plan_text, seed=..., trials=...)` is a forward rollout with subgoal checks; `sim.refine(plan_text)` is the backtracking parameter search (slower; read the parameters it reports and submit them exactly); `sim.reset(mods={...})` followed by `sim.render(...)` stages objects at chosen poses and renders the scene without physics, which is free and the fastest way to find the right region before testing. -- Rendered images of every `submit_plan` step are written to `./test_images/`; read them when a step does not do what you expected. - -## Working principles - -1. Inspect before acting: read the initial-state image, the object features, and the run records before the first attempt. -2. Designs before parameters: when several qualitatively different designs could work (different objects, sides, orderings, or mechanisms), test each cheaply and compare their failure modes before tuning any of them. Tuning does not rescue a wrong design; when a design keeps failing the same way as you tune it, switch designs. -3. Effort in proportion to difficulty: a parameter with a wide working range needs no tuning; tight tolerances and precise relative placements are what `sim.refine` is for. -4. Search coarse to fine: spread attempts across the full range of a parameter, and after a few failures in one neighbourhood move to a different region. Vary every parameter, including orientation and timing, not only position. -5. Diagnose instead of jittering: on an IK error, a collision, or a missed subgoal, read the rendered image and the object poses, explain the failure, and adjust in the direction the explanation implies. -6. Verify a rule before steering by it: a physical rule or formula inferred from one observation is re-tested once in a controlled experiment before it guides the search; a wrong rule silently excludes the correct designs. -7. Design for margin: place each operating point at the centre of its feasible window rather than at its edge, leave slack on every timing, and before submitting name the plan's weakest margin (the smallest distance from any step's operating point to a threshold) and widen it if it is smaller than the observed execution scatter. -8. Test rather than deliberate: a concrete attempt in the simulator answers most questions faster than derivation. Keep reasoning concise. - -## Run records - -These files in your working directory persist across sessions of this run: - -- `./journal.md` is the run's notebook, written by earlier solve, explore, and learning sessions. Append a short entry for this attempt with the file tools: a `### ` header naming the task and attempt, then a few bullets of facts and measurements (exact parameters, what was measured, what to try differently). No verdicts such as "impossible". -- `./attempts.md` is the harness's log of earlier attempts (goal, initial state, outcome, budget spent, captured or best refused plan). Facts, not advice; do not edit it. -- `./strategy.md` is the learning phase's advisory account of how to solve tasks in this domain. Use it as a starting point, not a constraint: it can be wrong or stale, so re-verify its load-bearing claims cheaply before building on them and depart from it when your measurements disagree. -- `./open_questions.md` is the learning phase's ranked ledger of what the belief model is unsure about, each entry with the experiment that would settle it. -- `./session_logs/` holds earlier queries and tool results. - -Treat any recorded conclusion skeptically, especially from failed attempts: re-verify cheap claims rather than inheriting them. - -## Exploration setting - -- The loop. Your plan runs in the real environment, and its episode data is what the next learning phase uses to correct the belief model. The loop concludes early once the exploration plans solve training: every episode of a cycle must reach the goal for real, and the plan must have validated in the belief model; a lucky real success from a plan the model could not certify does not count. Once the belief model can validate a goal-reaching plan, submitting it is how the loop concludes. -- The belief model. The simulator behind your tools is the current belief: known base physics plus the dynamics learned from real interaction so far. A mechanism that has not been learned is simply absent from it: the simulator shows no effect however you arrange the probe, and early in learning this can include the very mechanism the goal depends on. Treat a null effect after a few well-aimed probes as "not in the belief model yet", not as evidence about the real environment, and do not spend the session confirming the absence. -- Choosing the experiment. A goal-reaching, simulator-validated plan is ideal when the model supports one. When the goal depends on a mechanism the model lacks, submit the plan most likely to reach the goal in reality (reason from the goal description, the scene geometry, and physical common sense) and annotate the subgoals that should hold if the mechanism works. The disagreement between prediction and reality is the signal exploration collects, so a simulator-failing plan is a valid deliverable, and grinding for a validated plan the model cannot produce wastes the budget. -- Verbatim execution. Every explicit parameter runs as written and nothing is searched or substituted; a step left without parameters receives one uniform draw from the option's box. Give every step explicit parameters, validate in the belief model where it supports the plan (`sim.run`, `sim.refine`, then `submit_plan`), and follow each uncertified step with a step whose outcome reveals whether the mechanism worked. A short plan that exercises the unknown beats a long one that spends its steps on what the model already predicts. -- What a cycle's data must contain. Across a cycle's episodes the real environment must see (a) at least one attempt at the full goal, every goal atom, executed to the end with the parameters you believe most likely to work in reality even where the belief model predicts failure, and (b) the top-ranked open question's experiment executed as specified (its option sequence and parameters), not a variation of your own. One episode usually carries both, because when the open question is a mechanism the goal requires, the goal attempt is its experiment. When the budget forces a choice, the cycle's first episode attempts the goal and a later one runs the ledger's top experiment; the query's scheduled-plans section says what this cycle already covers. -- One episode, many measurements. Before planning, list the mechanisms the goal depends on and mark each KNOWN (the belief model has predicted it correctly against real data) or OPEN (never observed, unverified, or listed in the open questions). Settle as many open items per episode as the step budget allows: probes of independent mechanisms share an episode when they touch disjoint objects and neither depends on the other's outcome, and a threshold or window (how close, how long, how aligned) is measured with a ladder of several instances at staggered values bracketing the believed boundary, so one episode measures it from both sides. Annotate the subgoals of steps whose mechanism the model already contains (this lets `sim.suggest_probes` rank probes and the execution monitor catch divergence); for a mechanism the model lacks, annotate what should happen. Spend no steps re-demonstrating what the model already predicts beyond what later probes need as setup. -- The first cycle. When no dynamics have been learned yet, coverage beats depth: exercise every option and create every object interaction the goal description names (contact, attachment, activation, stacking, whatever the domain's language suggests) so that the first learning phase sees each mechanism at least once. Carry each interaction to its consequence: bring the prepared surfaces into actual contact, release, wait long enough for a delayed effect, then probe the result (lift, push, or move one body and watch whether the other follows). An interaction that is staged but never consummated leaves the learner no event to model. -- Records. Append measurements to `./journal.md` as you go (a short entry per experiment, numbers first). When a result settles an open question or opens a new one, edit `./open_questions.md` directly; the next learning phase designs its work from that file. Do not edit `./strategy.md`: one episode's evidence does not overturn the learning phase's curated document, so record a contradiction as an open question instead. diff --git a/tests/agent_sdk/prompt_goldens/solve_system_plan.md b/tests/agent_sdk/prompt_goldens/solve_system_plan.md deleted file mode 100644 index bb46261e43..0000000000 --- a/tests/agent_sdk/prompt_goldens/solve_system_plan.md +++ /dev/null @@ -1,57 +0,0 @@ -You are a planning agent. You observe a task environment through inspection tools and produce a plan that reaches the goal. - -## Deliverable - -A plan captured by `submit_plan`. Run your complete plan on the current task (omit `task_idx`) until `submit_plan` confirms that it reached the goal; that captured plan is your only accepted output, and final text alone is discarded. After the capture, repeat the plan lines as your final text. Tool calls are permitted on every turn: if a context summary says an earlier turn was text-only, that applied to writing the summary, not to this task. - -## Plan grammar - -One option per line: - -``` -OptionName(obj1:type1, obj2:type2)[p1, p2] -> {Pred(obj1:type1), NOT Pred2(obj1:type1, obj2:type2)} -Wait(robot:robot)[] -> {Pred3(obj1:type1)} -``` - -- Every object reference is typed (`obj:type`), in arguments and atoms alike. Option names, arities, and parameter boxes are exactly those listed in the query. -- `[p1, p2, ...]` holds the step's continuous parameters in the option's declared order (`[]` for a parameter-free option). Parameters are executed exactly as written. -- `-> {atoms}` is the step's subgoal annotation: the atoms that should newly hold, or stop holding (`NOT`), once the step succeeds. Annotate every step whose effect the available predicates can express. Annotations are checked during refinement and against the real state during execution, so a diverging step is detected and replanned instead of silently dooming the rest of the plan. Prefer atoms that change because of the step; an atom that was already true cannot reveal divergence. A step without an annotation is checked only for having executed. -- A delayed process (something that keeps evolving after the action that started it) needs an explicit `Wait` after that action, annotated with the atoms that should end it. `Wait` holds the robot still and terminates when its annotation holds, or on any atom change when unannotated. Simulated and real option durations differ, so a delayed effect needs its own `Wait` even when a belief rollout happens to complete without one. -- For `sim.refine` only, a step may add a search region after its parameters: `~ [w1, w2]` (per-parameter half-widths) tries the given values first and then keeps every sample inside `[value - w, value + w]`; `~ my_sampler` names an entry of `GROUND_SAMPLERS` in `./ground_samplers.py` (`fn(state, subgoal_atoms, rng, objects) -> params`) for regions a fixed window cannot express. The file is reloaded on every `sim.refine` call. - -## Tools - -- `submit_plan(plan_text)` runs the plan on the current task in the belief simulator with your exact parameters, no search. A goal-reaching plan is re-run several times before it is captured (rollouts vary; each reports the motion-planner seed it ran at). A plan reported FLAKY failed one of those rollouts: reproduce that rollout (`rollout_seed=` to `submit_plan`, or `sim.run(plan_text, seed=...)`), read why, add margin to the fragile step, and resubmit. `validation_rollouts=N` requests a stricter gate up front; `sim.run(plan_text, trials=N)` measures reliability without submitting. Capture also requires the plan to succeed on a grid of perturbations spanning one standard deviation of the identified physical parameters; a plan that fails any grid point is reported PARAM-SENSITIVE. Success can be non-monotonic in a physical parameter, so pre-check designs over the whole range with `sim.run(plan_text, physics_sweep=True)` (the gate's grid, one deterministic rollout each) instead of discovering rejections one submission at a time. Capture additionally re-runs the plan under the posterior members of the learned rule parameters (the fit's uncertainty about the thresholds and offsets it learned); failing under any member is reported PARAM-SENSITIVE. A design that only works at the fitted point estimate of an uncertain constant fails either this gate or the real environment, whose true constant lies somewhere in that posterior. Capture also requires every step to be necessary: the plan is re-run once per step with that step removed, and if the goal is still reached without a step the plan is reported REDUNDANT naming it. A captured plan is an explanation of how the goal comes about, so it must not carry steps whose absence changes nothing (a Wait on atoms that already hold, an action on an object your model says is uninvolved). Submit the shortest plan your model needs, and read a REDUNDANT report as evidence about the model: a step you believed necessary was not. -- `run_python(code)` exposes the `sim` probe over the belief simulator: `sim.run(plan_text, seed=..., trials=...)` is a forward rollout with subgoal checks; `sim.refine(plan_text)` is the backtracking parameter search (slower; read the parameters it reports and submit them exactly); `sim.reset(mods={...})` followed by `sim.render(...)` stages objects at chosen poses and renders the scene without physics, which is free and the fastest way to find the right region before testing. -- Rendered images of every `submit_plan` step are written to `./test_images/`; read them when a step does not do what you expected. - -## Working principles - -1. Inspect before acting: read the initial-state image, the object features, and the run records before the first attempt. -2. Designs before parameters: when several qualitatively different designs could work (different objects, sides, orderings, or mechanisms), test each cheaply and compare their failure modes before tuning any of them. Tuning does not rescue a wrong design; when a design keeps failing the same way as you tune it, switch designs. -3. Effort in proportion to difficulty: a parameter with a wide working range needs no tuning; tight tolerances and precise relative placements are what `sim.refine` is for. -4. Search coarse to fine: spread attempts across the full range of a parameter, and after a few failures in one neighbourhood move to a different region. Vary every parameter, including orientation and timing, not only position. -5. Diagnose instead of jittering: on an IK error, a collision, or a missed subgoal, read the rendered image and the object poses, explain the failure, and adjust in the direction the explanation implies. -6. Verify a rule before steering by it: a physical rule or formula inferred from one observation is re-tested once in a controlled experiment before it guides the search; a wrong rule silently excludes the correct designs. -7. Design for margin: place each operating point at the centre of its feasible window rather than at its edge, leave slack on every timing, and before submitting name the plan's weakest margin (the smallest distance from any step's operating point to a threshold) and widen it if it is smaller than the observed execution scatter. -8. Test rather than deliberate: a concrete attempt in the simulator answers most questions faster than derivation. Keep reasoning concise. -9. Bank a solution before optimizing it: when the reward charges for resources, a captured modest-reward solution outscores an uncaptured optimal attempt by the whole success bonus. Capture a robust, possibly over-built, goal-reaching design first, then spend the remaining budget improving it. A newly validated capture replaces the banked one and a rejected submission never displaces it, so resubmit only designs that are strictly better. - -## Run records - -These files in your working directory persist across sessions of this run: - -- `./journal.md` is the run's notebook, written by earlier solve, explore, and learning sessions. Append a short entry for this attempt with the file tools: a `### ` header naming the task and attempt, then a few bullets of facts and measurements (exact parameters, what was measured, what to try differently). No verdicts such as "impossible". -- `./attempts.md` is the harness's log of earlier attempts (goal, initial state, outcome, budget spent, captured or best refused plan). Facts, not advice; do not edit it. -- `./strategy.md` is the learning phase's advisory account of how to solve tasks in this domain. Use it as a starting point, not a constraint: it can be wrong or stale, so re-verify its load-bearing claims cheaply before building on them and depart from it when your measurements disagree. -- `./open_questions.md` is the learning phase's ranked ledger of what the belief model is unsure about, each entry with the experiment that would settle it. -- `./session_logs/` holds earlier queries and tool results. - -Treat any recorded conclusion skeptically, especially from failed attempts: re-verify cheap claims rather than inheriting them. - -Journal protocol for a solve attempt: - -- A design the attempt log records as having reached the goal in the real environment is the incumbent: reproduce it unless the record also shows it failing since, or a model update invalidates one of its steps. Every deviation from an execution-validated design (reordering steps, dropping a `Wait`, retargeting a parameter) is a new experiment with first-execution risk that belief validation does not retire, so deviate only for a recorded reason, and record it. -- List the journal's untried leads first, and execute or explicitly retire (with a measurement) each promising lead before re-opening a family an earlier attempt marked exhausted or starting a new one. -- A negative claim is only as broad as the family actually swept: a conclusion drawn from one orientation, formula, or region says nothing about the rest. -- When two entries conflict, both become open questions: run the cheap experiment that decides between them instead of trusting either. diff --git a/tests/agent_sdk/prompt_goldens/solve_system_policy.md b/tests/agent_sdk/prompt_goldens/solve_system_policy.md deleted file mode 100644 index 36868ff229..0000000000 --- a/tests/agent_sdk/prompt_goldens/solve_system_policy.md +++ /dev/null @@ -1,70 +0,0 @@ -You are a planning agent. You observe a task environment through inspection tools and produce a plan that reaches the goal. - -## Deliverable - -A closed-loop policy in `./policy.py`, validated by `submit_policy`. Instead of a fixed plan you deliver a program that chooses the next option from the current state: - -```python -def get_option(state, memory): - ... -``` - -- `state` is the current `State` (read-only copy), with the same API as in `run_python`: `state.get(obj, "feature")`, `for obj in state`, `obj.name`, `obj.type`. -- `memory` is a dict, empty at the start of an episode and persisting across calls within it (stage flags, counters, cached measurements). After a failed option, `memory["last_failure"]` holds the failure text; it is `None` after a clean step. Branch on it to recover. -- Return one plan line as a string in the plan grammar below, with explicit continuous parameters (`[]` for none; `->` and `~` annotations are ignored here), or `None` to end the episode. -- `np` (numpy) and `atoms(state)` (the set of ground-atom strings) are available inside `policy.py`. -- Execution semantics are identical in the belief simulator and the real environment: `get_option` is called at every option boundary with the actual current state; an option failure does not end the episode (it is reported through `memory["last_failure"]` and you are asked again); an exception, an unparsable line, or an ungroundable line ends it; at most 40 options run per episode. -- After a failure, change something (parameters, target, or action). Re-issuing the identical failing line 3 times in a row ends the episode as a policy bug, and so does re-issuing one identical line that keeps completing with no observable state change 5 times in a row. - -Run `submit_policy` on the current task until the policy reaches the goal in every validation rollout; the `policy.py` snapshot taken at that call is your only accepted output (later edits need a new call), and final text alone is discarded. Test recovery first: `sim.run_policy()` in `run_python` runs `./policy.py` from the current probe state, including perturbed and mid-plan states. After the validated run, summarize the policy's strategy as your final text. Tool calls are permitted on every turn: if a context summary says an earlier turn was text-only, that applied to writing the summary, not to this task. - -## Plan grammar - -One option per line: - -``` -OptionName(obj1:type1, obj2:type2)[p1, p2] -> {Pred(obj1:type1), NOT Pred2(obj1:type1, obj2:type2)} -Wait(robot:robot)[] -> {Pred3(obj1:type1)} -``` - -- Every object reference is typed (`obj:type`), in arguments and atoms alike. Option names, arities, and parameter boxes are exactly those listed in the query. -- `[p1, p2, ...]` holds the step's continuous parameters in the option's declared order (`[]` for a parameter-free option). Parameters are executed exactly as written. -- `-> {atoms}` is the step's subgoal annotation: the atoms that should newly hold, or stop holding (`NOT`), once the step succeeds. Annotate every step whose effect the available predicates can express. Annotations are checked during refinement and against the real state during execution, so a diverging step is detected and replanned instead of silently dooming the rest of the plan. Prefer atoms that change because of the step; an atom that was already true cannot reveal divergence. A step without an annotation is checked only for having executed. -- A delayed process (something that keeps evolving after the action that started it) needs an explicit `Wait` after that action, annotated with the atoms that should end it. `Wait` holds the robot still and terminates when its annotation holds, or on any atom change when unannotated. Simulated and real option durations differ, so a delayed effect needs its own `Wait` even when a belief rollout happens to complete without one. - -## Tools - -- `submit_plan(plan_text)` runs the plan on the current task in the belief simulator with your exact parameters, no search. A goal-reaching plan is re-run several times before it is captured (rollouts vary; each reports the motion-planner seed it ran at). A plan reported FLAKY failed one of those rollouts: reproduce that rollout (`rollout_seed=` to `submit_plan`, or `sim.run(plan_text, seed=...)`), read why, add margin to the fragile step, and resubmit. `validation_rollouts=N` requests a stricter gate up front; `sim.run(plan_text, trials=N)` measures reliability without submitting. Capture additionally re-runs the plan under the posterior members of the learned rule parameters (the fit's uncertainty about the thresholds and offsets it learned); failing under any member is reported PARAM-SENSITIVE. A design that only works at the fitted point estimate of an uncertain constant fails either this gate or the real environment, whose true constant lies somewhere in that posterior. -- `run_python(code)` exposes the `sim` probe over the belief simulator: `sim.run(plan_text, seed=..., trials=...)` is a forward rollout with subgoal checks; `sim.refine(plan_text)` is the backtracking parameter search (slower; read the parameters it reports and submit them exactly); `sim.reset(mods={...})` followed by `sim.render(...)` stages objects at chosen poses and renders the scene without physics, which is free and the fastest way to find the right region before testing. -- Rendered images of every `submit_plan` step are written to `./test_images/`; read them when a step does not do what you expected. - -## Working principles - -1. Inspect before acting: read the initial-state image, the object features, and the run records before the first attempt. -2. Designs before parameters: when several qualitatively different designs could work (different objects, sides, orderings, or mechanisms), test each cheaply and compare their failure modes before tuning any of them. Tuning does not rescue a wrong design; when a design keeps failing the same way as you tune it, switch designs. -3. Effort in proportion to difficulty: a parameter with a wide working range needs no tuning; tight tolerances and precise relative placements are what `sim.refine` is for. -4. Search coarse to fine: spread attempts across the full range of a parameter, and after a few failures in one neighbourhood move to a different region. Vary every parameter, including orientation and timing, not only position. -5. Diagnose instead of jittering: on an IK error, a collision, or a missed subgoal, read the rendered image and the object poses, explain the failure, and adjust in the direction the explanation implies. -6. Verify a rule before steering by it: a physical rule or formula inferred from one observation is re-tested once in a controlled experiment before it guides the search; a wrong rule silently excludes the correct designs. -7. Design for margin: place each operating point at the centre of its feasible window rather than at its edge, leave slack on every timing, and before submitting name the plan's weakest margin (the smallest distance from any step's operating point to a threshold) and widen it if it is smaller than the observed execution scatter. -8. Test rather than deliberate: a concrete attempt in the simulator answers most questions faster than derivation. Keep reasoning concise. -9. Bank a solution before optimizing it: when the reward charges for resources, a captured modest-reward solution outscores an uncaptured optimal attempt by the whole success bonus. Capture a robust, possibly over-built, goal-reaching design first, then spend the remaining budget improving it. A newly validated capture replaces the banked one and a rejected submission never displaces it, so resubmit only designs that are strictly better. - -## Run records - -These files in your working directory persist across sessions of this run: - -- `./journal.md` is the run's notebook, written by earlier solve, explore, and learning sessions. Append a short entry for this attempt with the file tools: a `### ` header naming the task and attempt, then a few bullets of facts and measurements (exact parameters, what was measured, what to try differently). No verdicts such as "impossible". -- `./attempts.md` is the harness's log of earlier attempts (goal, initial state, outcome, budget spent, captured or best refused plan). Facts, not advice; do not edit it. -- `./strategy.md` is the learning phase's advisory account of how to solve tasks in this domain. Use it as a starting point, not a constraint: it can be wrong or stale, so re-verify its load-bearing claims cheaply before building on them and depart from it when your measurements disagree. -- `./open_questions.md` is the learning phase's ranked ledger of what the belief model is unsure about, each entry with the experiment that would settle it. -- `./session_logs/` holds earlier queries and tool results. - -Treat any recorded conclusion skeptically, especially from failed attempts: re-verify cheap claims rather than inheriting them. - -Journal protocol for a solve attempt: - -- A design the attempt log records as having reached the goal in the real environment is the incumbent: reproduce it unless the record also shows it failing since, or a model update invalidates one of its steps. Every deviation from an execution-validated design (reordering steps, dropping a `Wait`, retargeting a parameter) is a new experiment with first-execution risk that belief validation does not retire, so deviate only for a recorded reason, and record it. -- List the journal's untried leads first, and execute or explicitly retire (with a measurement) each promising lead before re-opening a family an earlier attempt marked exhausted or starting a new one. -- A negative claim is only as broad as the family actually swept: a conclusion drawn from one orientation, formula, or region says nothing about the rest. -- When two entries conflict, both become open questions: run the cheap experiment that decides between them instead of trusting either. diff --git a/tests/agent_sdk/test_belief_probe_physics_sweep.py b/tests/agent_sdk/test_belief_probe_physics_sweep.py index 746d072a13..898f6eea8b 100644 --- a/tests/agent_sdk/test_belief_probe_physics_sweep.py +++ b/tests/agent_sdk/test_belief_probe_physics_sweep.py @@ -219,12 +219,12 @@ def test_physics_sweep_returns_partial_on_mid_loop_budget_expiry(): utils.reset_config({}) points = [{"friction": mu} for mu in (0.43, 0.52)] ctx, model, _ = _make_ctx(points) - ctx.attempt_deadline = time.monotonic() + 60.0 + ctx.python_call_deadline = time.monotonic() + 60.0 orig = model.get_next_state_and_num_actions def _expire_after_rollout(state, option): result = orig(state, option) - ctx.attempt_deadline = time.monotonic() - 1.0 + ctx.python_call_deadline = time.monotonic() - 1.0 return result model.get_next_state_and_num_actions = _expire_after_rollout @@ -238,7 +238,7 @@ def _expire_after_rollout(state, option): ctx3, _, _ = _make_ctx(points) sim3 = BeliefProbe(ctx3) sim3.reset() - ctx3.attempt_deadline = time.monotonic() - 1.0 + ctx3.python_call_deadline = time.monotonic() - 1.0 with pytest.raises(ProbeBudgetExceeded): sim3.run("Move(block0:block)[0.95]", render=False, physics_sweep=True) diff --git a/tests/agent_sdk/test_bilevel_sketch_regions.py b/tests/agent_sdk/test_bilevel_sketch_regions.py index aed08f6b55..e2094cc2ba 100644 --- a/tests/agent_sdk/test_bilevel_sketch_regions.py +++ b/tests/agent_sdk/test_bilevel_sketch_regions.py @@ -7,9 +7,6 @@ instead of the full option box. """ -import asyncio -from typing import Any - import numpy as np import pytest from gym.spaces import Box @@ -20,7 +17,7 @@ parse_sketch_from_text, strip_region_annotations from predicators.agent_sdk.sketch_refinement import refine_sketch from predicators.agent_sdk.sketch_types import GroundSampler, SketchStep -from predicators.agent_sdk.tools import ToolContext, create_mcp_tools +from predicators.agent_sdk.tools import ToolContext from predicators.structs import Action, GroundAtom, Object, \ ParameterizedOption, Predicate, State, Task, Type @@ -423,7 +420,6 @@ def fn(*_args): def _tool_ctx(ground_samplers=True, sandbox_dir=None): utils.reset_config({ - "agent_bilevel_use_llm_initial_params": True, "agent_bilevel_max_samples_per_step": 200, "agent_bilevel_ground_samplers": ground_samplers, }) @@ -441,21 +437,6 @@ def _tool_ctx(ground_samplers=True, sandbox_dir=None): ) -def _run_tool(tool_name, args, ground_samplers=True, sandbox_dir=None): - ctx = _tool_ctx(ground_samplers=ground_samplers, sandbox_dir=sandbox_dir) - tools = { - t.name: t.handler - for t in create_mcp_tools(ctx, tool_names=[tool_name]) - } - try: - loop = asyncio.get_event_loop() - except RuntimeError: - loop = asyncio.new_event_loop() - asyncio.set_event_loop(loop) - result: Any = loop.run_until_complete(tools[tool_name](args)) - return result["content"][0]["text"] - - def _probe_refine(plan, ground_samplers=True, sandbox_dir=None): """``sim.refine`` on the fake model - the agent-facing refinement surface (same parser and search core as the explorer's refinement).""" @@ -483,23 +464,6 @@ def test_probe_refine_rejects_bad_region(): _probe_refine("Move(block0:block)[0.85] ~ [0.1, 0.2]") -def test_submit_plan_ignores_region(): - """submit_plan runs the exact center; the region is inert.""" - text = _run_tool( - "submit_plan", { - "plan": ("Move(block0:block)[0.95] ~ [0.05] -> " - "{ReachedHi(block0:block)}"), - "include_states": - False, - "include_atoms": - False, - }) - # Goal achieved proves the exact center 0.95 ran (only x >= 0.9 passes); - # a searched/perturbed value could not be distinguished, so also check - # the report is a plain execution (no refinement verdict lines). - assert "Goal achieved: True" in text - - def test_probe_refine_ignores_region_when_disabled(): """With agent_bilevel_ground_samplers off, the annotation is a no-op. diff --git a/tests/agent_sdk/test_capture_decision.py b/tests/agent_sdk/test_capture_decision.py deleted file mode 100644 index f6a18f2431..0000000000 --- a/tests/agent_sdk/test_capture_decision.py +++ /dev/null @@ -1,263 +0,0 @@ -"""Direct unit tests for the pure ``_decide_capture`` function. - -The e2e harness (``test_submit_plan_capture.py``) drives the -same policy through the real tool handler; these tests pin the decision -table itself, one test per :class:`CaptureDecision` case, including -guard combinations the e2e tests do not reach: - -* capture disabled (``capture_goal_reaching_plans=False``) is silent for - every otherwise-triggering combination; -* an empty grounded plan never captures (unreachable via the handler, - whose parser rejects empty plans; the guard is preserved); -* best-effort mode with an existing validated capture ("best-effort - never displaces validated") falls through to the loud refusals or to - silence; -* a non-terminated evaluator rejection (illegitimate verdict whose - ``terminated`` is False while the env goal-check passed) does not - block a validated capture; -* flaky + evaluator-rejected cannot co-occur in the handler (validation - repeats are skipped on a rejection) but the ``not - evaluator_rejected`` guard on the FLAKY refusal is preserved; -* a wrong-task run without a reached goal is silent. -""" - -from typing import Any, Dict - -from predicators.agent_sdk.tools.capture import BestEffortReason, \ - CaptureDecision, CaptureOutcome, _decide_capture - - -def _decide(**overrides: Any) -> CaptureOutcome: - """Call ``_decide_capture`` on the validated-solve baseline with - overrides.""" - kwargs: Dict[str, Any] = dict(capture_enabled=True, - is_current_task=True, - have_plan=True, - goal_achieved=True, - evaluator_rejected=False, - reward_hack=False, - flaky=False, - best_effort_mode=False, - have_validated_capture=False) - kwargs.update(overrides) - return _decide_capture(**kwargs) - - -def test_validated_capture(): - """A clean goal-reaching plan on the current task is a validated solve. - - e2e: test_robust_plan_is_captured_with_validation_note. - """ - outcome = _decide() - assert outcome.decision is CaptureDecision.VALIDATED_CAPTURE - assert outcome.best_effort_reason is None - assert outcome.captured - - -def test_flaky_no_capture(): - """A flaky validation repeat refuses the capture. - - e2e: test_flaky_plan_is_not_captured. - """ - outcome = _decide(flaky=True) - assert outcome.decision is CaptureDecision.FLAKY_NO_CAPTURE - assert outcome.best_effort_reason is None - assert not outcome.captured - - -def test_param_sensitive_no_capture(): - """A physics-margin rollout failure refuses the capture. - - e2e: test_param_sensitive_plan_is_not_captured. - """ - outcome = _decide(param_sensitive=True) - assert outcome.decision is CaptureDecision.PARAM_SENSITIVE_NO_CAPTURE - assert outcome.best_effort_reason is None - assert not outcome.captured - - -def test_redundant_no_capture(): - """A plan that reaches the goal without one of its steps is refused. - - e2e: test_redundant_step_is_not_captured. - """ - outcome = _decide(redundant=True) - assert outcome.decision is CaptureDecision.REDUNDANT_NO_CAPTURE - assert outcome.best_effort_reason is None - assert not outcome.captured - - -def test_best_effort_redundant(): - """Best-effort mode captures a redundant submission, flagged.""" - outcome = _decide(best_effort_mode=True, redundant=True) - assert outcome.decision is CaptureDecision.BEST_EFFORT_CAPTURE - assert outcome.best_effort_reason is BestEffortReason.REDUNDANT - - -def test_param_sensitive_outranks_redundant_reason(): - """Param-sensitivity is the reason when both are set in best-effort mode - (the handler skips the necessity sweep after a margin failure, but the - ordering is pinned).""" - outcome = _decide(best_effort_mode=True, - param_sensitive=True, - redundant=True) - assert outcome.best_effort_reason is BestEffortReason.PARAM_SENSITIVE - - -def test_best_effort_param_sensitive(): - """Best-effort mode captures a param-sensitive submission.""" - outcome = _decide(best_effort_mode=True, param_sensitive=True) - assert outcome.decision is CaptureDecision.BEST_EFFORT_CAPTURE - assert outcome.best_effort_reason is BestEffortReason.PARAM_SENSITIVE - - -def test_flaky_outranks_param_sensitive_reason(): - """When both gates fail in best-effort mode, FLAKY is the reason (the. - - handler never produces this combination - the margin loop is skipped - after a flaky repeat - but the reason ordering is pinned). - """ - outcome = _decide(best_effort_mode=True, flaky=True, param_sensitive=True) - assert outcome.decision is CaptureDecision.BEST_EFFORT_CAPTURE - assert outcome.best_effort_reason is BestEffortReason.FLAKY - - -def test_reward_hack_no_capture(): - """A goal-atoms-reaching but evaluator-rejected rollout is refused. - - e2e: test_illegitimate_plan_is_not_captured_and_skips_repeats. - """ - outcome = _decide(evaluator_rejected=True, reward_hack=True) - assert outcome.decision is CaptureDecision.REWARD_HACK_NO_CAPTURE - assert not outcome.captured - - -def test_wrong_task_note(): - """A success on a train task is flagged, never captured.""" - outcome = _decide(is_current_task=False) - assert outcome.decision is CaptureDecision.WRONG_TASK_NOTE - assert not outcome.captured - - -def test_no_capture_on_honest_failure(): - """A plan that simply misses the goal is silent - no capture, no flag.""" - outcome = _decide(goal_achieved=False) - assert outcome.decision is CaptureDecision.NO_CAPTURE - assert not outcome.captured - - -def test_best_effort_honest_shortfall(): - """Best-effort mode captures a goal-missing submission as a shortfall. - - e2e: test_best_effort_honest_shortfall_is_captured. - """ - outcome = _decide(best_effort_mode=True, goal_achieved=False) - assert outcome.decision is CaptureDecision.BEST_EFFORT_CAPTURE - assert outcome.best_effort_reason is BestEffortReason.GOAL_NOT_REACHED - assert outcome.captured - - -def test_best_effort_reward_hack(): - """Best-effort mode captures an evaluator-rejected submission. - - e2e: test_best_effort_certificate_rejected_is_captured. - """ - outcome = _decide(best_effort_mode=True, - evaluator_rejected=True, - reward_hack=True) - assert outcome.decision is CaptureDecision.BEST_EFFORT_CAPTURE - assert outcome.best_effort_reason is BestEffortReason.REWARD_HACK - - -def test_best_effort_flaky(): - """Best-effort mode captures a flaky submission. - - e2e: test_best_effort_flaky_plan_is_captured. - """ - outcome = _decide(best_effort_mode=True, flaky=True) - assert outcome.decision is CaptureDecision.BEST_EFFORT_CAPTURE - assert outcome.best_effort_reason is BestEffortReason.FLAKY - - -def test_validated_solve_wins_over_best_effort_mode(): - """Edge: best-effort mode does not demote a fully validated solve.""" - outcome = _decide(best_effort_mode=True) - assert outcome.decision is CaptureDecision.VALIDATED_CAPTURE - assert outcome.best_effort_reason is None - - -def test_best_effort_never_displaces_validated_capture(): - """Edge: with a validated capture already recorded, best-effort mode is - inert - the loud refusals still fire, everything else is silent.""" - # A flaky resubmission is refused (and would escalate the gate) - # instead of overwriting the validated capture. - outcome = _decide(best_effort_mode=True, - have_validated_capture=True, - flaky=True) - assert outcome.decision is CaptureDecision.FLAKY_NO_CAPTURE - # A reward-hack resubmission is likewise refused. - outcome = _decide(best_effort_mode=True, - have_validated_capture=True, - evaluator_rejected=True, - reward_hack=True) - assert outcome.decision is CaptureDecision.REWARD_HACK_NO_CAPTURE - # An honest shortfall is silent. - outcome = _decide(best_effort_mode=True, - have_validated_capture=True, - goal_achieved=False) - assert outcome.decision is CaptureDecision.NO_CAPTURE - - -def test_capture_disabled_is_always_silent(): - """Edge: without capture_goal_reaching_plans every combination is - NO_CAPTURE - the open-loop planner must never record captures.""" - for overrides in ( - {}, # would be a validated solve - { - "flaky": True - }, - { - "evaluator_rejected": True, - "reward_hack": True - }, - { - "is_current_task": False - }, - { - "best_effort_mode": True, - "goal_achieved": False - }, - ): - outcome = _decide(capture_enabled=False, **overrides) - assert outcome.decision is CaptureDecision.NO_CAPTURE - - -def test_empty_plan_never_captures(): - """Edge: no grounded steps means no capture, even for a would-be - validated solve or best-effort submission.""" - assert _decide(have_plan=False).decision is CaptureDecision.NO_CAPTURE - assert _decide(have_plan=False, best_effort_mode=True, - goal_achieved=False).decision is CaptureDecision.NO_CAPTURE - - -def test_non_terminated_evaluator_rejection_does_not_block(): - """Edge: an illegitimate verdict with terminated=False (env goal-check - passed, evaluator's own termination did not) is not a reward hack and - does not block the capture.""" - outcome = _decide(evaluator_rejected=True, reward_hack=False) - assert outcome.decision is CaptureDecision.VALIDATED_CAPTURE - - -def test_flaky_refusal_requires_no_evaluator_rejection(): - """Edge: the FLAKY refusal's ``not evaluator_rejected`` guard - with a - non-terminated rejection alongside flakiness the result is silence, not - a FLAKY refusal (the handler never produces this combination because a - rejection skips validation repeats).""" - outcome = _decide(flaky=True, evaluator_rejected=True) - assert outcome.decision is CaptureDecision.NO_CAPTURE - - -def test_wrong_task_note_requires_goal(): - """Edge: a train-task run that misses the goal is silent, not flagged.""" - outcome = _decide(is_current_task=False, goal_achieved=False) - assert outcome.decision is CaptureDecision.NO_CAPTURE diff --git a/tests/agent_sdk/test_clearance_probe.py b/tests/agent_sdk/test_clearance_probe.py deleted file mode 100644 index 0caa102988..0000000000 --- a/tests/agent_sdk/test_clearance_probe.py +++ /dev/null @@ -1,202 +0,0 @@ -"""Tests for the capture gate's robot-clearance probe (tools/clearance.py). - -Regression for the 2026-09-02 bridge seed3 rerun: two belief-certified -explore plans (8/8 validation rollouts each) died on real contacts of -6.5 mm and 9.9 mm against a block every rollout had cleared - the gate -certified against the belief's own variability but never measured how -close the robot came to a bystander, so plans tighter than the -executor's realization slop passed by the luck of the draw. -""" -from collections import namedtuple -from types import SimpleNamespace -from typing import List, cast - -import numpy as np -import pybullet as p - -from predicators import utils -from predicators.agent_sdk.tools.clearance import RobotClearanceProbe, \ - clearance_lines, phase_skill_of -from predicators.structs import _Option - - -class _FakeSkill: - """A stand-in with only the config the verdict reads.""" - - def __init__(self, tol: float) -> None: - self._config = SimpleNamespace(move_to_pose_tol=tol, simulator=None) - - -def test_verdict_uses_executor_pose_slop() -> None: - """The bar is sqrt(move_to_pose_tol): 1e-4 -> 10 mm.""" - probe = RobotClearanceProbe(_FakeSkill(1e-4)) - assert abs(probe.threshold - 0.01) < 1e-12 - probe.num_probes = 5 - probe.min_dist = 0.0042 - probe.where = "rollout 2, step PickBlock(span2) vs span1" - ok, summary, detail = probe.verdict() - assert not ok - assert "4.2 mm" in summary and "10 mm" in summary - assert "span1" in detail and "inside the executor's 10 mm" in detail - probe.min_dist = 0.0123 - ok, summary, detail = probe.verdict() - assert ok and detail == "" and "12.3 mm" in summary - - -def test_verdict_without_probes_or_approach_is_ok() -> None: - """No probes, or nothing within the query distance, is a pass.""" - probe = RobotClearanceProbe(_FakeSkill(1e-4)) - assert probe.verdict() == (True, "", "") - assert not clearance_lines(probe) - assert not clearance_lines(None) - probe.num_probes = 3 # nothing came within the query distance - ok, summary, detail = probe.verdict() - assert ok and detail == "" and summary.startswith("min robot clearance: >") - - -def test_phase_skill_of_requires_a_planning_simulator() -> None: - """Only a skill-factory option with a planning simulator qualifies.""" - no_sim = cast( - _Option, - SimpleNamespace(parent=SimpleNamespace(policy=SimpleNamespace( - __self__=_FakeSkill(1e-4))))) - assert phase_skill_of([no_sim]) is None - plain = cast( - _Option, - SimpleNamespace(parent=SimpleNamespace(policy=lambda s: None))) - assert phase_skill_of([plain]) is None - - -_Obj = namedtuple("_Obj", ["name"]) # hashable stand-in for an Object - - -class _RecordingProbe(RobotClearanceProbe): - """Records the exempt set handed to each clearance query.""" - - def __init__(self, skill, held: str = "") -> None: - super().__init__(skill, stride=1) - self.exempts: List[set] = [] - self._held = held - - def _min_robot_clearance(self, state, exempt): - self.exempts.append(set(exempt)) - return 0.02, "wall" - - def _held_object_name(self, state): - return self._held or None - - -def _grounded(name: str, objects, skill=None) -> _Option: - parent = SimpleNamespace(policy=SimpleNamespace(__self__=skill)) - return cast(_Option, - SimpleNamespace(name=name, parent=parent, objects=objects)) - - -def test_observe_exempts_declared_contacts_and_the_held_object() -> None: - """A push's switch (skill-declared contact) and the object held when the - option starts join the option's arguments in the exempt set; a skill - without the hook, or a failing hook, exempts only the arguments and the - held object.""" - faucet = _Obj("faucet") - switch = _Obj("faucet_switch") - push_skill = SimpleNamespace( - _config=SimpleNamespace(move_to_pose_tol=1e-4, simulator=None), - contact_objects=lambda state, objects: {switch}) - probe = _RecordingProbe(_FakeSkill(1e-4)) - probe.observe("rollout 1", _grounded("SwitchOn", [faucet], push_skill), - ["s0", "s1"]) - assert probe.exempts == [{"faucet", "faucet_switch"}] * 2 - # A Place holds domino_5 when it starts: the released domino is - # exempt for the whole option (its retreat passes it by design). - probe = _RecordingProbe(_FakeSkill(1e-4), held="domino_5") - probe.observe("rollout 1", _grounded("Place", [], _FakeSkill(1e-4)), - ["s0"]) - assert probe.exempts == [{"domino_5"}] - # A failing contact hook is best-effort: only the arguments remain. - bad_skill = SimpleNamespace(contact_objects=lambda s, o: 1 / 0) - probe = _RecordingProbe(_FakeSkill(1e-4)) - probe.observe("rollout 1", _grounded("Push", [faucet], bad_skill), ["s0"]) - assert probe.exempts == [{"faucet"}] - assert probe.min_dist == 0.02 and "vs wall" in probe.where - - -def test_push_contact_object_is_the_body_at_the_target_pose() -> None: - """object_at_pose picks the posed object nearest the push target within the - contact radius, skipping the grounding's own objects and pose-less - objects.""" - # pylint: disable=import-outside-toplevel - from predicators.ground_truth_models.skill_factories.push import \ - object_at_pose - from predicators.structs import Object, State, Type - posed = Type("posed", ["x", "y", "z"]) - plain = Type("plain", ["is_on"]) - robot = Object("robot", posed) - faucet = Object("faucet", plain) - switch = Object("faucet_switch", posed) - other = Object("burner_switch0", posed) - state = State({ - robot: np.array([1.0, 1.45, 0.7]), - faucet: np.array([0.0]), - switch: np.array([1.0, 1.45, 0.65]), - other: np.array([0.6, 1.3, 0.65]), - }) - target = (1.0, 1.452, 0.65) - assert object_at_pose(state, target, {robot, faucet}) == switch - # The robot is the nearest posed body but is excluded. - assert object_at_pose(state, (1.0, 1.45, 0.66), {robot, faucet}) \ - == switch - # Nothing within the radius: no contact target. - assert object_at_pose(state, (2.0, 2.0, 0.65), {robot, faucet}) is None - - -def test_bridge_probe_measures_and_exempts(tmp_path) -> None: - """On the bridge env's own planning simulator: a block moved under the - gripper reads as a small clearance, the option's argument objects are - exempt, and a far scene reads as beyond the query distance.""" - del tmp_path - utils.reset_config({ - "env": "pybullet_bridge", - "seed": 0, - "num_train_tasks": 1, - "num_test_tasks": 0, - "skill_phase_use_motion_planning": True, - }) - # pylint: disable=import-outside-toplevel - from predicators.envs.pybullet_bridge import PyBulletBridgeEnv - from predicators.ground_truth_models import get_gt_options - env = PyBulletBridgeEnv(use_gui=False) - try: - task = env._generate_train_tasks()[0] # pylint: disable=protected-access - options = {o.name: o for o in get_gt_options(env.get_name())} - robot = env._robot # pylint: disable=protected-access - span1 = next(b for b in env._blocks if b.name == "span1") # pylint: disable=protected-access - pick = options["PickBlock"].ground([robot, span1], - np.array([0.0], dtype=np.float32)) - skill = phase_skill_of([pick]) - assert skill is not None - probe = RobotClearanceProbe(skill, stride=1) - init = task.init - # Far scene: nothing within the query distance of the home pose. - far, _ = probe._min_robot_clearance(init, set()) # pylint: disable=protected-access - assert far >= 0.05 - # Park span1 right under the gripper: a real, small clearance - # (the robot's z is the fingertip point, the block sits 9 cm - # below it, i.e. ~3 cm from the finger geometry). - near = init.copy() - for feat, val in (("x", init.get(robot, - "x")), ("y", init.get(robot, "y")), - ("z", init.get(robot, "z") - 0.09)): - near.set(span1, feat, val) - dist, body = probe._min_robot_clearance(near, set()) # pylint: disable=protected-access - assert body == "span1" and dist < 0.05 - # The pick's own target is exempt: the measurement moves on. - dist_exempt, body_exempt = probe._min_robot_clearance( # pylint: disable=protected-access - near, {"span1"}) - assert body_exempt != "span1" and dist_exempt >= dist - # observe() on the pick's trajectory records the closest approach - # against non-argument bodies only. - probe.observe("rollout 1", pick, [near]) - assert probe.num_probes == 1 - assert not probe.where.endswith("vs span1") - finally: - p.disconnect(env._physics_client_id) # pylint: disable=protected-access diff --git a/tests/agent_sdk/test_prompt_goldens.py b/tests/agent_sdk/test_prompt_goldens.py index 774a55c8dd..f61ca84d5c 100644 --- a/tests/agent_sdk/test_prompt_goldens.py +++ b/tests/agent_sdk/test_prompt_goldens.py @@ -16,82 +16,19 @@ import os import re -import numpy as np import pytest -from gym.spaces import Box from predicators import utils -from predicators.agent_sdk import learn_prompts, play_prompts +from predicators.agent_sdk import play_prompts from predicators.agent_sdk.prompt_templates import _PROMPTS_DIR, \ load_sections, placeholders, render from predicators.agent_sdk.sandbox_prompts import build_claude_md -from predicators.agent_sdk.sketch_prompts import build_early_stop_note, \ - build_solve_prompt, build_solve_system_prompt from predicators.agent_sdk.tools.continual_tools import CONTINUAL_TOOL_NAMES from predicators.approaches.agent_continual_frozen_approach import \ AgentContinualOracleDynamicsApproach -from predicators.structs import Action, GroundAtom, Object, \ - ParameterizedOption, Predicate, State, Task, TaskEvaluator, Type _GOLDEN_DIR = os.path.join(os.path.dirname(__file__), "prompt_goldens") -# -- fixtures ---------------------------------------------------------------- - -_THING = Type("thing", ["x", "y"]) -_FIXTURE = Type("fixture", ["x", "y", "is_on"]) - - -def _noop_policy(_s, _m, _o, _p): - return Action(np.zeros(1, dtype=np.float32)) - - -_MOVE = ParameterizedOption( - "MoveTo", - types=[_THING, _FIXTURE], - params_space=Box(low=np.array([-0.5, -0.5], dtype=np.float32), - high=np.array([0.5, 0.5], dtype=np.float32)), - policy=_noop_policy, - initiable=lambda _s, _m, _o, _p: True, - terminal=lambda _s, _m, _o, _p: True, - params_description=("dx", "dy"), -) -_WAIT = ParameterizedOption( - "Wait", - types=[], - params_space=Box(low=np.zeros(0, dtype=np.float32), - high=np.zeros(0, dtype=np.float32)), - policy=_noop_policy, - initiable=lambda _s, _m, _o, _p: True, - terminal=lambda _s, _m, _o, _p: True, -) -_AT = Predicate( - "At", [_THING, _FIXTURE], - lambda s, o: abs(s.get(o[0], "x") - s.get(o[1], "x")) < 0.1, - natural_language_assertion=lambda names: f"{names[0]} rests on {names[1]}") -_ON = Predicate("Active", [_FIXTURE], lambda s, o: s.get(o[0], "is_on") > 0.5) - - -class _Evaluator(TaskEvaluator): - """Evaluator whose objective statement reaches the prompt.""" - - def objective_description(self) -> str: - return "reward = (1.0 if success else 0.0) - 0.1 x moves used" - - -def _make_task(with_evaluator: bool) -> Task: - thing = Object("thing0", _THING) - fixture = Object("fixture0", _FIXTURE) - state = State({ - thing: np.array([0.0, 0.0]), - fixture: np.array([1.0, 0.0, 0.0]) - }) - goal = {GroundAtom(_AT, [thing, fixture]), GroundAtom(_ON, [fixture])} - return Task(state, - goal, - goal_nl="Put the thing on the fixture and switch it on.", - evaluator=_Evaluator(goal) if with_evaluator else None) - - # -- golden comparison ------------------------------------------------------- @@ -148,113 +85,22 @@ def test_render_rejects_missing_and_unused_placeholders() -> None: """A template placeholder without a value, or a value without a placeholder, fails loudly instead of shipping literal text.""" with pytest.raises(AssertionError): - render("solve_query", "goal_nl") + render("subclass_model", "physical_params") with pytest.raises(AssertionError): - render("solve_query", "goal_nl", goal_nl="g", extra="x") + render("subclass_model", "physical_params", param_list="p", extra="x") assert placeholders("a __B__ c __B__ __D__") == ["B", "D"] def test_rendered_prompts_have_no_leftover_placeholders() -> None: """No rendered prompt carries an unsubstituted ``__NAME__``.""" marker = re.compile(r"__[A-Z][A-Z0-9_]*__") - for text in (build_solve_system_prompt(explore=False), - build_solve_system_prompt(explore=True), build_claude_md()): + contract = play_prompts.build_model_contract(partially_observable=True) + tools = ["run_python"] + list(CONTINUAL_TOOL_NAMES) + for text in (play_prompts.build_play_system_prompt( + tools, model_contract=contract), build_claude_md()): assert not marker.search(text), marker.search(text) -# -- solve / explore --------------------------------------------------------- - - -def test_golden_solve_system_plan() -> None: - """Solve-phase system prompt, captured-plan deliverable, all gates.""" - _check_golden( - "solve_system_plan", - build_solve_system_prompt(explore=False, - propose_params=True, - ground_samplers=True, - physics_margin=True, - rule_param_margin=True, - necessity=True, - use_journal=True)) - - -def test_golden_solve_system_policy() -> None: - """Solve-phase system prompt, closed-loop policy deliverable.""" - _check_golden( - "solve_system_policy", - build_solve_system_prompt(explore=False, - policy_mode=True, - rule_param_margin=True, - policy_max_options=40, - policy_max_repeated_failures=3, - policy_max_repeated_noops=5)) - - -def test_golden_solve_system_explore() -> None: - """Explore-phase system prompt with the train-driven early-stop note.""" - utils.reset_config({ - "seed": - 0, - "online_learning_early_stopping": - True, - "online_learning_early_stopping_require_all_attempts": - True, - }) - _check_golden( - "solve_system_explore", - build_solve_system_prompt(explore=True, - physics_margin=True, - rule_param_margin=True, - early_stop_note=build_early_stop_note())) - - -def test_golden_solve_query() -> None: - """Solve query with scoring, run records, and the capture instructions.""" - utils.reset_config({"seed": 0}) - _check_golden( - "solve_query", - build_solve_prompt( - _make_task(with_evaluator=True), - all_predicates={_AT, _ON}, - all_options={_MOVE, _WAIT}, - tool_names=["submit_plan", "run_python", "Read", "Write"], - initial_image_section=( - "## Initial State Image\n\nA rendering of the initial " - "scene is at `./test_images/task000_initial_state.png`. " - "Read it first."), - propose_params=True, - require_tool_validation=True, - journal="### task 0 attempt 1\n- MoveTo dx=0.2 reached At", - strategy="## Approach\n- move, then activate", - attempts="### task 0 attempt 1/3\n- outcome: no capture", - )) - - -def test_golden_explore_query() -> None: - """Explore query with open questions and a belief-certified scheduled - plan.""" - utils.reset_config({"seed": 0}) - _check_golden( - "explore_query", - build_solve_prompt( - _make_task(with_evaluator=False), - all_predicates={_AT, _ON}, - all_options={_MOVE, _WAIT}, - tool_names=["submit_plan", "run_python"], - experiment_guidance=("The learning phase left this ranked " - "ledger of open questions:\n" - "1. Does Active require At? Experiment: " - "MoveTo then Wait."), - scheduled_plans=[ - " 0: MoveTo(thing0, fixture0)[0.2000, 0.0000]\n" - " NOTE: belief-certified; executes verbatim as a solve " - "attempt." - ], - propose_params=True, - explore_mode=True, - )) - - def test_golden_sandbox_claude_md() -> None: """The sandbox CLAUDE.md.""" _check_golden("sandbox_claude_md", build_claude_md()) @@ -263,116 +109,6 @@ def test_golden_sandbox_claude_md() -> None: # -- learn ------------------------------------------------------------------- -def test_golden_learn_system() -> None: - """Fully observable learn system prompt with a physical-parameter menu.""" - physical = learn_prompts.render_physical_params_section({ - "lateral_friction": { - "default": 0.5, - "lo": 0.05, - "hi": 2.0, - "scale": "log", - "description": "sliding friction of every body" - } - }) - _check_golden( - "learn_system", - learn_prompts.build_learn_system_prompt( - partially_observable=False, - residual_rule_signature="def residual_rule(state, updates, " - "params):", - scene_viz_hint="stage and render the scene", - physical_params_section=physical)) - - -def test_golden_learn_system_po_invention() -> None: - """Partially observable learn system prompt with predicate invention.""" - _check_golden( - "learn_system_po_invention", - learn_prompts.build_learn_system_prompt( - partially_observable=True, - residual_rule_signature="def residual_rule(observation, latent, " - "history, updates, params):", - scene_viz_hint="stage and render the scene", - extra_sections=[ - learn_prompts.render_predicate_invention_section( - "the scene workbench"), - ], - latent_extra_sections=[ - learn_prompts.render_predicate_latent_section(), - ], - workflow_extra=learn_prompts.render_predicate_workflow_extra())) - - -def test_golden_learn_message() -> None: - """Learn first message with a prior model, an objective, base-sim source, - and the predicate-invention and partial-observability additions.""" - _check_golden( - "learn_message", - learn_prompts.build_learn_message( - n_trajs=3, - n_transitions=42, - n_demos=1, - n_interaction=2, - trajectory_listing="[0] demo, task 0\n[1] interaction, task 0", - structs_ref="./reference/structs.py", - inferred_hint="{'fixture': ['is_on']}", - predicate_listing=" Holding(robot, thing)", - types_digest="thing: x, y\nfixture: x, y, is_on", - options_digest="MoveTo(thing, fixture)[dx, dy]", - simulator_file="./simulator.py", - objective_block=learn_prompts.render_objective_block( - "reward = success - 0.1 x moves used"), - prior_state_block=learn_prompts.render_prior_state_block( - ["`./simulator.py`", "`./predicates.py`"]), - divergence_block=learn_prompts.render_divergence_block( - "fixture.is_on: 4 mismatches", has_prior_model=True), - base_sim_block=learn_prompts.render_base_sim_block( - ["./reference/base_sim/scene.py"]), - tools_block=learn_prompts.render_tools_block( - ["run_python", "Read", "Write", "Edit"]), - extra_messages=[ - learn_prompts.render_predicate_invention_message( - "./predicates.py", - "Goal (natural language): switch the fixture on."), - learn_prompts.render_partial_observability_message(), - ])) - - -def test_golden_learn_program_system() -> None: - """The program-world-model learn system prompt with predicate invention - (paper arm C4).""" - _check_golden( - "learn_program_system", - learn_prompts.build_program_learn_system_prompt( - scene_viz_hint="stage and render the scene", - extra_sections=[ - learn_prompts.render_predicate_invention_section( - "the scene workbench"), - ], - workflow_extra=learn_prompts.render_predicate_workflow_extra())) - - -def test_golden_learn_program_message() -> None: - """The program-world-model learn first message, zero-shot variant.""" - _check_golden( - "learn_program_message", - learn_prompts.build_program_learn_message( - n_trajs=0, - n_transitions=0, - n_demos=0, - n_interaction=0, - trajectory_listing="", - structs_ref="./reference/structs.py", - predicate_listing="- Holding(robot:robot, block:block)", - types_digest="- robot: hand\n- block: x, y, held", - options_digest="- Pick(robot:robot, block:block)[]", - world_model_file="./world_model.py", - tools_block=learn_prompts.render_tools_block(["run_python"]), - extra_messages=[ - learn_prompts.render_program_zero_shot_message(), - ])) - - @pytest.mark.parametrize("model_based,noise,repair", [ (True, False, False), (True, True, True), diff --git a/tests/agent_sdk/test_refine_evaluator_gate.py b/tests/agent_sdk/test_refine_evaluator_gate.py index 53da53e2b2..9917f23314 100644 --- a/tests/agent_sdk/test_refine_evaluator_gate.py +++ b/tests/agent_sdk/test_refine_evaluator_gate.py @@ -88,7 +88,6 @@ def _make_probe(evaluator, model=None): # (p = 0.1 each) a 1e-9 event, so the certified/rejected paths # below are exercised at every probe seed, not just lucky ones. "agent_bilevel_max_samples_per_step": 200, - "agent_bilevel_use_llm_initial_params": False, }) init = State({_block: np.array([0.0], dtype=np.float32)}) goal = {GroundAtom(_ReachedHi, [_block])} diff --git a/tests/agent_sdk/test_session_fatal.py b/tests/agent_sdk/test_session_fatal.py index ffa01c604b..680f1866cf 100644 --- a/tests/agent_sdk/test_session_fatal.py +++ b/tests/agent_sdk/test_session_fatal.py @@ -463,7 +463,6 @@ async def _fake_sleep(secs): class _Ctx: attempt_start = 100.0 - attempt_deadline = 2800.0 python_call_deadline = None paused = 0.0 @@ -471,7 +470,6 @@ def pause_attempt_clock(self, seconds): """Shift the armed marks like the real ToolContext does.""" self.paused += seconds self.attempt_start += seconds - self.attempt_deadline += seconds ctx = _Ctx() monkeypatch.setattr(sb, "stream_agent_response", _fake_stream) @@ -485,7 +483,7 @@ def pause_attempt_clock(self, seconds): assert collected == healthy_resp assert sleeps == [sb._LIMIT_POLL_SECS] * 3 assert ctx.paused == sum(sleeps) - assert ctx.attempt_deadline == 2800.0 + sum(sleeps) + assert ctx.attempt_start == 100.0 + sum(sleeps) # Past the total cap the limited response is handed back as-is. monkeypatch.setattr(sb, "_LIMIT_MAX_TOTAL_WAIT_SECS", 1.0) responses[:] = [list(limit_resp), list(healthy_resp)] diff --git a/tests/agent_sdk/test_solve_prompt_strategy.py b/tests/agent_sdk/test_solve_prompt_strategy.py deleted file mode 100644 index 10353c78d2..0000000000 --- a/tests/agent_sdk/test_solve_prompt_strategy.py +++ /dev/null @@ -1,190 +0,0 @@ -"""Behavioral checks on the solve/explore prompts. - -Each test pins one rule the prompts must carry (the goldens in -``test_prompt_goldens.py`` pin the full text): the reward form reaches -the solver, solutions are banked before being optimized, the run records -are used skeptically, and exploration is framed as experiment design -against a belief model. -""" -import numpy as np - -from predicators import utils -from predicators.agent_sdk.sketch_prompts import build_early_stop_note, \ - build_solve_prompt, build_solve_system_prompt -from predicators.structs import Object, State, Task, TaskEvaluator, Type - -_DOM = Type("thing", ["x"]) - - -class _StubEvaluator(TaskEvaluator): - """Evaluator whose objective statement must reach the prompt.""" - - def objective_description(self) -> str: - return "reward = (1.0 if success else 0.0) - 0.2 x widgets used" - - -def _make_task(evaluator=None) -> Task: - obj = Object("thing0", _DOM) - return Task(State({obj: np.array([0.0])}), - set(), - goal_nl="Do the thing.", - evaluator=evaluator) - - -def _render(task: Task, journal: str = "", strategy: str = "") -> str: - return build_solve_prompt(task, - all_predicates=set(), - all_options=set(), - propose_params=True, - require_tool_validation=True, - journal=journal, - strategy=strategy) - - -def _render_explore(task: Task, propose_params: bool = True) -> str: - return build_solve_prompt(task, - all_predicates=set(), - all_options=set(), - propose_params=propose_params, - require_tool_validation=False, - explore_mode=True) - - -def test_scoring_section_from_evaluator() -> None: - """The evaluator's public reward form renders as a Scoring section; without - an evaluator the section is absent.""" - utils.reset_config({"seed": 0}) - prompt = _render(_make_task(_StubEvaluator(set()))) - assert "## Scoring (env ground-truth reward)" in prompt - assert "0.2 x widgets used" in prompt - assert "Decode every reward you observe" in prompt - assert "## Scoring" not in _render(_make_task(None)) - - -def test_solve_system_prompt_banks_before_optimizing() -> None: - """The solve system prompt states the banking semantics (a validated - capture replaces the banked one, a rejection never does); the explore - system prompt, which has no capture deliverable, omits them.""" - prompt = build_solve_system_prompt(explore=False) - assert "Bank a solution before optimizing it" in prompt - assert "a rejected submission never displaces it" in prompt - assert "strictly better" in prompt - assert "Bank a solution" not in build_solve_system_prompt(explore=True) - - -def test_solve_system_prompt_journal_protocol() -> None: - """The run-record protocol (incumbent, untried leads, scoped negatives, - conflicting entries) lives in the solve system prompt and only there.""" - prompt = build_solve_system_prompt(explore=False, use_journal=True) - assert "is the incumbent" in prompt - assert "untried leads first" in prompt - assert "only as broad as the family actually swept" in prompt - assert "both become open questions" in prompt - assert "Journal protocol" not in build_solve_system_prompt( - explore=False, use_journal=False) - assert "Journal protocol" not in build_solve_system_prompt(explore=True) - - -def test_solve_system_prompt_verifies_rules_before_steering() -> None: - """A rule inferred from one observation is re-tested before it guides the - search.""" - prompt = build_solve_system_prompt(explore=False) - assert "Verify a rule before steering by it" in prompt - - -def test_query_carries_run_records_not_their_protocol() -> None: - """The query renders the journal and strategy contents under their headers; - the rules for using them are not repeated there.""" - utils.reset_config({"seed": 0}) - prompt = _render(_make_task(None), - journal="- notes", - strategy="## Glue first\n- dab twice") - assert "## Solve Journal (./journal.md)" in prompt - assert "- notes" in prompt - assert "## Domain Strategy (advisory, written during learning)" in prompt - assert "- dab twice" in prompt - assert "incumbent" not in prompt - bare = _render(_make_task(None)) - assert "## Solve Journal" not in bare - assert "## Domain Strategy" not in bare - - -def test_explore_system_prompt_states_the_setting() -> None: - """The explore system prompt discloses the belief model, accepts a - simulator-failing plan, states the cycle data contract and first-cycle - coverage, and executes parameters verbatim; the solve prompt has none.""" - prompt = build_solve_system_prompt(explore=True) - assert "## Exploration setting" in prompt - assert "not in the belief model yet" in prompt - assert "a simulator-failing plan is a valid deliverable" in prompt - assert "at least one attempt at the full goal" in prompt - assert ("top-ranked open question's experiment executed as specified" - in prompt) - assert "Carry each interaction to its consequence" in prompt - assert "nothing is searched or substituted" in prompt - solve = build_solve_system_prompt(explore=False) - assert "## Exploration setting" not in solve - assert "belief model yet" not in solve - - -def test_explore_deliverable_and_certified_note() -> None: - """The explore deliverable is the final plan text; the certified-plan note - follows the execute-certified-plan flag.""" - with_note = build_solve_system_prompt(explore=True, - execute_certified_plan=True) - assert "Your final plan text: the experiment" in with_note - assert "executed verbatim as this episode's solve attempt" in with_note - without = build_solve_system_prompt(explore=True, - execute_certified_plan=False) - assert "executed verbatim" not in without - - -def test_early_stop_note_follows_config() -> None: - """The early-stop note credits certified exploration plans that solve for - real (train-driven) or perfect test phases (test-driven), and is absent - when early stopping is off.""" - utils.reset_config({ - "seed": - 0, - "online_learning_early_stopping": - True, - "online_learning_early_stopping_require_all_attempts": - True, - }) - note = build_early_stop_note() - assert note.startswith("The loop concludes early once the exploration") - assert "every episode of a cycle" in note - prompt = build_solve_system_prompt(explore=True, early_stop_note=note) - assert "exploration plans solve training" in prompt - utils.reset_config({ - "seed": - 0, - "online_learning_early_stopping": - True, - "online_learning_early_stopping_by_test_solve_rate": - True, - "online_learning_early_stopping_consecutive_perfect_tests": - 2, - }) - assert "2 consecutive test phases" in build_early_stop_note() - utils.reset_config({"seed": 0, "online_learning_early_stopping": False}) - assert build_early_stop_note() == "" - - -def test_query_openings_by_phase() -> None: - """Explore queries open with the information-gathering framing; solve - queries with the solving framing. - - Both end in the instructions. - """ - utils.reset_config({"seed": 0}) - explore = _render_explore(_make_task(None)) - assert explore.startswith("Design this episode's experiment") - assert "part of information gathering" in explore - assert "output the plan lines as your final text" in explore - solve = _render(_make_task(None)) - assert solve.startswith("Solve the task below") - assert "deliver it through the capture gate" in solve - # Param-free sketch mode keeps the same delivery semantics. - sketch = _render_explore(_make_task(None), propose_params=False) - assert "output the plan lines as your final text" in sketch diff --git a/tests/agent_sdk/test_solve_restart_journal.py b/tests/agent_sdk/test_solve_restart_journal.py index ca151bedc3..8254a4833b 100644 --- a/tests/agent_sdk/test_solve_restart_journal.py +++ b/tests/agent_sdk/test_solve_restart_journal.py @@ -1,10 +1,8 @@ -"""Tests for the solve journal and the wall-clock exploration budgets. +"""Tests for the run journal and ``run_python``'s budgets. Covers the journal module (entry caps, prompt-injection trimming, the -harness-owned attempt log), the cooperative probe deadline -(:class:`ProbeBudgetExceeded`), and ``run_python``'s budget handling -(refusal after the attempt deadline, per-call timeout with partial -output, ``[budget]`` footer). +harness-owned round log) and ``run_python``'s budget handling (per-call +timeout with partial output, ``[budget]`` footer). """ # pylint: disable=protected-access import asyncio @@ -17,7 +15,7 @@ from predicators import utils from predicators.agent_sdk import journal as journal_mod -from predicators.agent_sdk.belief_probe import BeliefProbe, ProbeBudgetExceeded +from predicators.agent_sdk.belief_probe import BeliefProbe from predicators.agent_sdk.tools import ToolContext, create_mcp_tools from predicators.structs import Action, GroundAtom, LowLevelTrajectory, \ Object, ParameterizedOption, Predicate, State, Task, Type @@ -107,19 +105,17 @@ def test_journal_append_and_read(tmp_path): """Entries append under headers and read back verbatim.""" sandbox = str(tmp_path) assert journal_mod.read_journal(sandbox) == "" - assert journal_mod.append_entry(sandbox, "task 0 attempt 1 (auto)", - "- outcome: no capture") is None + journal_mod.append_entry(sandbox, "Round 1", "- no environment action") content = journal_mod.read_journal(sandbox, filename=journal_mod.ATTEMPTS_FILENAME) - assert "### task 0 attempt 1 (auto)" in content - assert "- outcome: no capture" in content + assert "### Round 1" in content + assert "- no environment action" in content def test_journal_entry_truncated_at_cap(tmp_path): - """Oversize entries are truncated with a notice.""" + """Oversize entries are truncated with a marker.""" sandbox = str(tmp_path) - note = journal_mod.append_entry(sandbox, "big", "x" * 10000) - assert note is not None and "truncated" in note + journal_mod.append_entry(sandbox, "big", "x" * 10000) content = journal_mod.read_journal(sandbox, max_chars=10**6, filename=_A) assert "[entry truncated at the per-entry size cap]" in content assert len(content) < 5000 @@ -145,26 +141,6 @@ def test_journal_read_no_sandbox(): assert journal_mod.read_journal(None) == "" -def test_journal_read_raw_and_restore(tmp_path): - """read_raw snapshots faithfully and restore rolls entries back.""" - sandbox = str(tmp_path) - assert journal_mod.read_raw(None) is None - assert journal_mod.read_raw(sandbox, filename=_A) is None - journal_mod.append_entry(sandbox, "Agent notes (pre-test phase)", - "- learning fact") - snapshot = journal_mod.read_raw(sandbox, filename=_A) - assert snapshot is not None and "- learning fact" in snapshot - journal_mod.append_entry(sandbox, "Agent notes (test task 0)", - "- test-phase fact") - journal_mod.restore(sandbox, snapshot, filename=_A) - assert journal_mod.read_raw(sandbox, filename=_A) == snapshot - # A None snapshot means no journal file existed: restore deletes. - journal_mod.restore(sandbox, None, filename=_A) - assert journal_mod.read_raw(sandbox, filename=_A) is None - # Deleting an already-absent journal is a no-op, not an error. - journal_mod.restore(sandbox, None, filename=_A) - - # --------------------------------------------------------------------------- # attempt log (harness-owned file next to the agent's journal) # --------------------------------------------------------------------------- @@ -172,99 +148,27 @@ def test_journal_read_raw_and_restore(tmp_path): def test_attempt_log_is_a_separate_file(tmp_path): """Harness entries land in attempts.md; the agent's journal.md is a plain - file the harness never writes, and each is read, snapshotted and restored - on its own.""" + file the harness never writes, and each is read on its own.""" sandbox = str(tmp_path) - assert journal_mod.append_entry(sandbox, "task 0 attempt 1/1 (auto)", - "- outcome: no capture") is None - assert not os.path.isfile(journal_mod.journal_path(sandbox)) - assert os.path.isfile(journal_mod.attempts_path(sandbox)) + journal_path = os.path.join(sandbox, journal_mod.JOURNAL_FILENAME) + journal_mod.append_entry(sandbox, "Round 1", "- no environment action") + assert not os.path.isfile(journal_path) + assert os.path.isfile(os.path.join(sandbox, _A)) assert journal_mod.read_journal(sandbox) == "" - attempts = journal_mod.read_journal(sandbox, - filename=journal_mod.ATTEMPTS_FILENAME) - assert "### task 0 attempt 1/1 (auto)" in attempts + assert "### Round 1" in journal_mod.read_journal(sandbox, filename=_A) # The agent writes its journal with the file tools. - with open(journal_mod.journal_path(sandbox), "w", encoding="utf-8") as f: - f.write("### task 0 attempt 1\n- tried x=0.5: stopped 3 cm short\n") - assert "stopped 3 cm short" in journal_mod.read_journal(sandbox) - snapshot = journal_mod.read_raw(sandbox, - filename=journal_mod.ATTEMPTS_FILENAME) - journal_mod.append_entry(sandbox, "task 1 attempt 1/1 (auto)", - "- outcome: captured") - journal_mod.restore(sandbox, - snapshot, - filename=journal_mod.ATTEMPTS_FILENAME) - assert "task 1" not in journal_mod.read_journal( - sandbox, filename=journal_mod.ATTEMPTS_FILENAME) + with open(journal_path, "w", encoding="utf-8") as f: + f.write("### Level 1\n- tried x=0.5: stopped 3 cm short\n") assert "stopped 3 cm short" in journal_mod.read_journal(sandbox) + assert "stopped 3 cm short" not in journal_mod.read_journal(sandbox, + filename=_A) # --------------------------------------------------------------------------- -# probe deadline +# probe rollout metering # --------------------------------------------------------------------------- -def test_probe_raises_after_attempt_deadline(): - """Past the attempt deadline every probe sim call raises.""" - utils.reset_config({}) - ctx = _make_ctx() - ctx.attempt_deadline = time.monotonic() - 1.0 - sim = BeliefProbe(ctx) - try: - sim.reset() - assert False, "expected ProbeBudgetExceeded" - except ProbeBudgetExceeded as e: - assert "submit your single best plan" in str(e) - - -def test_probe_deadline_skipped_during_best_effort_nudge(): - """The final-submission nudge is never blocked by the spent budget.""" - utils.reset_config({}) - ctx = _make_ctx() - ctx.attempt_deadline = time.monotonic() - 1.0 - ctx.capture_best_effort_plan = True - sim = BeliefProbe(ctx) - sim.reset() # must not raise - - -def test_probe_trials_returns_partial_on_mid_loop_budget_expiry(): - """A budget stop mid-trials returns the completed trials (they are minutes - of sim time living in the return value, not stdout) instead of discarding - them.""" - utils.reset_config({}) - ctx = _make_ctx() - ctx.attempt_deadline = time.monotonic() + 60.0 - model = ctx.option_model - orig = model.get_next_state_and_num_actions - - def _expire_after_rollout(state, option): - result = orig(state, option) - ctx.attempt_deadline = time.monotonic() - 1.0 - return result - - model.get_next_state_and_num_actions = _expire_after_rollout - sim = BeliefProbe(ctx) - sim.reset() - res = sim.run("Move(block0:block)[0.95]", render=False, trials=5) - assert len(res.trials) == 1 - assert res.successes == 1 - assert any("time budget expired after 1/5 trials" in n for n in res.notes) - - -def test_probe_trials_reraises_when_nothing_completed(): - """With zero completed trials there is nothing to salvage.""" - utils.reset_config({}) - ctx = _make_ctx() - sim = BeliefProbe(ctx) - sim.reset() - ctx.attempt_deadline = time.monotonic() - 1.0 - try: - sim.run("Move(block0:block)[0.95]", render=False, trials=3) - assert False, "expected ProbeBudgetExceeded" - except ProbeBudgetExceeded: - pass - - def test_probe_counts_rollouts(): """run() meters full-plan rollouts (single and trials).""" utils.reset_config({}) @@ -283,19 +187,6 @@ def test_probe_counts_rollouts(): # --------------------------------------------------------------------------- -def test_run_python_refuses_after_attempt_deadline(tmp_path): - """A call arriving past the attempt deadline is refused unrun.""" - utils.reset_config({}) - ctx = _make_ctx(sandbox_dir=str(tmp_path)) - ctx.attempt_start = time.monotonic() - 10.0 - ctx.attempt_deadline = time.monotonic() - 1.0 - text = _call(_get_tool(ctx, "run_python"), - {"code": "print('should not run')"}) - assert "wall-clock exploration budget" in text - assert "should not run" not in text - assert "[budget]" in text - - def test_python_call_timeout_returns_partial_output(tmp_path): """A per-call timeout stops the sweep and returns printed output.""" utils.reset_config({ @@ -317,12 +208,10 @@ def test_run_python_budget_footer(tmp_path): }) ctx = _make_ctx(sandbox_dir=str(tmp_path)) ctx.attempt_start = time.monotonic() - ctx.attempt_deadline = ctx.attempt_start + 2700 code = "sim.reset(); print(sim.run('Move(block0:block)[0.95]', " \ "render=False).goal_reached)" text = _call(_get_tool(ctx, "run_python"), {"code": code}) assert "[budget] attempt time" in text - assert "/45 min" in text assert "sim rollouts this attempt: 1 (+1 this call)" in text @@ -379,55 +268,6 @@ def test_run_python_no_footer_outside_attempt(tmp_path): assert "[budget]" not in text -# --------------------------------------------------------------------------- -# prompt injection -# --------------------------------------------------------------------------- - - -def test_solve_prompt_includes_journal_section(): - """build_solve_prompt renders the journal and attempt log contents.""" - # pylint: disable-next=import-outside-toplevel - from predicators.agent_sdk.sketch_prompts import build_solve_prompt - utils.reset_config({}) - ctx = _make_ctx() - task = ctx.train_tasks[0] - journal_text = ("### task 0 attempt 1/3 (auto)\n" - "- outcome: no capture") - prompt = build_solve_prompt(task, - all_predicates={_ReachedHi}, - all_options={_Move}, - journal="### notes\n- tried x=0.5", - attempts=journal_text) - assert "## Attempt Log" in prompt - assert "- outcome: no capture" in prompt - assert "## Solve Journal" in prompt - assert "- tried x=0.5" in prompt - assert "./journal.md" in prompt - assert "record_journal" not in prompt - # Without journal content the section is absent entirely. - prompt_no_journal = build_solve_prompt(task, - all_predicates={_ReachedHi}, - all_options={_Move}) - assert "## Solve Journal" not in prompt_no_journal - assert "## Attempt Log" not in prompt_no_journal - - -def test_read_strategy_absent_present_and_truncated(tmp_path): - """read_strategy: "" when absent, verbatim when small, head-kept cap.""" - sandbox = str(tmp_path) - assert journal_mod.read_strategy(sandbox) == "" - assert journal_mod.read_strategy(None) == "" - with open(journal_mod.strategy_path(sandbox), "w", encoding="utf-8") as f: - f.write("## Approach\n- glue both faces\n") - assert "- glue both faces" in journal_mod.read_strategy(sandbox) - with open(journal_mod.strategy_path(sandbox), "w", encoding="utf-8") as f: - f.write("HEADLINE\n" + "x" * 10000) - content = journal_mod.read_strategy(sandbox) - assert content.startswith("HEADLINE") - assert "[strategy truncated at the prompt cap" in content - assert len(content) < 4300 - - # --------------------------------------------------------------------------- # run_python path argument # --------------------------------------------------------------------------- diff --git a/tests/agent_sdk/test_submit_plan_capture.py b/tests/agent_sdk/test_submit_plan_capture.py deleted file mode 100644 index a4b0babcce..0000000000 --- a/tests/agent_sdk/test_submit_plan_capture.py +++ /dev/null @@ -1,1122 +0,0 @@ -"""Capture-gating tests for the ``submit_plan`` tool. - -Drives the real MCP tool handler with a fake option model (no PyBullet), -covering the two gates in front of ``ctx.solved_plan``: - -* multi-rollout validation - the shared sim env is nondeterministic - across repeats, so a goal-reaching plan is captured only after every - one of ``CFG.agent_plan_validation_rollouts`` rollouts succeeds; a - flaky plan is reported to the agent instead of captured; -* task-evaluator legitimacy - a goal-reaching but ``legitimate=False`` - rollout is refused as a reward hack (the real evaluator applies the - same certificate). The refusal is internal: the agent-facing report - speaks only in (terminated, reward) terms and never leaks the - certificate's legitimacy bool or reason string. - -Under ``capture_best_effort_plan`` (the final-submission nudge) neither -gate refuses: the submission is captured regardless - honest shortfall, -certificate-rejected rollout, or flaky repeat - but is marked as NOT a -validated solve (``solved_plan_reached_goal=False``), so it executes for -its honest reward without counting as a solve. -""" - -import asyncio -import contextlib -from typing import Any - -import numpy as np -import pytest -from gym.spaces import Box - -from predicators import utils -from predicators.agent_sdk.tools import ToolContext, create_mcp_tools -from predicators.structs import Action, GroundAtom, LowLevelTrajectory, \ - Object, ParameterizedOption, Predicate, State, Task, TaskEvaluator, Type - -_block_type = Type("block", ["x"]) -_block = Object("block0", _block_type) - -_ReachedHi = Predicate("ReachedHi", [_block_type], - lambda s, o: s.get(o[0], "x") >= 0.9) - - -def _noop_policy(_s, _m, _o, _p): - return Action(np.zeros(1, dtype=np.float32)) - - -_Move = ParameterizedOption( - "Move", - types=[_block_type], - params_space=Box(low=np.array([0.0], dtype=np.float32), - high=np.array([1.0], dtype=np.float32)), - policy=_noop_policy, - initiable=lambda _s, _m, _o, _p: True, - terminal=lambda _s, _m, _o, _p: False, -) - -_PLAN_TEXT = "Move(block0:block)[0.95] -> {ReachedHi(block0:block)}" -# A plan that lands short of the goal (x=0.5 < 0.9), so ReachedHi never -# holds: an honest shortfall, not a reward hack. -_SHORTFALL_PLAN_TEXT = "Move(block0:block)[0.5]" - - -class _Model: - """Fake option model: Move sets block.x to its parameter value. - - ``succeed_first_n`` bounds how many calls apply the parameter; later - calls leave the state unchanged, emulating a flaky plan whose repeat - rollout misses the goal. Exposes ``last_trajectory`` so evaluator - verdicts are NON-coarse (a coarse verdict never blocks capture). - """ - - last_execution_failure = None - - def __init__(self, succeed_first_n=10**9): - self.num_calls = 0 - self._succeed_first_n = succeed_first_n - self.last_trajectory = None - - def get_next_state_and_num_actions(self, state, option): - """Roll the option forward one step, counting the call.""" - self.num_calls += 1 - nxt = state.copy() - if self.num_calls <= self._succeed_first_n and len(option.params): - nxt.set(_block, "x", float(option.params[0])) - self.last_trajectory = LowLevelTrajectory( - [state, nxt], [Action(np.zeros(1, dtype=np.float32))]) - return nxt, 1 - - -class _StubEvaluator(TaskEvaluator): - """Deterministic legitimacy verdict.""" - - def __init__(self, goal, legit): - super().__init__(goal) - self._legit = legit - - def _certify(self, states, step_options, sim_env=None): - if self._legit: - return True, "" - return False, "stub: the cascade was staged, not pushed" - - -def _make_ctx(model, evaluator=None, best_effort=False, goal_nl=None): - init = State({_block: np.array([0.0], dtype=np.float32)}) - goal = {GroundAtom(_ReachedHi, [_block])} - task = Task(init, goal, evaluator=evaluator, goal_nl=goal_nl) - ctx = ToolContext( - types={_block_type}, - predicates={_ReachedHi}, - processes=set(), - options={_Move}, - train_tasks=[task], - example_state=init, - option_model=model, - current_task=task, - ) - ctx.capture_goal_reaching_plans = True - ctx.capture_best_effort_plan = best_effort - return ctx - - -def _call_tool(ctx, plan_text=_PLAN_TEXT, extra_args=None): - """Invoke the real tool handler once against ``ctx``.""" - tools = { - t.name: t.handler - for t in create_mcp_tools(ctx, tool_names=["submit_plan"]) - } - try: - loop = asyncio.get_event_loop() - except RuntimeError: - loop = asyncio.new_event_loop() - asyncio.set_event_loop(loop) - call_args = {"plan": plan_text} - if extra_args: - call_args.update(extra_args) - result: Any = loop.run_until_complete(tools["submit_plan"](call_args)) - return result["content"][0]["text"] - - -def _run_tool(model, - evaluator=None, - rollouts=3, - plan_text=_PLAN_TEXT, - best_effort=False, - goal_nl=None, - extra_args=None): - utils.reset_config({"agent_plan_validation_rollouts": rollouts}) - ctx = _make_ctx(model, - evaluator=evaluator, - best_effort=best_effort, - goal_nl=goal_nl) - return _call_tool(ctx, plan_text, extra_args=extra_args), ctx - - -def test_robust_plan_is_captured_with_validation_note(): - """All rollouts succeed: captured, with the K/K validation note.""" - model = _Model() - text, ctx = _run_tool(model, rollouts=3) - assert "Captured as the current answer" in text - assert "Validated 3/3 rollouts" in text - assert ctx.solved_plan is not None - # One reported rollout + two validation repeats. - assert model.num_calls == 3 - - -def test_flaky_plan_is_not_captured(): - """A repeat rollout that misses the goal blocks capture, loudly.""" - model = _Model(succeed_first_n=1) - text, ctx = _run_tool(model, rollouts=3) - assert "FLAKY (plan NOT captured)" in text - assert "rollout 2/3 (planner seed" in text - assert "goal not reached" in text - assert "ReachedHi" in text - assert "Captured as the current answer" not in text - assert ctx.solved_plan is None - # The first rollout DID reach the goal - that is what makes it flaky. - assert "Goal achieved: True" in text - - -def test_single_rollout_config_disables_repeats(): - """``agent_plan_validation_rollouts=1`` restores single-rollout capture.""" - model = _Model() - text, ctx = _run_tool(model, rollouts=1) - assert "Captured as the current answer" in text - assert "Validated" not in text - assert ctx.solved_plan is not None - - -def test_flaky_message_reports_all_rollout_outcomes(): - """The FLAKY report lists EVERY rollout's outcome and an estimated. - - reliability, instead of stopping at the first failure - the - per-rollout list is what distinguishes failure modes. - """ - model = _Model(succeed_first_n=1) - text, _ = _run_tool(model, rollouts=3) - assert "estimated reliability 1/3" in text - assert "rollout 1 (planner seed" in text - assert "): goal reached" in text - assert "rollout 2 (planner seed" in text - assert "rollout 3 (planner seed" in text - assert text.count("FAILED -") >= 2 - # All three rollouts actually ran (no early break). - assert model.num_calls == 3 - - -def test_flaky_verdict_line_labeled_as_rollout_1(): - """When the submission is rejected as FLAKY, rollout 1's evaluator. - - verdict is labeled as such - unlabeled it read as a second, - contradictory verdict in the same message. - """ - model = _Model(succeed_first_n=1) - evaluator = _StubEvaluator({GroundAtom(_ReachedHi, [_block])}, True) - text, _ = _run_tool(model, evaluator=evaluator, rollouts=3) - assert "FLAKY (plan NOT captured)" in text - assert "[rollout 1 only - NOT the operative outcome" in text - - -def test_missing_goal_atoms_printed_even_with_goal_nl(): - """A goal-nl task still names the missing goal atoms on a shortfall - 'Goal - achieved: False' alone left agents unable to tell a near-miss from a non- - starter.""" - model = _Model() - text, _ = _run_tool(model, - plan_text=_SHORTFALL_PLAN_TEXT, - goal_nl="topple the target") - assert "Goal achieved: False" in text - assert "Missing goal atoms" in text - assert "ReachedHi" in text - assert model.num_calls == 1 - - -def test_illegitimate_plan_is_not_captured_and_skips_repeats(): - """A non-coarse ``legitimate=False`` verdict refuses capture before any - validation repeats are spent, reported in reward terms only: the - certificate's reason string never reaches the agent.""" - model = _Model() - goal = {GroundAtom(_ReachedHi, [_block])} - text, ctx = _run_tool(model, - evaluator=_StubEvaluator(goal, legit=False), - rollouts=3) - assert "NOT CAPTURED" in text - assert "scores it as a non-solve" in text - assert "stub: the cascade was staged" not in text - assert "legitimate" not in text - assert "Captured as the current answer" not in text - assert ctx.solved_plan is None - assert model.num_calls == 1 - - -def test_legitimate_plan_passes_both_gates(): - """A legitimate goal-reaching plan validates and captures normally.""" - model = _Model() - goal = {GroundAtom(_ReachedHi, [_block])} - text, ctx = _run_tool(model, - evaluator=_StubEvaluator(goal, legit=True), - rollouts=3) - assert "Captured as the current answer" in text - assert "Validated 3/3 rollouts" in text - assert "Task evaluator" in text and "reward=" in text - assert "solved=True" in text - assert "legitimate" not in text - assert ctx.solved_plan is not None - assert model.num_calls == 3 - - -def test_best_effort_honest_shortfall_is_captured(): - """An honest best-effort shortfall is captured, not refused. - - The plan does not reach the goal (terminated=False), so the - evaluator marks it legitimate=False - there is no genuine cascade to - certify. But it is not a reward hack, so under a best-effort - submission it is captured and executes for its honest reward instead - of being forfeited. - """ - model = _Model() - goal = {GroundAtom(_ReachedHi, [_block])} - text, ctx = _run_tool(model, - evaluator=_StubEvaluator(goal, legit=False), - rollouts=3, - plan_text=_SHORTFALL_PLAN_TEXT, - best_effort=True) - assert "Captured as the current answer" in text - assert "best-effort: goal NOT reached" in text - assert "will not count as a solve" in text - assert "NOT CAPTURED" not in text - assert ctx.solved_plan is not None - assert ctx.solved_plan_reached_goal is False - assert "Goal achieved: False" in text - - -def test_best_effort_certificate_rejected_is_captured(): - """A best-effort submission captures even a certificate-rejected rollout. - - The plan reaches the goal atoms (terminated=True) but the evaluator - scores the route as a non-solve. Outside best-effort mode that is - refused as a reward hack, but at final submission the budget is - spent: the plan is captured to execute for its honest reward, marked - as NOT a validated solve (run_20260714_145053 task 4: this refusal - forfeited the task entirely). - """ - model = _Model() - goal = {GroundAtom(_ReachedHi, [_block])} - text, ctx = _run_tool(model, - evaluator=_StubEvaluator(goal, legit=False), - rollouts=3, - best_effort=True) - assert "Captured as the current answer" in text - assert "best-effort" in text - assert "will not count as a solve" in text - assert "stub: the cascade was staged" not in text - assert "legitimate" not in text - assert "NOT CAPTURED" not in text - assert ctx.solved_plan is not None - assert ctx.solved_plan_reached_goal is False - # Certificate rejection skips the validation repeats. - assert model.num_calls == 1 - - -def test_flaky_rejection_escalates_later_captures(): - """A FLAKY rejection escalates the gate for later captures on the task. - - A flaky submission is evidence the agent is tuning in a marginal - parameter region, where a lucky streak can pass the base 3-rollout - gate and die on the single real episode (run_20260717_182321: a - 20/20-swept relay placement validated 3/3, then missed the target - for real). Resubmissions must therefore clear the escalated - ``agent_plan_validation_rollouts_after_flaky`` gate. - """ - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_rollouts_after_flaky": 6, - }) - flaky_model = _Model(succeed_first_n=1) - ctx = _make_ctx(flaky_model) - text = _call_tool(ctx) - assert "FLAKY (plan NOT captured)" in text - assert "captures require 6/6 successful rollouts" in text - # The resubmission (robust this time) faces the 6-rollout gate. - robust_model = _Model() - ctx.option_model = robust_model - text2 = _call_tool(ctx) - assert "Captured as the current answer" in text2 - assert "Validated 6/6 rollouts" in text2 - assert ctx.solved_plan is not None - assert robust_model.num_calls == 6 - - -def test_flaky_escalation_is_per_task(): - """Escalation is keyed to the task: a different test task keeps the base - gate.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_rollouts_after_flaky": 6, - }) - flaky_model = _Model(succeed_first_n=1) - ctx = _make_ctx(flaky_model) - ctx.test_task_idx = 0 - text = _call_tool(ctx) - assert "FLAKY (plan NOT captured)" in text - robust_model = _Model() - ctx.option_model = robust_model - ctx.test_task_idx = 1 - text2 = _call_tool(ctx) - assert "Validated 3/3 rollouts" in text2 - assert robust_model.num_calls == 3 - - -def test_validation_rollouts_enter_fresh_env_scope(): - """Every gate rollout - the reported first one included - runs inside - ``ctx.validation_env_scope``, so the whole gate shares one substrate - (reproducible in-session via ``sim.run(plan, trials=N)``).""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - }) - entered = [] - - @contextlib.contextmanager - def _scope(): - entered.append(True) - yield - - model = _Model() - ctx = _make_ctx(model) - ctx.validation_env_scope = _scope - text = _call_tool(ctx) - assert "Captured as the current answer" in text - assert "freshly constructed simulator" in text - # All 3 rollouts, the reported first one included. - assert len(entered) == 3 - - -def test_fresh_env_scope_disabled_by_config(): - """``agent_plan_validation_fresh_env=False`` keeps repeats on the shared - env even when a scope is installed.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": False, - }) - entered = [] - - @contextlib.contextmanager - def _scope(): - entered.append(True) - yield - - model = _Model() - ctx = _make_ctx(model) - ctx.validation_env_scope = _scope - text = _call_tool(ctx) - assert "Captured as the current answer" in text - assert "freshly constructed simulator" not in text - assert not entered - - -class _PhysicsAwareModel(_Model): - """Fake model whose success depends on a physics 'parameter'. - - Move only applies its parameter while ``friction >= 0.5``, emulating - a plan whose success band excludes part of the fit posterior. The - physics-margin scope perturbs ``friction`` the way the real scope - perturbs the fresh env's physical params. - """ - - def __init__(self): - super().__init__() - self.friction = 0.53 - - def get_next_state_and_num_actions(self, state, option): - if self.friction >= 0.5: - return super().get_next_state_and_num_actions(state, option) - self.num_calls += 1 - nxt = state.copy() - self.last_trajectory = LowLevelTrajectory( - [state, nxt], [Action(np.zeros(1, dtype=np.float32))]) - return nxt, 1 - - -def _physics_scope_ctx(model, points): - """A ctx whose fresh-env scope applies physics overrides to ``model``.""" - ctx = _make_ctx(model) - scope_overrides = [] - - @contextlib.contextmanager - def _scope(physical_overrides=None): - scope_overrides.append(physical_overrides) - prev = model.friction - if physical_overrides: - model.friction = physical_overrides["lateral_friction"] - try: - yield - finally: - model.friction = prev - - ctx.validation_env_scope = _scope - ctx.physics_margin_provider = lambda: list(points) - return ctx, scope_overrides - - -class _AdditiveModel(_PhysicsAwareModel): - """Fake model where each Move ADDS its parameter to block.x. - - Two Move(0.5) steps are both necessary to reach x >= 0.9, while a - Move(0.95) makes any other Move padding - the two shapes the - necessity gate has to tell apart. - """ - - def get_next_state_and_num_actions(self, state, option): - self.num_calls += 1 - nxt = state.copy() - nxt.set(_block, "x", - float(state.get(_block, "x")) + float(option.params[0])) - self.last_trajectory = LowLevelTrajectory( - [state, nxt], [Action(np.zeros(1, dtype=np.float32))]) - return nxt, 1 - - -_REDUNDANT_PLAN_TEXT = ( - "Move(block0:block)[0.95]\n" - "Move(block0:block)[0.95] -> {ReachedHi(block0:block)}") -_TWO_STEP_PLAN_TEXT = ("Move(block0:block)[0.5]\n" - "Move(block0:block)[0.5] -> {ReachedHi(block0:block)}") - - -def test_redundant_step_is_not_captured(): - """A plan that still reaches the goal with a step removed is refused. - - Regression for run_20260902_152811: a validated capture pressed - three of four buttons and released one that was never on, for a goal - its own model reached with two presses and a Wait. Every gate ran at - the full plan, so none could see the padding. - """ - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": False, - "agent_plan_validation_necessity": True, - }) - ctx, _ = _physics_scope_ctx(_AdditiveModel(), []) - text = _call_tool(ctx, plan_text=_REDUNDANT_PLAN_TEXT) - assert "REDUNDANT (plan NOT captured)" in text - assert "without step 0 (Move(block0)): goal STILL reached" in text - assert ctx.solved_plan is None - # The best refused submission is stashed for the best-effort nudge. - assert ctx.best_uncaptured_plan_lines is not None - - -def test_necessary_steps_are_captured_with_note(): - """A plan whose every step is needed captures with the check's note.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": False, - "agent_plan_validation_necessity": True, - }) - ctx, _ = _physics_scope_ctx(_AdditiveModel(), []) - text = _call_tool(ctx, plan_text=_TWO_STEP_PLAN_TEXT) - assert "Captured as the current answer" in text - assert "Necessity check passed" in text - assert ctx.solved_plan is not None - - -def test_necessity_gate_disabled_by_config(): - """The default-off flag captures the padded plan without ablations.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": False, - "agent_plan_validation_necessity": False, - }) - ctx, scope_overrides = _physics_scope_ctx(_AdditiveModel(), []) - text = _call_tool(ctx, plan_text=_REDUNDANT_PLAN_TEXT) - assert "Captured as the current answer" in text - assert "REDUNDANT" not in text - # Main rollout + 2 execution repeats, no ablation rollouts. - assert scope_overrides == [None, None, None] - - -def test_param_sensitive_plan_is_not_captured(): - """A plan that fails at a -1-sigma physics point is refused. - - Regression for run_20260723_091108: a capture validated 8/8 at the - fitted lateral_friction 0.5319 failed deterministically at true 0.5 - - execution repeats at the fitted values cannot see zero margin to - the fit's parameter error. - """ - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": True, - }) - model = _PhysicsAwareModel() - ctx, scope_overrides = _physics_scope_ctx(model, [{ - "lateral_friction": 0.48 - }, { - "lateral_friction": 0.59 - }]) - text = _call_tool(ctx) - assert "PARAM-SENSITIVE (plan NOT captured)" in text - assert "lateral_friction=0.48" in text - assert ctx.solved_plan is None - # Main rollout + 2 execution repeats (no overrides) + 2 physics - # points. - assert scope_overrides == [ - None, None, None, { - "lateral_friction": 0.48 - }, { - "lateral_friction": 0.59 - } - ] - - -def test_param_sensitive_refusal_names_the_straddle(): - """Under the interval belief the refusal carries the certified fraction, - the passing and failing ranges and the probe cue.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": True, - "code_sim_learning_interval_belief": True, - }) - model = _PhysicsAwareModel() - ctx, _ = _physics_scope_ctx(model, [{ - "lateral_friction": 0.48 - }, { - "lateral_friction": 0.59 - }]) - text = _call_tool(ctx) - assert "PARAM-SENSITIVE (plan NOT captured)" in text - assert ("(1/2 belief-interval points passed; lateral_friction: fails at " - "0.48, passes at 0.59)") in text - assert "straddles this plan's success boundary" in text - assert "sim.suggest_probes" in text - assert ctx.param_sensitive_refusal_pending - assert ctx.solved_plan is None - - -def test_physics_margin_pass_is_captured_with_note(): - """Margin points inside the success band capture with the margin note.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": True, - }) - model = _PhysicsAwareModel() - ctx, _ = _physics_scope_ctx(model, [{ - "lateral_friction": 0.51 - }, { - "lateral_friction": 0.59 - }]) - text = _call_tool(ctx) - assert "Captured as the current answer" in text - assert "Physics-margin check passed" in text - - -def test_physics_margin_disabled_by_config(): - """The default-off flag skips the margin rollouts entirely.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": False, - }) - model = _PhysicsAwareModel() - ctx, scope_overrides = _physics_scope_ctx(model, [{ - "lateral_friction": 0.48 - }]) - text = _call_tool(ctx) - assert "Captured as the current answer" in text - assert "PARAM-SENSITIVE" not in text - assert scope_overrides == [None, None, None] - - -def test_physics_margin_vacuous_without_points(): - """An empty provider (no fit / degenerate posterior) adds no note.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": True, - }) - model = _PhysicsAwareModel() - ctx, scope_overrides = _physics_scope_ctx(model, []) - text = _call_tool(ctx) - assert "Captured as the current answer" in text - assert "Physics-margin check passed" not in text - assert scope_overrides == [None, None, None] - - -def _rule_param_scope_ctx(model, members): - """A ctx whose rule-param override scope applies members to ``model``.""" - ctx = _make_ctx(model) - applied = [] - - @contextlib.contextmanager - def _fresh_scope(physical_overrides=None): - del physical_overrides - yield - - @contextlib.contextmanager - def _override(point): - applied.append(point) - prev = model.friction - model.friction = point["dab_tol"] - try: - yield - finally: - model.friction = prev - - ctx.validation_env_scope = _fresh_scope - ctx.rule_param_margin_provider = lambda: list(members) - ctx.rule_param_override_scope = _override - return ctx, applied - - -def test_rule_param_sensitive_plan_is_not_captured(): - """A plan that fails under a calibrated rule-param ensemble member is - refused as PARAM-SENSITIVE. - - Regression for the bridge cycles 5-7: plans centered on marginal - operating points of uncertain learned constants (a glue dab at 16 mm - of the true 20 mm radius) passed 5/5 nominal validation rollouts and - the (empty) physics sweep, then failed in the real environment. - """ - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": False, - "agent_plan_validation_rule_param_margin": True, - }) - model = _PhysicsAwareModel() - ctx, applied = _rule_param_scope_ctx(model, [{ - "dab_tol": 0.59 - }, { - "dab_tol": 0.48 - }]) - text = _call_tool(ctx) - assert "PARAM-SENSITIVE (plan NOT captured)" in text - assert "rule-param ensemble member 2/2" in text - assert "dab_tol=0.48" in text - assert ctx.solved_plan is None - assert applied == [{"dab_tol": 0.59}, {"dab_tol": 0.48}] - - -def test_rule_param_margin_pass_is_captured_with_note(): - """Members inside the success band capture with the margin note.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": False, - "agent_plan_validation_rule_param_margin": True, - }) - model = _PhysicsAwareModel() - ctx, _ = _rule_param_scope_ctx(model, [{ - "dab_tol": 0.51 - }, { - "dab_tol": 0.59 - }]) - text = _call_tool(ctx) - assert "Captured as the current answer" in text - assert "Rule-parameter margin check passed" in text - - -def test_rule_param_margin_disabled_by_config(): - """The default-off flag skips the rule-param sweep entirely.""" - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_plan_validation_fresh_env": True, - "agent_plan_validation_physics_margin": False, - "agent_plan_validation_rule_param_margin": False, - }) - model = _PhysicsAwareModel() - ctx, applied = _rule_param_scope_ctx(model, [{"dab_tol": 0.48}]) - text = _call_tool(ctx) - assert "Captured as the current answer" in text - assert "PARAM-SENSITIVE" not in text - assert not applied - - -def test_best_effort_flaky_plan_is_captured(): - """A best-effort submission captures a flaky plan instead of refusing. - - Rollout 1 solves, rollout 2 misses. Outside best-effort mode that is - refused as FLAKY (the agent can add margin and resubmit), but at - final submission there is no budget left, so the plan is captured - with the flaky detail in the note and marked as NOT a validated - solve. - """ - model = _Model(succeed_first_n=1) - text, ctx = _run_tool(model, rollouts=3, best_effort=True) - assert "Captured as the current answer" in text - assert "best-effort" in text - assert "rollout 2/3 (planner seed" in text - assert "FLAKY (plan NOT captured)" not in text - assert ctx.solved_plan is not None - assert ctx.solved_plan_reached_goal is False - - -def test_capture_stashes_evaluator_reward(): - """A capture records the evaluator verdict's reward for the restart loop's - cross-attempt ranking.""" - model = _Model() - goal = {GroundAtom(_ReachedHi, [_block])} - _, ctx = _run_tool(model, - evaluator=_StubEvaluator(goal, legit=True), - rollouts=1) - assert ctx.solved_plan is not None - assert isinstance(ctx.solved_plan_eval_reward, float) - - -def test_capture_without_evaluator_has_no_reward(): - """No evaluator: the reward stash stays None (ranked below any rewarded - capture, above no capture).""" - model = _Model() - _, ctx = _run_tool(model, rollouts=1) - assert ctx.solved_plan is not None - assert ctx.solved_plan_eval_reward is None - - -def test_budget_footer_during_attempt(): - """The footer reports THIS call's rollout delta, not the attempt's - cumulative total masquerading as one.""" - import time as _time # pylint: disable=import-outside-toplevel - utils.reset_config({"agent_plan_validation_rollouts": 1}) - model = _Model() - ctx = _make_ctx(model) - ctx.attempt_start = _time.monotonic() - # Simulate a prior explore sweep this attempt. - ctx.attempt_rollout_count = 50 - text = _call_tool(ctx, _PLAN_TEXT) - assert "[budget] attempt time" in text - assert "sim rollouts this attempt: 51 (+1 this call)" in text - - -def test_no_budget_footer_outside_attempt(): - """No attempt in flight: no footer noise.""" - model = _Model() - text, _ctx = _run_tool(model, rollouts=1) - assert "[budget]" not in text - - -def test_validation_repeats_use_decorrelated_planner_seeds(): - """Each validation repeat rolls out under its own ``CFG.seed``. - - A fresh env per repeat is not enough for independent samples: the - skills' motion planning reads the constant ``CFG.seed`` at call - time, so identical-seed repeats are bit-identical replays and the - flaky gate detects nothing (run_20260722_204632: a 13/13-validated - capture was a coin flip on the real episode). The capture rollout - itself must keep the base seed; the repeats offset it; the base seed - must be restored afterward. - """ - - class _SeedRecordingModel(_Model): - """Records ``CFG.seed`` at each rollout step.""" - - def __init__(self): - super().__init__() - self.seeds = [] - - def get_next_state_and_num_actions(self, state, option): - from predicators.settings import \ - CFG # pylint: disable=import-outside-toplevel - self.seeds.append(CFG.seed) - return super().get_next_state_and_num_actions(state, option) - - model = _SeedRecordingModel() - _, ctx = _run_tool(model, rollouts=3) - from predicators.settings import \ - CFG # pylint: disable=import-outside-toplevel - base = CFG.seed - assert ctx.solved_plan is not None - # One capture rollout at the base seed, two decorrelated repeats. - assert model.seeds == [base, base + 1, base + 2] - - -class _SeedRecordingModel2(_Model): - """Records ``CFG.seed`` at each rollout step (module-level reuse).""" - - def __init__(self, succeed_first_n=10**9): - super().__init__(succeed_first_n=succeed_first_n) - self.seeds = [] - - def get_next_state_and_num_actions(self, state, option): - from predicators.settings import \ - CFG # pylint: disable=import-outside-toplevel - self.seeds.append(CFG.seed) - return super().get_next_state_and_num_actions(state, option) - - -def test_validation_rollouts_arg_raises_the_gate(): - """``validation_rollouts=N`` requests a stricter gate than configured.""" - model = _Model() - text, ctx = _run_tool(model, - rollouts=3, - extra_args={"validation_rollouts": 5}) - assert "Validated 5/5 rollouts" in text - assert ctx.solved_plan is not None - assert model.num_calls == 5 - - -def test_validation_rollouts_arg_cannot_lower_the_gate(): - """A request below the configured gate is ignored: the gate is a floor - - letting the agent lower it would let a lucky draw bypass validation.""" - model = _Model() - text, ctx = _run_tool(model, - rollouts=3, - extra_args={"validation_rollouts": 1}) - assert "Validated 3/3 rollouts" in text - assert ctx.solved_plan is not None - assert model.num_calls == 3 - - -def test_flaky_report_names_seeds_and_reproduction_path(): - """A FLAKY rejection names each rollout's planner seed and tells the agent - how to reproduce the failed rollout (``sim.run(plan, seed=S)``).""" - model = _Model(succeed_first_n=1) - text, _ = _run_tool(model, rollouts=3) - from predicators.settings import \ - CFG # pylint: disable=import-outside-toplevel - base = CFG.seed - assert f"rollout 1 (planner seed {base}): goal reached" in text - assert f"(planner seed {base + 1}): FAILED" in text - assert "sim.run(plan, seed=" in text - - -# --------------------------------------------------------------------------- -# annotation intersection over passing validation rollouts -# --------------------------------------------------------------------------- - -_MidHi = Predicate("MidHi", [_block_type], - lambda s, o: s.get(o[0], "x") >= 0.4) - -_TWO_STEP_PLAN = ("Move(block0:block)[0.5] -> {MidHi(block0:block)}\n" - "Move(block0:block)[0.95] -> {ReachedHi(block0:block)}") - -_NEG_STEP_PLAN = ("Move(block0:block)[0.5] -> {NOT ReachedHi(block0:block)}\n" - "Move(block0:block)[0.95] -> {ReachedHi(block0:block)}") - - -class _RepeatDriftModel(_Model): - """Two-step plans; validation repeats drift the FIRST step's landing. - - Rollout 1 applies each Move's parameter exactly. Later rollouts land - the first step of each pair at ``repeat_first_step_x`` instead, - while the second step still applies its parameter, so repeats reach - the goal (pass) with a different intermediate state. - """ - - def __init__(self, repeat_first_step_x): - super().__init__() - self._repeat_x = repeat_first_step_x - - def get_next_state_and_num_actions(self, state, option): - nxt, n = super().get_next_state_and_num_actions(state, option) - rollout_idx = (self.num_calls - 1) // 2 - step_in_rollout = (self.num_calls - 1) % 2 - if rollout_idx >= 1 and step_in_rollout == 0: - nxt.set(_block, "x", self._repeat_x) - return nxt, n - - -def test_annotation_pruned_when_absent_in_a_passing_repeat(): - """An atom that held in rollout 1 by luck is pruned by the repeats. - - The repeats pass (goal reached), but the intermediate MidHi does not - hold there, so the captured sketch drops it - keeping it would arm - the closed-loop monitor with a divergence the plan does not need. - """ - model = _RepeatDriftModel(repeat_first_step_x=0.3) - utils.reset_config({"agent_plan_validation_rollouts": 3}) - ctx = _make_ctx(model) - ctx.predicates.add(_MidHi) - text = _call_tool(ctx, _TWO_STEP_PLAN) - assert "Captured as the current answer" in text - sketch = ctx.solved_sketch - assert sketch is not None - assert not sketch[0].subgoal_atoms # MidHi pruned - assert {str(a) for a in sketch[1].subgoal_atoms} == \ - {"ReachedHi(block0:block)"} - assert ctx.solved_plan_validation_summary == \ - "validation: 3/3 rollouts ok" - - -def test_annotation_kept_when_only_a_failing_repeat_disagrees(): - """Failing rollouts contribute no evidence to the intersection. - - Repeats fail outright here (goal never reached), so under the best- - effort nudge the flaky capture falls back to the rollout-1 filter - and MidHi survives. - """ - model = _RepeatDriftModel(repeat_first_step_x=0.3) - # Make repeats FAIL: the second step of later rollouts also misses. - orig = _RepeatDriftModel.get_next_state_and_num_actions - - def _failing(self, state, option): - nxt, n = orig(self, state, option) - rollout_idx = (self.num_calls - 1) // 2 - if rollout_idx >= 1: - nxt.set(_block, "x", 0.3) - return nxt, n - - # pylint: disable-next=no-value-for-parameter - model.get_next_state_and_num_actions = _failing.__get__(model) - utils.reset_config({"agent_plan_validation_rollouts": 3}) - ctx = _make_ctx(model, best_effort=True) - ctx.predicates.add(_MidHi) - text = _call_tool(ctx, _TWO_STEP_PLAN) - assert "best-effort" in text - sketch = ctx.solved_sketch - assert sketch is not None - assert {str(a) for a in sketch[0].subgoal_atoms} == \ - {"MidHi(block0:block)"} - assert "first failure: rollout" in ctx.solved_plan_validation_summary - - -def test_negative_annotation_pruned_when_violated_in_a_passing_repeat(): - """The mirrored rule: a NOT atom must be absent in every passing repeat's - post-state to survive.""" - model = _RepeatDriftModel(repeat_first_step_x=0.95) - utils.reset_config({"agent_plan_validation_rollouts": 3}) - ctx = _make_ctx(model) - text = _call_tool(ctx, _NEG_STEP_PLAN) - assert "Captured as the current answer" in text - sketch = ctx.solved_sketch - assert sketch is not None - # NOT ReachedHi held after step 1 of rollout 1 (x=0.5) but is - # violated in the passing repeats (x=0.95), so it is pruned. - assert not sketch[0].subgoal_neg_atoms - - -def test_plan_capture_carries_validation_summary(): - """take_plan_capture surfaces the summary alongside the plan.""" - model = _Model() - _, ctx = _run_tool(model, rollouts=3) - capture = ctx.take_plan_capture() - assert capture.validation_summary == "validation: 3/3 rollouts ok" - assert ctx.solved_plan_validation_summary is None - - -def test_parallel_repeats_capture_robust_plan(): - """With agent_validation_parallel_workers set, the validation repeats run - in forked children: the verdict and note are identical to sequential mode, - and the parent-side model counter proves the repeats did NOT run in this - process (fork isolation).""" - from predicators.agent_sdk.parallel_rollouts import \ - parallel_rollouts_available # pylint: disable=import-outside-toplevel - if not parallel_rollouts_available(): - pytest.skip("fork not available on this platform") - model = _Model() - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_validation_parallel_workers": 2, - }) - ctx = _make_ctx(model) - text = _call_tool(ctx) - assert "Captured as the current answer" in text - assert "Validated 3/3 rollouts" in text - assert ctx.solved_plan is not None - # Only the reported rollout ran in the parent; both repeats ran in - # forked children whose counter increments never propagate back. - assert model.num_calls == 1 - # The parent-side rollout accounting still counts every repeat. - assert ctx.attempt_rollout_count >= 3 - - -def test_parallel_repeats_still_reject_flaky_plan(): - """Child-side failures propagate through the result queue.""" - from predicators.agent_sdk.parallel_rollouts import \ - parallel_rollouts_available # pylint: disable=import-outside-toplevel - if not parallel_rollouts_available(): - pytest.skip("fork not available on this platform") - model = _Model(succeed_first_n=1) - utils.reset_config({ - "agent_plan_validation_rollouts": 3, - "agent_validation_parallel_workers": 2, - }) - ctx = _make_ctx(model) - text = _call_tool(ctx) - assert "FLAKY (plan NOT captured)" in text - assert "Captured as the current answer" not in text - assert ctx.solved_plan is None - - -# ── Execution-verifiability probe ──────────────────────────────────── -# The monitor evaluates captured annotations on real observations, which -# carry no latent. An annotation whose truth in the certifying post-state -# depends on the belief latent (or whose classifier raises without one) -# is excluded from the monitored sketch at capture time. - - -class _LatentModel(_Model): - """Post-states carry a belief latent, as belief-sim rollouts do.""" - - def __init__(self, latent_value, **kwargs): - super().__init__(**kwargs) - self._latent_value = latent_value - - def get_next_state_and_num_actions(self, state, option): - nxt, n = super().get_next_state_and_num_actions(state, option) - nxt.latent = {"_bonds": self._latent_value} - return nxt, n - - -def _latent_only_classifier(s, o, latent=None): - del s, o - return bool((latent or {}).get("_bonds")) - - -_LatentBonded = Predicate("LatentBonded", [_block_type], - _latent_only_classifier) - - -def _raising_without_latent(s, o, latent=None): - del s, o - return bool(latent["_bonds"]) # TypeError when latent is None - - -_RaisingBond = Predicate("RaisingBond", [_block_type], _raising_without_latent) - -_LATENT_PLAN = ("Move(block0:block)[0.95] -> " - "{ReachedHi(block0:block), LatentBonded(block0:block)}") - -_NEG_RAISING_PLAN = ( - "Move(block0:block)[0.5] -> {NOT RaisingBond(block0:block)}\n" - "Move(block0:block)[0.95] -> {ReachedHi(block0:block)}") - - -def test_latent_only_annotation_excluded_from_monitoring(): - """A positive annotation that only holds through the belief latent is - dropped from the captured sketch (it would read false on every real - observation and abort a healthy episode), and the capture message says so; - the observable annotation survives.""" - model = _LatentModel({"a|b"}) - utils.reset_config({"agent_plan_validation_rollouts": 3}) - ctx = _make_ctx(model) - ctx.predicates.add(_LatentBonded) - text = _call_tool(ctx, _LATENT_PLAN) - assert "Captured as the current answer" in text - assert "cannot be verified from a real observation" in text - assert "LatentBonded" in text - sketch = ctx.solved_sketch - assert sketch is not None - assert {str(a) for a in sketch[0].subgoal_atoms} == \ - {"ReachedHi(block0:block)"} - - -def test_observation_backed_annotation_survives_probe(): - """An annotation that holds from observable features alone is kept: - - the probe only drops latent-dependent atoms. - """ - model = _LatentModel({"a|b"}) - utils.reset_config({"agent_plan_validation_rollouts": 3}) - ctx = _make_ctx(model) - text = _call_tool(ctx, _PLAN_TEXT) - assert "Captured as the current answer" in text - assert "cannot be verified from a real observation" not in text - sketch = ctx.solved_sketch - assert sketch is not None - assert {str(a) for a in sketch[0].subgoal_atoms} == \ - {"ReachedHi(block0:block)"} diff --git a/tests/agent_sdk/test_submit_policy_capture.py b/tests/agent_sdk/test_submit_policy_capture.py deleted file mode 100644 index 511e34683a..0000000000 --- a/tests/agent_sdk/test_submit_policy_capture.py +++ /dev/null @@ -1,219 +0,0 @@ -"""Capture-gating tests for the ``submit_policy`` tool (policy mode). - -Mirrors test_submit_plan_capture.py's fixtures: drives the real -MCP handler with a fake option model, covering the policy-mode gates in -front of ``ctx.solved_policy_source`` - multi-rollout validation with -fresh policy memory per rollout, source snapshotting, the -recovered-option-failure semantics, and the mode gate itself. -""" -import asyncio -import os -from typing import Any - -import numpy as np -from gym.spaces import Box - -from predicators import utils -from predicators.agent_sdk.tools import ToolContext, create_mcp_tools -from predicators.structs import Action, GroundAtom, LowLevelTrajectory, \ - Object, ParameterizedOption, Predicate, State, Task, Type - -_block_type = Type("block", ["x"]) -_block = Object("block0", _block_type) -_ReachedHi = Predicate("ReachedHi", [_block_type], - lambda s, o: s.get(o[0], "x") >= 0.9) - - -def _noop_policy(_s, _m, _o, _p): - return Action(np.zeros(1, dtype=np.float32)) - - -_Move = ParameterizedOption( - "Move", - types=[_block_type], - params_space=Box(low=np.array([0.0], dtype=np.float32), - high=np.array([1.0], dtype=np.float32)), - policy=_noop_policy, - initiable=lambda _s, _m, _o, _p: True, - terminal=lambda _s, _m, _o, _p: False, -) - -_GOAL_POLICY = ''' -def get_option(state, memory): - for obj in state: - if state.get(obj, "x") >= 0.9: - return None - return "Move(block0:block)[0.95]" -''' - - -class _Model: - """Move sets block.x to its parameter; flaky after N calls.""" - - last_execution_failure = None - - def __init__(self, succeed_first_n=10**9): - self.num_calls = 0 - self._succeed_first_n = succeed_first_n - self.last_trajectory = None - - def get_next_state_and_num_actions(self, state, option): - """Roll the option forward one step, counting the call.""" - self.num_calls += 1 - nxt = state.copy() - if self.num_calls <= self._succeed_first_n and len(option.params): - nxt.set(_block, "x", float(option.params[0])) - self.last_trajectory = LowLevelTrajectory( - [state, nxt], [Action(np.zeros(1, dtype=np.float32))]) - return nxt, 1 - - -def _make_ctx(model, sandbox_dir, best_effort=False): - init = State({_block: np.array([0.0], dtype=np.float32)}) - goal = {GroundAtom(_ReachedHi, [_block])} - task = Task(init, goal) - ctx = ToolContext( - types={_block_type}, - predicates={_ReachedHi}, - processes=set(), - options={_Move}, - train_tasks=[task], - example_state=init, - option_model=model, - current_task=task, - sandbox_dir=sandbox_dir, - log_dir=sandbox_dir, - ) - ctx.capture_goal_reaching_plans = True - ctx.capture_best_effort_plan = best_effort - ctx.policy_capture_mode = True - return ctx - - -def _write_policy(sandbox_dir, source): - path = os.path.join(sandbox_dir, "policy.py") - with open(path, "w", encoding="utf-8") as f: - f.write(source) - return path - - -def _call_tool(ctx, extra_args=None): - tools = { - t.name: t.handler - for t in create_mcp_tools(ctx, tool_names=["submit_policy"]) - } - try: - loop = asyncio.get_event_loop() - except RuntimeError: - loop = asyncio.new_event_loop() - asyncio.set_event_loop(loop) - result: Any = loop.run_until_complete(tools["submit_policy"](extra_args - or {})) - return result["content"][0]["text"] - - -def _run_tool(model, - tmp_path, - source=_GOAL_POLICY, - rollouts=3, - best_effort=False, - extra_args=None): - # The local-sandbox path resolution reads /sandbox. - sandbox = os.path.join(str(tmp_path), "sandbox") - os.makedirs(sandbox, exist_ok=True) - utils.reset_config({ - "agent_plan_validation_rollouts": rollouts, - "agent_solve_policy_mode": True, - "agent_sdk_use_local_sandbox": True, - }) - _write_policy(sandbox, source) - ctx = _make_ctx(model, str(tmp_path), best_effort=best_effort) - return _call_tool(ctx, extra_args=extra_args), ctx, sandbox - - -def test_robust_policy_is_captured_with_validation_note(tmp_path): - """All rollouts succeed: source captured with the K/K note.""" - model = _Model() - text, ctx, _ = _run_tool(model, tmp_path, rollouts=3) - assert "Captured policy.py as the current answer" in text - assert "Validated 3/3 rollouts" in text - assert ctx.solved_policy_source is not None - assert "get_option" in ctx.solved_policy_source - assert ctx.solved_plan is None - assert ctx.solved_plan_reached_goal is True - assert ctx.solved_plan_validation_summary == \ - "validation: 3/3 rollouts ok" - - -def test_flaky_policy_is_not_captured(tmp_path): - """A failing validation repeat blocks the capture, loudly.""" - model = _Model(succeed_first_n=1) - text, ctx, _ = _run_tool(model, tmp_path, rollouts=3) - assert "FLAKY (policy NOT captured)" in text - assert ctx.solved_policy_source is None - - -def test_flaky_policy_best_effort_captured(tmp_path): - """Under the final nudge a flaky policy is captured as best-effort.""" - model = _Model(succeed_first_n=1) - text, ctx, _ = _run_tool(model, tmp_path, rollouts=3, best_effort=True) - assert "best-effort" in text - assert ctx.solved_policy_source is not None - assert ctx.solved_plan_reached_goal is False - - -def test_source_snapshot_not_rereading_file(tmp_path): - """Editing policy.py after the call cannot swap unvalidated code.""" - model = _Model() - _, ctx, sandbox = _run_tool(model, tmp_path, rollouts=1) - assert ctx.solved_policy_source is not None - _write_policy(sandbox, "def get_option(state, memory):\n return None\n") - assert "Move(block0:block)" in ctx.solved_policy_source - - -def test_recovered_option_failure_still_captures(tmp_path): - """Closed-loop: a surfaced-and-recovered failure does not disqualify.""" - model = _Model() - orig = _Model.get_next_state_and_num_actions - - def _first_call_fails(self, state, option): - if self.num_calls == 0: - self.num_calls += 1 - self.last_execution_failure = "simulated failure" - return state.copy(), 0 - return orig(self, state, option) - - # pylint: disable-next=no-value-for-parameter - model.get_next_state_and_num_actions = _first_call_fails.__get__(model) - text, ctx, _ = _run_tool(model, tmp_path, rollouts=1) - assert "OPTION FAILURE (surfaced to the policy" in text - assert "Captured policy.py as the current answer" in text - assert ctx.solved_policy_source is not None - - -def test_mode_gate_refuses_outside_policy_mode(tmp_path): - """The tool refuses when the attempt is not in policy mode.""" - model = _Model() - sandbox = os.path.join(str(tmp_path), "sandbox") - os.makedirs(sandbox, exist_ok=True) - utils.reset_config({ - "agent_solve_policy_mode": False, - "agent_sdk_use_local_sandbox": True, - }) - _write_policy(sandbox, _GOAL_POLICY) - ctx = _make_ctx(model, str(tmp_path)) - ctx.policy_capture_mode = False - text = _call_tool(ctx) - assert "only available in policy mode" in text - - -def test_missing_policy_file_is_instructive(tmp_path): - """A missing policy.py errors with writing instructions.""" - model = _Model() - utils.reset_config({ - "agent_solve_policy_mode": True, - "agent_sdk_use_local_sandbox": True, - }) - ctx = _make_ctx(model, str(tmp_path)) - text = _call_tool(ctx) - assert "No ./policy.py found" in text diff --git a/tests/agent_sdk/test_tool_registry.py b/tests/agent_sdk/test_tool_registry.py index 9d67423531..e581cca5db 100644 --- a/tests/agent_sdk/test_tool_registry.py +++ b/tests/agent_sdk/test_tool_registry.py @@ -119,12 +119,12 @@ def test_list_session_tool_names_filters_and_combines() -> None: """Filtered MCP names drop unknowns; ``extra_mcp_tools`` pass through.""" fake = SimpleNamespace(name="run_python") grouped = list_session_tool_names( - mcp_filter=["submit_plan", "not_a_tool", "run_python"], + mcp_filter=["not_a_tool", "run_python"], extra_mcp_tools=[fake], include_builtin=False, ) assert grouped == { - "mcp": ["submit_plan", "run_python"], + "mcp": ["run_python"], "extra": ["run_python"], } @@ -143,13 +143,13 @@ def test_solve_and_synthesis_tool_names_are_independent() -> None: class _Approach(AgentSessionMixin): def _get_solve_tool_names(self) -> Optional[List[str]]: - return ["run_python", "submit_plan"] + return ["run_python", "skills_execute_plan"] def _get_synthesis_tool_names(self) -> Optional[List[str]]: return ["run_python"] obj = _Approach() - assert obj._get_solve_tool_names() == ["run_python", "submit_plan"] + assert obj._get_solve_tool_names() == ["run_python", "skills_execute_plan"] assert obj._get_synthesis_tool_names() == ["run_python"] @@ -158,13 +158,11 @@ def test_get_allowed_tool_list_passes_dynamic_names_through() -> None: list is the single source of truth, with no silent filtering against ``ALL_TOOL_NAMES``.""" allowed = get_allowed_tool_list([ - "submit_plan", # static - "run_python", # dynamic synthesis tool + "run_python", # static, or a dynamic synthesis instance "my_dynamic_tool", # a dynamic tool the roster never lists ]) prefix = f"mcp__{MCP_SERVER_NAME}__" assert allowed == [ - f"{prefix}submit_plan", f"{prefix}run_python", f"{prefix}my_dynamic_tool", ] diff --git a/tests/agent_sdk/test_trajectory_summary.py b/tests/agent_sdk/test_trajectory_summary.py deleted file mode 100644 index 4d7b9bfbd3..0000000000 --- a/tests/agent_sdk/test_trajectory_summary.py +++ /dev/null @@ -1,110 +0,0 @@ -"""Tests for the ``## Trajectory Summary`` query section. - -The section is the only outcome feedback an agent without a learn phase -gets about its earlier episodes, so it must say what plan ran and how -the env judged it, not only how many steps it took. -""" -from typing import Sequence - -import numpy as np -from gym.spaces import Box - -from predicators import utils -from predicators.agent_sdk.sketch_prompts import summarize_trajectories -from predicators.structs import Action, GroundAtom, LowLevelTrajectory, \ - Object, ParameterizedOption, Predicate, State, Task, Type - -_block_type = Type("block", ["x"]) -_block = Object("block0", _block_type) -_Far = Predicate("Far", [_block_type], lambda s, o: s.get(o[0], "x") > 0.5) -_Move = ParameterizedOption( - "Move", - types=[_block_type], - params_space=Box(0.0, 1.0, (1, )), - policy=lambda s, m, o, p: Action(np.zeros(1, dtype=np.float32)), - initiable=lambda s, m, o, p: True, - terminal=lambda s, m, o, p: True, -) -_Wait = ParameterizedOption( - "Wait", - types=[], - params_space=Box(0.0, 1.0, (0, )), - policy=lambda s, m, o, p: Action(np.zeros(1, dtype=np.float32)), - initiable=lambda s, m, o, p: True, - terminal=lambda s, m, o, p: True, -) - - -def _state(x: float) -> State: - return State({_block: np.array([x], dtype=np.float32)}) - - -def _traj(xs: Sequence[float], options: Sequence[str], - **kwargs) -> LowLevelTrajectory: - states = [_state(x) for x in xs] - actions = [] - for name in options: - act = Action(np.zeros(1, dtype=np.float32)) - if name == "Move": - act.set_option(_Move.ground([_block], np.array([0.5]))) - elif name == "Wait": - act.set_option(_Wait.ground([], np.array([]))) - actions.append(act) - return LowLevelTrajectory(states, actions, _train_task_idx=0, **kwargs) - - -def _setup() -> None: - utils.reset_config({"agent_sdk_max_trajectories_in_context": 2}) - - -def test_summary_reports_plan_verdict_and_stable_numbers() -> None: - """Each recent trajectory shows the option plan it executed (repeats - collapsed), the env's reward and goal verdict, and keeps its global index - so a number means the same episode in every session.""" - _setup() - trajs = [ - _traj([0.0, 0.0], ["Move"], _env_reward=0.0, _env_terminated=False), - _traj([0.0, 0.2, 0.9], ["Move", "Move"], - _env_reward=0.0, - _env_terminated=False), - _traj([0.0, 0.9, 0.9], ["Move", "Wait"], - _env_reward=1.0, - _env_terminated=True), - ] - text = summarize_trajectories(trajs, {_Far}) - assert "(3 total, showing last 2)" in text - assert "Trajectory 0:" not in text - assert "Trajectory 1: 2 steps" in text - assert "Executed: Move(block0)\n" in text - assert "Trajectory 2: 2 steps" in text - assert "Executed: Move(block0) -> Wait()" in text - assert "Outcome: env reward 0.00, goal NOT reached" in text - assert "Outcome: env reward 1.00, goal atoms held at the end" in text - assert "Gained: Far(block0:block)" in text - - -def test_summary_falls_back_to_the_task_goal_without_a_verdict() -> None: - """A trajectory the env never evaluated (no reward) still gets a goal - verdict from the train task's goal atoms when the tasks are given.""" - _setup() - task = Task(_state(0.0), {GroundAtom(_Far, [_block])}) - solved = _traj([0.0, 0.9], ["Move"]) - unsolved = _traj([0.0, 0.1], ["Move"]) - text = summarize_trajectories([solved, unsolved], {_Far}, - train_tasks=[task]) - assert text.count("Outcome: goal atoms held at the end") == 1 - assert text.count("Outcome: goal NOT reached") == 1 - # Without tasks there is nothing to judge against: no outcome line. - assert "Outcome" not in summarize_trajectories([solved], {_Far}) - - -def test_summary_without_option_tags_omits_the_plan_line() -> None: - """Raw-action trajectories (no option on the actions) keep the old step- - count and atom-delta lines and simply have no plan to show.""" - _setup() - traj = LowLevelTrajectory([_state(0.0), _state(0.9)], - [Action(np.zeros(1, dtype=np.float32))]) - text = summarize_trajectories([traj], {_Far}) - assert "Trajectory 0: 1 steps" in text - assert "Executed" not in text - assert "Gained: Far(block0:block)" in text diff --git a/tests/approaches/test_agent_continual_real_to_sim_approach.py b/tests/approaches/test_agent_continual_real_to_sim_approach.py index 122ca9946e..2e53e009d4 100644 --- a/tests/approaches/test_agent_continual_real_to_sim_approach.py +++ b/tests/approaches/test_agent_continual_real_to_sim_approach.py @@ -29,8 +29,6 @@ "agent_sim_learn_declared_params_only": True, "continual_uncertainty_decisions": False, "agent_sim_learn_param_uncertainty": False, - "agent_plan_validation_rule_param_margin": False, - "agent_plan_validation_physics_margin": False, "agent_explorer_info_seeking": False, "agent_explorer_info_seeking_adaptive": False, "agent_explorer_info_seeking_noise_aware": False, diff --git a/tests/approaches/test_agent_program_world_model_approach.py b/tests/approaches/test_agent_program_world_model_approach.py index c7f50261c0..f136f68593 100644 --- a/tests/approaches/test_agent_program_world_model_approach.py +++ b/tests/approaches/test_agent_program_world_model_approach.py @@ -3,15 +3,12 @@ import os from typing import Any, List -import numpy as np - from predicators import utils -from predicators.agent_sdk import learn_prompts from predicators.agent_sdk.tools import ToolContext from predicators.approaches import agent_program_world_model_approach as apwm from predicators.approaches.agent_sim_learning_approach import _SynthesisPaths from predicators.code_sim_learning.program_world_model import \ - ProgramOptionModel, load_program_world_model + load_program_world_model from predicators.datasets import create_dataset from predicators.envs import create_new_env from predicators.ground_truth_models import get_gt_options @@ -63,37 +60,16 @@ def _bare(env: Any, train_tasks: List[Any], options: Any) -> Any: return approach -def test_belief_particles_and_override_scope() -> None: - """Particles, the nominal latent, the override scope, and the rolled - latents all come from the installed program.""" +def test_installed_program_rolls_latents() -> None: + """An installed program backs the option model, and materialise_latent + rolls it along a recorded trajectory.""" env, train_tasks, options = _cover() approach = _bare(env, train_tasks, options) - # No model yet: no particles. - assert not approach._belief_particles() program, err = load_program_world_model(_PROGRAM, env.types, env.predicates, options) assert err is None and program is not None approach._install_program(program) assert approach._option_model is approach._program_model - particles = approach._belief_particles() - # Distinct draws only: initial_latent has three outcomes. - assert 1 <= len(particles) <= 3 - assert len({p["phase"] for p in particles}) == len(particles) - # Deterministic across calls (seeded). - assert approach._belief_particles() == particles - # The current task drives the draw when one is set. - approach._tool_context.current_task = train_tasks[1] - assert approach._belief_particles() == particles - # Under the scope every latent-less start rolls from the particle. - model: ProgramOptionModel = approach._program_model - (pick_place, ) = [o for o in options if o.name == "PickPlace"] - option = pick_place.ground([], np.array([0.4], dtype=np.float32)) - with approach._particle_override_scope({"phase": 20}): - nxt, _ = model.get_next_state_and_num_actions(train_tasks[0].init, - option) - assert nxt.latent == {"phase": 21} - assert model.initial_latent_override is None - # materialise_latent rolls the program along a recorded trajectory. dataset = create_dataset(env, train_tasks, options, env.predicates) traj = dataset.trajectories[0] latents = approach.materialise_latent(traj) @@ -130,31 +106,3 @@ def test_rehydrate_from_world_model_file(tmp_path, monkeypatch) -> None: assert program.latent_features == {"robot": ["phase"]} assert "world_model.py" in approach._CHECKPOINT_SANDBOX_FILES assert "world_model_versions" in approach._CHECKPOINT_SANDBOX_DIRS - - -def test_program_prompts_render() -> None: - """System prompt and first message render without leftovers.""" - system = learn_prompts.build_program_learn_system_prompt( - scene_viz_hint="x", - extra_sections=[ - learn_prompts.render_predicate_invention_section("workbench") - ], - workflow_extra=learn_prompts.render_predicate_workflow_extra()) - assert "world_model.py" in system and "sim.score" in system - assert "Plan format" in system and "Predicate Invention" in system - assert "__" not in system.replace("__init__", "") - message = learn_prompts.build_program_learn_message( - n_trajs=0, - n_transitions=0, - n_demos=0, - n_interaction=0, - trajectory_listing="", - structs_ref="./reference/structs.py", - predicate_listing="- Holding(robot, block)", - types_digest="types", - options_digest="options", - world_model_file="./world_model.py", - extra_messages=[learn_prompts.render_program_zero_shot_message()]) - assert "./world_model.py" in message - assert "No trajectory has been recorded" in message - assert "__" not in message diff --git a/tests/approaches/test_agent_sim_learning_ablations.py b/tests/approaches/test_agent_sim_learning_ablations.py index 2c17d5ba08..431812b326 100644 --- a/tests/approaches/test_agent_sim_learning_ablations.py +++ b/tests/approaches/test_agent_sim_learning_ablations.py @@ -13,7 +13,6 @@ import pytest from predicators import utils -from predicators.agent_sdk import learn_prompts from predicators.agent_sdk.tools import create_synthesis_tools from predicators.approaches import agent_sim_learning_approach as asla from predicators.code_sim_learning.fit_space import ParamSpec, \ @@ -136,13 +135,12 @@ def test_deploy_declared_params_uses_the_declaration_as_the_estimate() -> None: assert approach._fit_sse == 1.5 -def test_rule_param_margin_alone_builds_the_ensemble() -> None: - """A6 (info-seeking off, gate on) keeps the validation ensemble; with both - consumers off (A6+A7) none is built.""" +def test_ensemble_follows_info_seeking() -> None: + """Info-seeking, the ensemble's consumer, builds it; with info-seeking off + none is built.""" utils.reset_config({ "agent_sim_learn_declared_params_only": True, - "agent_explorer_info_seeking": False, - "agent_plan_validation_rule_param_margin": True, + "agent_explorer_info_seeking": True, "agent_explorer_info_ensemble_size": 5, }) approach = _bare_approach() @@ -153,7 +151,6 @@ def test_rule_param_margin_alone_builds_the_ensemble() -> None: utils.reset_config({ "agent_sim_learn_declared_params_only": True, "agent_explorer_info_seeking": False, - "agent_plan_validation_rule_param_margin": False, }) approach = _bare_approach() approach._physical_param_specs = list(_PHYS_SPECS) @@ -215,24 +212,6 @@ def test_no_data_seeding_applies_declared_physical_inits() -> None: assert approach._last_fit_result is None -def test_declared_params_prompt_section_is_flag_gated() -> None: - """The no-estimation section renders only under the A3 flag.""" - kwargs: Dict[str, Any] = dict( - partially_observable=False, - residual_rule_signature="def rule(state, updates, params):", - scene_viz_hint="look", - ) - plain = learn_prompts.build_learn_system_prompt(**kwargs) - declared = learn_prompts.build_learn_system_prompt( - declared_params_only=True, **kwargs) - marker = "Harness parameter estimation is DISABLED" - assert marker not in plain - assert marker in declared - assert "__" not in declared.replace("__init__", "") - zero_shot = learn_prompts.render_zero_shot_message() - assert "No trajectory has been recorded" in zero_shot - - def test_estimation_surfaces_refuse_under_declared_params(tmp_path) -> None: """A4: sim.fit, fit_params and sweep_params refuse; the plain report runs.""" diff --git a/tests/approaches/test_agent_sim_learning_approach.py b/tests/approaches/test_agent_sim_learning_approach.py index 5941407ec1..dec1b59273 100644 --- a/tests/approaches/test_agent_sim_learning_approach.py +++ b/tests/approaches/test_agent_sim_learning_approach.py @@ -678,13 +678,11 @@ def test_base_sim_reference_provisioning() -> None: def test_synthesis_tool_names_are_run_python_only(): - """The learn session's only tool is run_python; the journal is a plain file - the agent edits, whatever the journal flag says.""" - stub = SimpleNamespace(_do_synthesize_samplers=False) - for use_journal in (True, False): - utils.reset_config({"agent_solve_use_journal": use_journal}) - names = AgentSimLearningApproach._get_synthesis_tool_names(stub) - assert names == ["run_python"] + """The synthesis session's only tool is run_python; the journal is a plain + file the agent edits.""" + names = AgentSimLearningApproach._get_synthesis_tool_names( + SimpleNamespace()) + assert names == ["run_python"] # --------------------------------------------------------------------------- diff --git a/tests/approaches/test_agent_sim_prompt_formatting.py b/tests/approaches/test_evaluate_trajectory_helper.py similarity index 86% rename from tests/approaches/test_agent_sim_prompt_formatting.py rename to tests/approaches/test_evaluate_trajectory_helper.py index 9b81853698..3b0209d49b 100644 --- a/tests/approaches/test_agent_sim_prompt_formatting.py +++ b/tests/approaches/test_evaluate_trajectory_helper.py @@ -1,5 +1,5 @@ -"""Tests for the ``evaluate_trajectory`` helper the synthesis namespace offers, -and for the learn system prompt's deliverables.""" +"""Tests for the ``evaluate_trajectory`` helper the synthesis namespace +offers.""" # pylint: disable=protected-access,import-outside-toplevel,unused-import from __future__ import annotations @@ -8,7 +8,7 @@ # Bootstrap circular imports before pulling from predicators.approaches. from predicators import utils -from predicators.structs import Action, LowLevelTrajectory, State, Task, Type +from predicators.structs import Action, State, Task, Type @pytest.fixture(name="approach_cls") @@ -157,22 +157,3 @@ def _scope(physical_overrides=None): assert physics == {"friction": 0.5} # the scope restored the physics stub._identified_physical_sigma_points = [] assert fn(states, None, task_idx=0, physics_sweep=True)["sweep"] is None - - -def test_learn_message_ships_goal_required_mechanisms_as_hypotheses(): - """The learn message distinguishes a hypothesis the goal can do without - (record, do not ship) from one the goal REQUIRES (ship as a labelled - hypothesis with declared ParamSpecs and a first-ranked confirming - experiment). - - The rule lives in the learn system prompt's deliverables section. - """ - from predicators.agent_sdk import learn_prompts - prompt = learn_prompts.build_learn_system_prompt( - partially_observable=False, - residual_rule_signature="def residual_rule(state, updates, params):", - scene_viz_hint="render the scene") - assert "When the goal requires it" in prompt - assert "labelled hypothesis" in prompt - assert "first entry of `./open_questions.md`" in prompt - assert "A GO that rests on a hypothesized mechanism" in prompt diff --git a/tests/code_sim_learning/test_program_world_model.py b/tests/code_sim_learning/test_program_world_model.py index 0016cffc31..49557cab7c 100644 --- a/tests/code_sim_learning/test_program_world_model.py +++ b/tests/code_sim_learning/test_program_world_model.py @@ -91,8 +91,8 @@ def _ground_pick_place(options: Any, param: float) -> Any: def test_program_option_model_steps_and_carries_latent() -> None: - """Transitions write features, seed and advance the latent, honor the - override, and never mutate the input state.""" + """Transitions write features, seed and advance the latent, and never + mutate the input state.""" env, train_tasks, options, _ = _cover() model = ProgramOptionModel(_load(_HAND_PROGRAM, env, options), seed=0) init = train_tasks[0].init @@ -110,12 +110,6 @@ def test_program_option_model_steps_and_carries_latent() -> None: nxt2, _ = model.get_next_state_and_num_actions(nxt, option) assert nxt2.latent is not None assert nxt2.latent["count"] == nxt.latent["count"] + 1 - # The override pins every latent-less start. - model.initial_latent_override = {"count": 40} - nxt3, _ = model.get_next_state_and_num_actions(init, option) - assert nxt3.latent is not None - assert nxt3.latent["count"] == 41 - model.initial_latent_override = None assert model.last_execution_failure is None diff --git a/tests/test_agent_harness_fixes.py b/tests/test_agent_harness_fixes.py index 13be018edb..a7023f7d55 100644 --- a/tests/test_agent_harness_fixes.py +++ b/tests/test_agent_harness_fixes.py @@ -13,9 +13,8 @@ from predicators import utils from predicators.agent_sdk.session_base import max_session_log_number -from predicators.agent_sdk.tools.testing import _missing_goal_atoms from predicators.structs import Action, GroundAtom, Object, \ - ParameterizedOption, Predicate, State, Task, Type + ParameterizedOption, Predicate, State, Type def test_max_session_log_number(tmp_path: Path) -> None: @@ -98,14 +97,3 @@ def test_strip_latent_wait_targets_keeps_observable_atoms() -> None: assert wait.memory["wait_target_atoms"] == {GroundAtom(far, [block])} assert "wait_target_neg_atoms" not in wait.memory assert other.memory["wait_target_atoms"] == {GroundAtom(far, [block])} - - -def test_missing_goal_atoms_uses_the_goal_classifiers() -> None: - """Atoms that hold are not reported missing even when the goal predicates - are absent from the agent's predicate set.""" - near, far_block = Object("near", _block_type), Object("far", _block_type) - state = State({near: np.array([0.0]), far_block: np.array([1.0])}) - far = _geometric_pred() - goal = {GroundAtom(far, [near]), GroundAtom(far, [far_block])} - task = Task(state, goal) - assert _missing_goal_atoms(task, state) == {GroundAtom(far, [near])} diff --git a/tests/test_agent_sdk_tools.py b/tests/test_agent_sdk_tools.py index 94a57f082d..0f874e005e 100644 --- a/tests/test_agent_sdk_tools.py +++ b/tests/test_agent_sdk_tools.py @@ -1,12 +1,10 @@ """Tests for agent SDK tool enhancements. Validates: -1. submit_plan always saves scene images -2. submit_plan shows "Missing goal atoms" when goal not achieved -3. submit_plan shows object poses on failure -4. format_object_poses helper -5. render_scene_image helper -6. _sync_tool_context sets ctx.env from option model +1. the run_python probe (reset, run, refine, render) +2. format_object_poses helper +3. render_scene_image helper +4. _sync_tool_context sets ctx.env from option model Usage: python tests/test_agent_sdk_tools.py @@ -163,63 +161,6 @@ def _get_valid_option_plan_step(ctx: Any) -> dict[str, Any] | None: return None -def _plan_to_text(plan: Any, ctx: Any) -> str: - """Render structured option-plan steps as the text grammar that submit_plan - now expects (typed object refs + params in []).""" - type_of = {o.name: o.type.name for o in ctx.current_task.init} - lines = [] - for step in plan: - objs = ", ".join(f"{n}:{type_of.get(n, 'object')}" - for n in step["object_names"]) - params = ", ".join(str(p) for p in step["params"]) - lines.append(f"{step['option_name']}({objs})[{params}]") - return "\n".join(lines) - - -def test_option_plan_missing_goal_atoms(ctx: Any) -> None: - """submit_plan reports missing goal atoms when goal not achieved.""" - tools = _make_tools(ctx, ["submit_plan"]) - - step = _get_valid_option_plan_step(ctx) - assert step is not None, "No valid option found for testing" - plan = [step] - - result = _run(tools["submit_plan"]({ - "plan": _plan_to_text(plan, ctx), - "include_atoms": True, - })) - text = result["content"][0]["text"] - - # Three possible outcomes: - if "Goal achieved: False" in text: - # Either the env exposes goal atoms (and we show "Missing goal - # atoms: ...") or it sets goal_nl (and we show that instead, - # to avoid leaking env predicate names to predicate-invention - # agents). - assert ("Missing goal atoms:" in text - or "Goal (natural language):" in text) - print(" PASS: submit_plan (failure diagnostic shown)") - elif "Goal achieved: True" in text: - assert "Missing goal atoms:" not in text - print(" PASS: submit_plan (goal achieved, no missing atoms)") - else: - # Plan failed early (grounding error, NOT INITIABLE, etc.) - assert ("NOT INITIABLE" in text or "FAILURE REASON:" in text - or "EXECUTION ERROR" in text or "Failed to ground" in text) - print(" PASS: submit_plan (plan failed early, " - "goal check not reached)") - - -def test_option_plan_description_submission_split(ctx: Any) -> None: - """submit_plan's description routes exploration to the probe and frames - this tool as the submission path.""" - from predicators.agent_sdk.tools import create_mcp_tools - tool_obj = create_mcp_tools(ctx, tool_names=["submit_plan"])[0] - desc = getattr(tool_obj, "description", "") - assert "run_python" in desc and "SUBMIT" in desc - print(" PASS: submit_plan (submission-split description)") - - def test_run_python_render_annotations(ctx: Any) -> None: """sim.render(annotations=...) overlays temporary geometry for one render (bodies removed after) and surfaces bad annotations as loud errors.""" @@ -272,11 +213,10 @@ def test_run_python_exec_and_persistence(ctx: Any) -> None: def test_run_python_probe_sim(ctx: Any) -> None: """BeliefProbe: reset with mods, full-precision state, run from the - modified state, snapshot/restore - and nothing is ever captured.""" + modified state, snapshot/restore.""" tools = _make_tools(ctx, ["run_python"]) domino = next(o for o in ctx.current_task.init if o.type.name == "domino") robot = next(o for o in ctx.current_task.init if o.type.name == "robot") - ctx.capture_goal_reaching_plans = True prior_dir = ctx.image_save_dir try: with tempfile.TemporaryDirectory() as tmpdir: @@ -298,16 +238,15 @@ def test_run_python_probe_sim(ctx: Any) -> None: result = _run(tools["run_python"]({"code": code})) saved = [f for f in os.listdir(tmpdir) if f.endswith(".png")] finally: - ctx.capture_goal_reaching_plans = False ctx.image_save_dir = prior_dir text = result["content"][0]["text"] assert "modx 0.95" in text assert "steps 1" in text assert "restx 0.95" in text assert "natoms" in text - # sim.run saves the same per-step audit images submit_plan - # does, and reports their paths on each step; render=False (for - # tight sweep loops) skips the render entirely. + # sim.run saves per-step audit images and reports their paths on + # each step; render=False (for tight sweep loops) skips the render + # entirely. assert "quietimg None" in text if saved: assert any("probe_step_0_" in f for f in saved) @@ -315,20 +254,16 @@ def test_run_python_probe_sim(ctx: Any) -> None: assert len(saved) == 1 else: print(" NOTE: rendering not available, image save not checked") - # The probe carries no scoring surface: nothing it ran was captured. - assert ctx.solved_plan is None - print(" PASS: run_python (BeliefProbe reset/run/snapshot, no capture)") + print(" PASS: run_python (BeliefProbe reset/run/snapshot)") def test_run_python_probe_refine(ctx: Any) -> None: - """BeliefProbe.refine searches params from the current state, reports per- - step samples and a refined plan line, and captures nothing.""" + """BeliefProbe.refine searches params from the current state and reports + per-step samples and a refined plan line.""" tools = _make_tools(ctx, ["run_python"]) domino = next(o for o in ctx.current_task.init if o.type.name == "domino") robot = next(o for o in ctx.current_task.init if o.type.name == "robot") - ctx.capture_goal_reaching_plans = True - try: - code = f""" + code = f""" sim.reset() res = sim.refine( "Pick({robot.name}:robot, {domino.name}:domino)[0.06] " @@ -338,15 +273,12 @@ def test_run_python_probe_refine(ctx: Any) -> None: print("samples", res.total_samples, res.step_samples) print("line", res.plan_lines[0]) """ - result = _run(tools["run_python"]({"code": code})) - finally: - ctx.capture_goal_reaching_plans = False + result = _run(tools["run_python"]({"code": code})) text = result["content"][0]["text"] assert "success True" in text assert "samples" in text assert "line Pick(" in text and "Holding(" in text - assert ctx.solved_plan is None - print(" PASS: run_python (BeliefProbe.refine, no capture)") + print(" PASS: run_python (BeliefProbe.refine)") def test_run_python_probe_refine_verdict_line(ctx: Any) -> None: @@ -526,112 +458,6 @@ def test_run_python_run_contacts(ctx: Any) -> None: print(" PASS: run_python (contact recording)") -def test_option_plan_not_initiable_shows_poses(ctx: Any) -> None: - """submit_plan shows object poses when option is NOT INITIABLE.""" - tools = _make_tools(ctx, ["submit_plan"]) - - # Find Place option and try it without Pick first - place_opt = None - for opt in ctx.options: - if opt.name == "Place": - place_opt = opt - break - - if place_opt is None: - print(" SKIP: submit_plan (no Place option)") - return - - # Build object names from types - state = ctx.current_task.init - obj_names = [] - for t in place_opt.types: - for obj in state: - if obj.type == t and obj.name not in obj_names: - obj_names.append(obj.name) - break - - low = place_opt.params_space.low - high = place_opt.params_space.high - params = ((low + high) / 2).tolist() - - plan = [{ - "option_name": "Place", - "object_names": obj_names, - "params": params, - }] - - result = _run(tools["submit_plan"]({ - "plan": _plan_to_text(plan, ctx), - })) - text = result["content"][0]["text"] - - if "NOT INITIABLE" in text: - assert "Object poses at failure:" in text - print(" PASS: submit_plan (NOT INITIABLE shows poses)") - elif "Failed to ground" in text: - print(" SKIP: submit_plan (Place could not be grounded)") - else: - print(" SKIP: submit_plan (Place was initiable, " - "can't test NOT INITIABLE path)") - - -def test_option_plan_saves_images(ctx: Any) -> None: - """submit_plan always saves scene images (never returns inline).""" - with tempfile.TemporaryDirectory() as tmpdir: - ctx.image_save_dir = tmpdir - - tools = _make_tools(ctx, ["submit_plan"]) - - step = _get_valid_option_plan_step(ctx) - assert step is not None, "No valid option found for testing" - plan = [step] - - result = _run(tools["submit_plan"]({ - "plan": _plan_to_text(plan, ctx), - })) - - content = result["content"] - # Should have text block only (no inline images) - assert any(b["type"] == "text" for b in content) - assert not any(b["type"] == "image" for b in content) - - # Check files were saved if env rendering works - saved = [f for f in os.listdir(tmpdir) if f.endswith(".png")] - if saved: - print(f" PASS: submit_plan ({len(saved)} images saved)") - else: - print(" SKIP: submit_plan (rendering not available)") - - ctx.image_save_dir = None - - -def test_option_plan_failure_shows_poses(ctx: Any) -> None: - """submit_plan shows object poses when option returns 0 actions.""" - tools = _make_tools(ctx, ["submit_plan"]) - - step = _get_valid_option_plan_step(ctx) - assert step is not None, "No valid option found for testing" - plan = [step] - - result = _run(tools["submit_plan"]({ - "plan": _plan_to_text(plan, ctx), - })) - text = result["content"][0]["text"] - - # Check the output is well-formed — it should have either step info - # or a grounding error - assert ("Step 0:" in text or "Failed to ground" in text - or "Testing option plan" in text) - if "FAILURE REASON:" in text: - assert "Object poses at failure:" in text - print(" PASS: submit_plan (failure shows poses)") - elif "NOT INITIABLE" in text: - assert "Object poses at failure:" in text - print(" PASS: submit_plan (NOT INITIABLE shows poses)") - else: - print(" PASS: submit_plan (no failures in output)") - - def testformat_object_poses(ctx: Any) -> None: """format_object_poses formats object positions correctly.""" from predicators.agent_sdk.tools import format_object_poses @@ -732,21 +558,14 @@ def main() -> None: print("=== Tool Enhancement Tests ===\n") - # submit_plan tests - print("1. submit_plan tests:") - test_option_plan_missing_goal_atoms(ctx) - test_option_plan_not_initiable_shows_poses(ctx) - test_option_plan_saves_images(ctx) - test_option_plan_failure_shows_poses(ctx) - # Helper function tests - print("\n2. Helper function tests:") + print("1. Helper function tests:") testformat_object_poses(ctx) testrender_scene_image(ctx) test_render_scene_no_env(ctx) # _sync_tool_context test (creates fresh env) - print("\n3. Context sync tests:") + print("\n2. Context sync tests:") test_sync_tool_context_sets_env() print("\n=== All tests passed! ===") diff --git a/tests/test_docker_option_plan.py b/tests/test_docker_option_plan.py index 3c8b177f9d..c94db7ea2a 100644 --- a/tests/test_docker_option_plan.py +++ b/tests/test_docker_option_plan.py @@ -1,4 +1,4 @@ -"""Test that submit_plan produces correct results. +"""Test that multi-step option plans execute correctly. Validates that multi-step option plans (Pick→Place→Pick→Place→Push) produce non-zero actions at every step, both in-process and in a subprocess that