From f8871b87218d17e340760f0385e091e09501eaa4 Mon Sep 17 00:00:00 2001 From: Yichao Liang Date: Sat, 26 Sep 2026 14:35:41 -0400 Subject: [PATCH] Rename the predicatorv3 configs to empiric with rounds and subsets scripts/configs/empiric/ holds only the paper's experiment: the seven arms (approaches.yaml) on the five benchmark settings (envs.yaml) with the shared flags and the principled joint belief (common.yaml), which benchmark.yaml includes. With --round joint_r1 it resolves to the same 35 runs as continual_principled_belief_r1.yaml, over seeds 0-4. A launch names its round with --round or a ROUND key, which suffixes every experiment id, and launch.py refuses a continual launch without one, since those runs auto-resume from their run folders. --envs, --approaches and --seeds pick a subset. EXTENDS, which gave arms round-specific ids, is gone. Deleted: the phased exp_*.yaml configs and their all.yaml menus, oracle.yaml, the dated continual launchers, the menu entries outside the benchmark, relaunch_on_timeout.py (the self-requeue trap covers timeouts) and scripts/domino_debug/. random_actions_pybullet.yaml moves next to the ExoPredicator configs it belongs with. Tests that loaded dated launchers now load the benchmark; the arms outside it (scene package, real-to-sim, from assets, scene only, zero shot) keep their tests with their flags defined there. Doc links to deleted launchers point at the iclr-empiric-submission tag. --- docs/amps/empiric-from-assets.md | 4 +- docs/amps/fan-development.md | 6 +- .../oracle-fan-ramp-cohort-investigation.md | 2 +- docs/comparisons/empiric-validation-r2.md | 2 +- docs/comparisons/five-seed-launch-review.md | 2 +- docs/comparisons/ten-agent-opus-benchmark.md | 16 +- docs/envs/bridge/README.md | 2 +- docs/protocol/design.md | 8 +- docs/protocol/overview.md | 4 +- docs/uncertainty/principled-belief.md | 2 +- .../approaches/agent_model_based_approach.py | 3 +- .../task_generators/min_block_generation.py | 24 +- predicators/explorers/fixed_plan_explorer.py | 2 +- predicators/settings.py | 16 +- scripts/cluster_utils.py | 103 +++-- .../random_actions_pybullet.yaml | 2 +- scripts/configs/empiric/approaches.yaml | 83 ++++ scripts/configs/empiric/benchmark.yaml | 29 ++ .../common.yaml} | 26 +- scripts/configs/empiric/envs.yaml | 136 ++++++ .../configs/predicatorv3/approaches/all.yaml | 388 ------------------ .../predicatorv3/approaches/continual.yaml | 203 --------- scripts/configs/predicatorv3/common.yaml | 35 -- .../continual_benchmark_five_seeds.yaml | 38 -- ...inual_direct_scene_files_benchmark_r1.yaml | 28 -- .../continual_eight_agent_noisy_sweep.yaml | 55 --- .../continual_empiric_benchmark_r2.yaml | 26 -- ...al_empiric_scene_package_benchmark_r1.yaml | 29 -- .../continual_fan_inertial_baselines_r1.yaml | 20 - ...ontinual_fan_inertial_confirmation_r1.yaml | 15 - .../continual_fan_inertial_pilot_r1.yaml | 27 -- .../continual_fan_oracle_prompt_r2.yaml | 17 - ...nual_fan_oracle_prompt_r2_retry_seed0.yaml | 5 - ...nual_fan_oracle_prompt_r2_retry_seed3.yaml | 5 - .../continual_fan_ramp_confirmation_r1.yaml | 15 - .../continual_fan_ramp_long_landing_r1.yaml | 16 - ...ual_fan_ramp_low_drop_confirmation_r1.yaml | 5 - .../continual_fan_ramp_low_drop_r1.yaml | 17 - .../continual_fan_ramp_oracle_extra_r1.yaml | 10 - .../continual_fan_ramp_oracle_extra_r2.yaml | 10 - .../continual_fan_ramp_oracle_extra_r3.yaml | 10 - .../continual_fan_ramp_pilot_r1.yaml | 28 -- ...al_fan_ramp_skill_repair_baselines_r1.yaml | 20 - .../continual_fan_ramp_skill_repair_r1.yaml | 11 - ...al_fan_ramp_skill_repair_seed0_resume.yaml | 5 - .../continual_fan_transfer_pilot_r1.yaml | 27 -- ...continual_five_ablations_benchmark_r1.yaml | 42 -- .../continual_from_assets_expansion_r1.yaml | 29 -- .../continual_from_assets_pilot_r1.yaml | 37 -- ...inual_no_uncertainty_raw_obs_seed0_r1.yaml | 23 -- .../continual_oracle_validation_r2.yaml | 19 - .../continual_principled_belief_r1.yaml | 44 -- .../continual_principled_belief_smoke_r1.yaml | 28 -- .../continual_real_to_sim_benchmark_r1.yaml | 29 -- .../continual_scene_only_benchmark_r1.yaml | 30 -- ...ontinual_standalone_no_uncertainty_r2.yaml | 37 -- scripts/configs/predicatorv3/envs/all.yaml | 352 ---------------- .../configs/predicatorv3/envs/continual.yaml | 266 ------------ .../configs/predicatorv3/exp_boil_sweep.yaml | 119 ------ scripts/configs/predicatorv3/exp_bridge.yaml | 33 -- .../predicatorv3/exp_bridge_sweep.yaml | 119 ------ .../configs/predicatorv3/exp_busyboard.yaml | 30 -- scripts/configs/predicatorv3/exp_domino.yaml | 34 -- .../predicatorv3/exp_domino_heavy.yaml | 20 - .../configs/predicatorv3/exp_domino_real.yaml | 196 --------- .../exp_domino_real_geometry.yaml | 58 --- .../predicatorv3/exp_domino_real_replay.yaml | 65 --- .../predicatorv3/exp_domino_sweep.yaml | 118 ------ scripts/configs/predicatorv3/exp_fan.yaml | 30 -- .../configs/predicatorv3/exp_fan_sweep.yaml | 129 ------ scripts/configs/predicatorv3/oracle.yaml | 30 -- scripts/domino_debug/__init__.py | 0 scripts/domino_debug/count_turns.py | 91 ---- .../domino_debug/measure_turn_diversity.py | 185 --------- scripts/domino_debug/probe_cascade.py | 113 ----- scripts/domino_debug/probe_infront_drift.py | 116 ------ scripts/domino_debug/probe_min_block_bands.py | 180 -------- scripts/domino_debug/probe_real_scene.py | 287 ------------- .../render_domino_initial_states.py | 83 ---- .../render_unsolved_domino_states.py | 173 -------- .../domino_debug/replay_domino_sketches.py | 226 ---------- scripts/domino_debug/replay_ikval_sweep.py | 196 --------- scripts/domino_debug/replay_plan.py | 218 ---------- .../domino_debug/reproduce_domino_failures.py | 146 ------- scripts/dump_continual_arm_prompts.py | 2 +- scripts/engaging/launch.py | 82 +++- scripts/engaging/relaunch_on_timeout.py | 156 ------- scripts/local/generate_random_action_gifs.py | 4 +- scripts/local/render_init_state_gifs.py | 36 +- scripts/plotting/monitor_benchmark_arms.py | 3 +- tests/agent_sdk/test_prompt_goldens.py | 3 +- ...test_agent_continual_direct_scene_files.py | 11 +- ...st_agent_continual_real_to_sim_approach.py | 94 ++--- .../test_agent_continual_scene_package.py | 27 +- .../test_continual_comparison_approach.py | 39 +- .../test_oracle_process_planning_boil.py | 6 +- .../test_oracle_process_planning_bridge.py | 12 +- .../test_bridge_transfer_oracle.py | 54 +-- .../test_continual_oracle.py | 8 +- tests/envs/test_pybullet_fan_transfer.py | 13 +- tests/test_benchmark_plots.py | 12 +- tests/test_cluster_utils_configs.py | 201 ++++++--- tests/test_docker_option_plan.py | 2 +- tests/test_five_seed_benchmark.py | 54 --- 104 files changed, 717 insertions(+), 5570 deletions(-) rename scripts/configs/{predicatorv3 => ExoPredicator}/random_actions_pybullet.yaml (96%) create mode 100644 scripts/configs/empiric/approaches.yaml create mode 100644 scripts/configs/empiric/benchmark.yaml rename scripts/configs/{predicatorv3/continual_common.yaml => empiric/common.yaml} (72%) create mode 100644 scripts/configs/empiric/envs.yaml delete mode 100644 scripts/configs/predicatorv3/approaches/all.yaml delete mode 100644 scripts/configs/predicatorv3/approaches/continual.yaml delete mode 100644 scripts/configs/predicatorv3/common.yaml delete mode 100644 scripts/configs/predicatorv3/continual_benchmark_five_seeds.yaml delete mode 100644 scripts/configs/predicatorv3/continual_direct_scene_files_benchmark_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_eight_agent_noisy_sweep.yaml delete mode 100644 scripts/configs/predicatorv3/continual_empiric_benchmark_r2.yaml delete mode 100644 scripts/configs/predicatorv3/continual_empiric_scene_package_benchmark_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_inertial_baselines_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_inertial_confirmation_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_inertial_pilot_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2_retry_seed0.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2_retry_seed3.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_confirmation_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_long_landing_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_low_drop_confirmation_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_low_drop_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r2.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r3.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_pilot_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_baselines_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_seed0_resume.yaml delete mode 100644 scripts/configs/predicatorv3/continual_fan_transfer_pilot_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_five_ablations_benchmark_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_from_assets_expansion_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_from_assets_pilot_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_no_uncertainty_raw_obs_seed0_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_oracle_validation_r2.yaml delete mode 100644 scripts/configs/predicatorv3/continual_principled_belief_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_principled_belief_smoke_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_real_to_sim_benchmark_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_scene_only_benchmark_r1.yaml delete mode 100644 scripts/configs/predicatorv3/continual_standalone_no_uncertainty_r2.yaml delete mode 100644 scripts/configs/predicatorv3/envs/all.yaml delete mode 100644 scripts/configs/predicatorv3/envs/continual.yaml delete mode 100644 scripts/configs/predicatorv3/exp_boil_sweep.yaml delete mode 100644 scripts/configs/predicatorv3/exp_bridge.yaml delete mode 100644 scripts/configs/predicatorv3/exp_bridge_sweep.yaml delete mode 100644 scripts/configs/predicatorv3/exp_busyboard.yaml delete mode 100644 scripts/configs/predicatorv3/exp_domino.yaml delete mode 100644 scripts/configs/predicatorv3/exp_domino_heavy.yaml delete mode 100644 scripts/configs/predicatorv3/exp_domino_real.yaml delete mode 100644 scripts/configs/predicatorv3/exp_domino_real_geometry.yaml delete mode 100644 scripts/configs/predicatorv3/exp_domino_real_replay.yaml delete mode 100644 scripts/configs/predicatorv3/exp_domino_sweep.yaml delete mode 100644 scripts/configs/predicatorv3/exp_fan.yaml delete mode 100644 scripts/configs/predicatorv3/exp_fan_sweep.yaml delete mode 100644 scripts/configs/predicatorv3/oracle.yaml delete mode 100644 scripts/domino_debug/__init__.py delete mode 100644 scripts/domino_debug/count_turns.py delete mode 100644 scripts/domino_debug/measure_turn_diversity.py delete mode 100644 scripts/domino_debug/probe_cascade.py delete mode 100644 scripts/domino_debug/probe_infront_drift.py delete mode 100644 scripts/domino_debug/probe_min_block_bands.py delete mode 100644 scripts/domino_debug/probe_real_scene.py delete mode 100644 scripts/domino_debug/render_domino_initial_states.py delete mode 100644 scripts/domino_debug/render_unsolved_domino_states.py delete mode 100644 scripts/domino_debug/replay_domino_sketches.py delete mode 100644 scripts/domino_debug/replay_ikval_sweep.py delete mode 100644 scripts/domino_debug/replay_plan.py delete mode 100644 scripts/domino_debug/reproduce_domino_failures.py delete mode 100644 scripts/engaging/relaunch_on_timeout.py delete mode 100644 tests/test_five_seed_benchmark.py diff --git a/docs/amps/empiric-from-assets.md b/docs/amps/empiric-from-assets.md index 1b8d3062dd..2c935e1868 100644 --- a/docs/amps/empiric-from-assets.md +++ b/docs/amps/empiric-from-assets.md @@ -29,7 +29,7 @@ This is a reduction in supplied domain implementation, not reconstruction from r ## Pilot -The configuration is [continual_from_assets_pilot_r1.yaml](../../scripts/configs/predicatorv3/continual_from_assets_pilot_r1.yaml). +The configuration is [continual_from_assets_pilot_r1.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_from_assets_pilot_r1.yaml). It runs Opus 5 on seeds 0 and 1 of Fan + ramp, four-span Bridge, Domino, Balloons and Boil. There are ten intended runs total, each with its training and test levels. Fan uses the previously reviewed 3 mm ramp with the 10 cm landing extension and the repaired shared skills, not Fan maze. @@ -49,7 +49,7 @@ Bridge seeds 0-1 are array `23407427`; Fan maze seeds 0-1 are array `23407428`. The user immediately corrected Fan maze to Fan + ramp; both tasks of array `23407428` were cancelled with their logs preserved. The abandoned maze runs are not part of the intended pilot and must not be resumed or counted as task failures. Bridge array `23407427` remains unchanged. -The [expansion-only configuration](../../scripts/configs/predicatorv3/continual_from_assets_expansion_r1.yaml) submits only the eight new tasks, with Bridge and Fan maze disabled to prevent duplicates. +The [expansion-only configuration](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_from_assets_expansion_r1.yaml) submits only the eight new tasks, with Bridge and Fan maze disabled to prevent duplicates. Its runtime menus are pinned to the frozen implementation. The resolved Fan + ramp flags match the repaired EMPIRIC ramp cohort, apart from making the default fitting flag explicit. The eight replacement/additional tasks were submitted on account d as the following two-seed arrays: diff --git a/docs/amps/fan-development.md b/docs/amps/fan-development.md index 8a5dbc6ecd..583b7d28a7 100644 --- a/docs/amps/fan-development.md +++ b/docs/amps/fan-development.md @@ -303,20 +303,20 @@ The backup account `dat` also passed but is not in the active pool. Account `c` has an active limit marker until September 23 at 00:00 UTC and is excluded. Usage percentages were unavailable from the service endpoint; successful probes establish current access, not a guarantee of sufficient remaining quota for full runs. Each task requests 8 CPUs and 16 GB, with requeue enabled for preemption, time limits, and recognized account-limit exits. -The launch configuration is [continual_fan_ramp_skill_repair_r1.yaml](../../scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml). +The launch configuration is [continual_fan_ramp_skill_repair_r1.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml). The shared repair and its measured extra interaction cost are documented in [the switch investigation](fan-switch-seed4-investigation.md). Seed 0's original job stopped after account `a` reported that its organization had disabled subscription access for Claude Code. This was an infrastructure interruption after training succeeded and the test reached 207 steps, not a task failure. With the user's approval, job `23395354` resumes only seed 0 on `dat`, from the same frozen runtime, experiment key, run directory, sandbox, and recorded state. -The resume configuration is [continual_fan_ramp_skill_repair_seed0_resume.yaml](../../scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_seed0_resume.yaml). +The resume configuration is [continual_fan_ramp_skill_repair_seed0_resume.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_seed0_resume.yaml). ### Six matched comparison arms The user approved five seeds each for all six other paper agents on this repaired Fan ramp setup. All 30 tasks were verified running on compute nodes after submission. They use the same frozen runtime `ff11bc76f4652c6964e73beda7e41a6655d670d9`, reviewed geometry, observation noise, step budget, and preflight-off setting as the repaired EMPIRIC cohort. -The [six-arm configuration](../../scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_baselines_r1.yaml) resolves to exactly six five-seed arrays, with only the intended approach and ablation flag differences. +The [six-arm configuration](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_baselines_r1.yaml) resolves to exactly six five-seed arrays, with only the intended approach and ablation flag differences. - Oracle dynamics: array `23398858`, seeds 0-4, accounts b/d. - Direct agent: array `23398859`, seeds 0-4, account dat. diff --git a/docs/amps/oracle-fan-ramp-cohort-investigation.md b/docs/amps/oracle-fan-ramp-cohort-investigation.md index 33ad547a2d..8c91f2d588 100644 --- a/docs/amps/oracle-fan-ramp-cohort-investigation.md +++ b/docs/amps/oracle-fan-ramp-cohort-investigation.md @@ -90,7 +90,7 @@ The driver is [audit_oracle_ramp_20260922.py](/home/ycliang/predicators/logs/aud Seeds 9 and 10 were submitted as array `23463117` and started on compute nodes using accounts b and d. The launch flags and arguments were checked against the previous extra-seed config; only the seed range changes. -The [launch configuration](/home/ycliang/predicators/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r3.yaml) retains the frozen runtime and original cohort identifier. +The [launch configuration](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r3.yaml) retains the frozen runtime and original cohort identifier. The Markdown benchmark tracks all eleven seeds, retaining all failures. Replacement plot monitor `23463368` refreshes the report and figures every 60 seconds when results change. The expanded-cohort report tests passed: 8 tests. diff --git a/docs/comparisons/empiric-validation-r2.md b/docs/comparisons/empiric-validation-r2.md index 47da3f16c8..c2384b783d 100644 --- a/docs/comparisons/empiric-validation-r2.md +++ b/docs/comparisons/empiric-validation-r2.md @@ -1,7 +1,7 @@ # EMPIRIC r2: prospective robustness checks The requested cohort is two additional seeds per domain, seeds 3 and 4, across Boil, Domino, Balloons, Bridge, and Fan. -The launcher is [continual_empiric_benchmark_r2.yaml](../../scripts/configs/predicatorv3/continual_empiric_benchmark_r2.yaml). +The launcher is [continual_empiric_benchmark_r2.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_empiric_benchmark_r2.yaml). Its experiment key is `-mb_opus_benchmark_r2`, under `agent_continual`. The main figures retain historical EMPIRIC and show this cohort separately as **EMPIRIC r2**. All finished outcomes count, including failures; an unfinished run is not a failed run. diff --git a/docs/comparisons/five-seed-launch-review.md b/docs/comparisons/five-seed-launch-review.md index 2a7155730e..e7f43ae094 100644 --- a/docs/comparisons/five-seed-launch-review.md +++ b/docs/comparisons/five-seed-launch-review.md @@ -35,7 +35,7 @@ These checks do not establish future solve rates or resolve the previously docum ## Launch and reporting -The launcher is [continual_benchmark_five_seeds.yaml](../../scripts/configs/predicatorv3/continual_benchmark_five_seeds.yaml). +The launcher is [continual_benchmark_five_seeds.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_benchmark_five_seeds.yaml). All 60 destination seed directories were checked absent before submission. Jobs use `mit_preemptable`, checkpoint resume, automatic requeue, and account labels `a,b,c,d`, all confirmed usable by the user during this launch review. The user's account e corresponds to the launcher's `dat` label and is reserved as backup, not included in the normal rotation. diff --git a/docs/comparisons/ten-agent-opus-benchmark.md b/docs/comparisons/ten-agent-opus-benchmark.md index e8a8b9a3c8..00d1b45d2b 100644 --- a/docs/comparisons/ten-agent-opus-benchmark.md +++ b/docs/comparisons/ten-agent-opus-benchmark.md @@ -46,20 +46,20 @@ python ../../../scripts/plotting/plot_benchmark_arms.py benchmark-arms-opus ## Settings and cohort selection -The five settings are the menu defaults in [envs/continual.yaml](../../scripts/configs/predicatorv3/envs/continual.yaml): Boil two-jug test, Domino high-friction turn, Fan ramp test, Bridge four-span test row and Balloons composition test levels. +The five settings are the menu defaults in [envs/continual.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/envs/continual.yaml): Boil two-jug test, Domino high-friction turn, Fan ramp test, Bridge four-span test row and Balloons composition test levels. EMPIRIC and the direct agent ran before the benchmark rounds under their own round keys; the four-span EMPIRIC runs are the preflight-off ones (r2 seed 0, r3 seeds 1-2). The other arms ran as `_opus_benchmark_` from frozen worktrees: -- Oracle dynamics, zero-shot and no harness fitting: [continual_five_ablations_benchmark_r1.yaml](../../scripts/configs/predicatorv3/continual_five_ablations_benchmark_r1.yaml) at `2982f5876` (/home/ycliang/predicators-five-arms-frozen-20260918). -- Standalone simulator and no explicit uncertainty: round r2 from [continual_standalone_no_uncertainty_r2.yaml](../../scripts/configs/predicatorv3/continual_standalone_no_uncertainty_r2.yaml) at `0ddc8f7f4` (/home/ycliang/predicators-standalone-nounc-frozen-20260918), after the September 18 revision of both surfaces. -- Scene only: [continual_scene_only_benchmark_r1.yaml](../../scripts/configs/predicatorv3/continual_scene_only_benchmark_r1.yaml). -- Agentic real-to-sim: [continual_real_to_sim_benchmark_r1.yaml](../../scripts/configs/predicatorv3/continual_real_to_sim_benchmark_r1.yaml) at `5ea35c91c` (/home/ycliang/predicators-real-to-sim-frozen-20260918). +- Oracle dynamics, zero-shot and no harness fitting: [continual_five_ablations_benchmark_r1.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_five_ablations_benchmark_r1.yaml) at `2982f5876` (/home/ycliang/predicators-five-arms-frozen-20260918). +- Standalone simulator and no explicit uncertainty: round r2 from [continual_standalone_no_uncertainty_r2.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_standalone_no_uncertainty_r2.yaml) at `0ddc8f7f4` (/home/ycliang/predicators-standalone-nounc-frozen-20260918), after the September 18 revision of both surfaces. +- Scene only: [continual_scene_only_benchmark_r1.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_scene_only_benchmark_r1.yaml). +- Agentic real-to-sim: [continual_real_to_sim_benchmark_r1.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_real_to_sim_benchmark_r1.yaml) at `5ea35c91c` (/home/ycliang/predicators-real-to-sim-frozen-20260918). Its prompt asks the agent to write its own `simulator.py` scene from the engine, the manifest and the assets and to model the mechanisms, the same workflow as EMPIRIC. -- EMPIRIC + scene package: [continual_empiric_scene_package_benchmark_r1.yaml](../../scripts/configs/predicatorv3/continual_empiric_scene_package_benchmark_r1.yaml). +- EMPIRIC + scene package: [continual_empiric_scene_package_benchmark_r1.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_empiric_scene_package_benchmark_r1.yaml). Fan and Balloons ran as r1 at `d575a8447` (/home/ycliang/predicators-empiric-pkg-frozen-20260918). Boil, Bridge and Domino ran as r2 at `8051333e5` (/home/ycliang/predicators-empiric-pkg-split-frozen-20260919), after their environment files were split so the observable sim core can be shared without the hidden mechanisms; their r1 runs were cancelled and are excluded. To resume the paused seeds, relaunch the same launcher and round key from the same worktree so auto-resume picks up the scorecard and checkpoint. -- Direct agent + scene assets: [continual_direct_scene_files_benchmark_r1.yaml](../../scripts/configs/predicatorv3/continual_direct_scene_files_benchmark_r1.yaml) at `f6f609636` (/home/ycliang/predicators-direct-scene-files-frozen-20260919). +- Direct agent + scene assets: [continual_direct_scene_files_benchmark_r1.yaml](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_direct_scene_files_benchmark_r1.yaml) at `f6f609636` (/home/ycliang/predicators-direct-scene-files-frozen-20260919). It is the direct agent plus the real-to-sim arm's engine wrapper, scene manifest and URDF and mesh files as read-only references; its prompt only asks it to solve the levels, with no simulator, model files, fitting or model gate. The rendered prompt is in [the prompt review](../prompt-review/2026-09-19-direct-scene-files/direct_agent_scene_files.md). @@ -80,7 +80,7 @@ This is not a matched preflight ablation. The archived Fan transfer experiment is a separate pilot with two seeds each for EMPIRIC and the direct agent. The other agents have not been launched on this variant; their empty rows are missing results, not failures. Only finished seeds enter the bars and curves; pending runs are listed below and do not count as zero successes. -See the [illustrated task description](../amps/fan-exposed-transfer.md) and [launch configuration](../../scripts/configs/predicatorv3/continual_fan_transfer_pilot_r1.yaml). +See the [illustrated task description](../amps/fan-exposed-transfer.md) and [launch configuration](https://github.com/BasisResearch/predicators/blob/iclr-empiric-submission/scripts/configs/predicatorv3/continual_fan_transfer_pilot_r1.yaml). This superseded pilot is omitted from the figure; its tables and logs remain archived below. It is also excluded from the paper figure and its data selection. diff --git a/docs/envs/bridge/README.md b/docs/envs/bridge/README.md index ba000619e2..9b3850a5a1 100644 --- a/docs/envs/bridge/README.md +++ b/docs/envs/bridge/README.md @@ -15,5 +15,5 @@ Gray pads mark the two leg sites; standing blocks are legs, lying blocks are spa ## Oracle solve trajectories `oracle_solve_.mp4`: `oracle_process_planning` solving one test task end-to-end (seed 0), recorded with `--make_test_videos`. -The launch flags mirror the `bridge` entry in `scripts/configs/predicatorv3/envs/all.yaml`, plus `--no_repeated_arguments_in_grounding True` (set globally by `common.yaml` for config-launched runs, required on a bare CLI for the full spec to plan). +The launch flags mirror the `bridge` entry of the retired phased menu `scripts/configs/predicatorv3/envs/all.yaml` (tag `iclr-empiric-submission`), plus `--no_repeated_arguments_in_grounding True` (set globally by that menu's `common.yaml` for config-launched runs, required on a bare CLI for the full spec to plan). The full-spec video was recorded with the default `pybullet_birrt_path_subsample_ratio 1`. diff --git a/docs/protocol/design.md b/docs/protocol/design.md index bde2331281..8a36619c0a 100644 --- a/docs/protocol/design.md +++ b/docs/protocol/design.md @@ -56,7 +56,7 @@ There is no human oracle for these envs, our levels are not ordered by difficult | Scorecard | `scorecard.json` per run, aggregated across runs | | Recording JSONL and replay viewer | Per-level recording plus the existing trajectory viewer | | Swarm | One Slurm job per env and seed, plus an aggregator | -| Benchmarking harness (model configs x games, tags) | The launcher configs in `scripts/configs/predicatorv3/` | +| Benchmarking harness (model configs x games, tags) | The benchmark config in `scripts/configs/empiric/` | ## 4. Protocol definition @@ -297,7 +297,7 @@ The prompt gives no schedule and no certification rule. - `predicators/agent_sdk/prompts/play_system.md` and `play_query.md`: the continual prompts, with golden renders like the existing templates. - `predicators/approaches/agent_continual_approach.py`: `AgentSessionMixin` plus the tool context, the session loop, session resume, and `learn.run` as a sub-session launcher over the existing synthesis code. - `scripts/aggregate_scorecards.py`: scorecards to tables and curves, parameterised by the aggregation chosen later. -- `scripts/configs/predicatorv3/continual_common.yaml` plus the menus `envs/continual.yaml` and `approaches/continual.yaml`: a launcher includes them, un-parks one env and the arms it compares, and names each arm with `EXTENDS` (for example `continual_balloons_compose_r2.yaml`); one job per env and seed and arm. +- `scripts/configs/empiric/benchmark.yaml`: the seven arms (`approaches.yaml`) on the five settings (`envs.yaml`) with the shared flags (`common.yaml`); a launch names its round with `--round` or a `ROUND` key, which suffixes every experiment id, and `--envs`, `--approaches` and `--seeds` pick a subset; one job per env and arm, one array task per seed. ### 6.2 Entry point @@ -423,7 +423,7 @@ Tests: `tests/run/test_continual.py` pins the counts, the recordings, the preemp Launching: ```bash -python scripts/engaging/launch.py -c predicatorv3/continual_balloons_compose_r2.yaml --partition mit_preemptable +python scripts/engaging/launch.py -c empiric/benchmark.yaml --round r2 --envs balloons --partition mit_preemptable ``` The launcher passes `--auto_resume`, so a requeue resumes from the scorecard and the level recording in the run directory it adopts (section 4.7). @@ -460,7 +460,7 @@ Step 2, the agent arm, landed the same day: Tests: `tests/agent_sdk/test_continual_tools.py` drives the tools over a real session on cover (a win through `skills_execute_plan`, divergences on positive and `NOT` expectations, parse errors, game over then reset, give up and run end, the cap hit inside a tool); `tests/approaches/test_agent_continual_approach.py` runs the play loop on boil with a scripted agent in place of the LLM (a round that acts and a round that gives up, the continuation of the conversation by its recorded id, the attempts record, the checkpoint, the resume of an in-flight round after a preemption, the idle guard). -Launching the agent arm: write a launcher that includes `continual_common.yaml` and the menus and un-parks `agent_continual` with `EXTENDS` (see the header of `continual_common.yaml`). +Launching the agent arm: `--approaches mb_opus` on `empiric/benchmark.yaml` (see the header of `benchmark.yaml`). First agent result, boil seed 0, job 21964274 (2026-09-04): both levels won, 2127 steps and 5 resets on level 1 (one session, 129 turns, 85 skill invocations, 44 failed, one horizon game over), 374 steps and no reset on level 2, 55 min active, $24.66. The agent asked for no learning session and ran no model rollout: it measured the dynamics by probing the real environment, wrote a recipe into its journal, and replayed it on level 2. diff --git a/docs/protocol/overview.md b/docs/protocol/overview.md index d9ea1a28c3..1f55741a57 100644 --- a/docs/protocol/overview.md +++ b/docs/protocol/overview.md @@ -133,10 +133,10 @@ An audit of every transcript (all tool calls, including the Python the agents ra ## How to run and view -Launch (un-skip the arms you want in the yaml; each job requeues and resumes itself): +Launch (name the round; `--envs`, `--approaches` and `--seeds` pick a subset; each job requeues and resumes itself): ```bash -PYTHONPATH=. python scripts/engaging/launch.py -c predicatorv3/continual_balloons_compose_r2.yaml --partition mit_preemptable +PYTHONPATH=. python scripts/engaging/launch.py -c empiric/benchmark.yaml --round r2 --envs balloons --partition mit_preemptable ``` Add `--accounts a,b` to spread the runs over several Claude accounts (one token file per account under `~/.claude-tokens/`, see `scripts/engaging/claude_accounts.py`); each seed is assigned round-robin and the scorecard records which account it used. diff --git a/docs/uncertainty/principled-belief.md b/docs/uncertainty/principled-belief.md index aff9a15b6a..c5a6e7f3eb 100644 --- a/docs/uncertainty/principled-belief.md +++ b/docs/uncertainty/principled-belief.md @@ -106,7 +106,7 @@ Scripts and logs: `~/claude_sbatch/principled_belief_20260925/interpenetration_c ## Implementation status (2026-09-25) -Built on the `principled-belief` branch, behind `belief_joint_draws` (settings default 0; `continual_common.yaml` sets 16, and the No uncertainty arm keeps 0). +Built on the `principled-belief` branch, behind `belief_joint_draws` (settings default 0; `scripts/configs/empiric/common.yaml` sets 16, and the No uncertainty arm keeps 0). - Parameter factor: `ParameterBelief` with line posteriors, discrete parameters as categorical lines (`DiscretePosterior`), fresh draws (`ParameterBelief.sample`), and `prior_parameter_belief` before a fit; the fit tool reports the belief in place of the identifiability section. - State factor: `change_point_belief` and `feature_ranges` in `observation_belief.py`; `ContinualRun.belief()` uses it when the joint belief is on, for every arm with a state estimate. diff --git a/predicators/approaches/agent_model_based_approach.py b/predicators/approaches/agent_model_based_approach.py index 1b91dad592..4b6c11676a 100644 --- a/predicators/approaches/agent_model_based_approach.py +++ b/predicators/approaches/agent_model_based_approach.py @@ -766,8 +766,7 @@ def _refine_sketch( first passes the task through :meth:`_attach_initial_latent` so partially-observable approaches can seed ``task.init.latent`` with the initial latent block. Used by mid-episode suffix - replans and by the offline replay scripts under - ``scripts/domino_debug/``. + replans. ``attempt`` perturbs the RNG so retries explore different samples - without it, refinement is deterministic in diff --git a/predicators/envs/pybullet_domino/task_generators/min_block_generation.py b/predicators/envs/pybullet_domino/task_generators/min_block_generation.py index 2c4607dded..1ec0cc17ee 100644 --- a/predicators/envs/pybullet_domino/task_generators/min_block_generation.py +++ b/predicators/envs/pybullet_domino/task_generators/min_block_generation.py @@ -69,8 +69,9 @@ def _domino_code_digest() -> str: # Default L-shape leg-sampling bands (entry_leg, exit_leg) for turn tasks # when no explicit differentiating band is configured - the natural-corner # region shared by the plain and heavy turn variants. Explicit differentiating -# bands come from CFG.domino_min_block_turn_* (probe with -# scripts/domino_debug/probe_min_block_bands.py when the friction pair moves). +# bands come from CFG.domino_min_block_turn_* (re-probe them when the friction +# pair moves, with scripts/domino_debug/probe_min_block_bands.py from tag +# iclr-empiric-submission). _DEFAULT_TURN_ENTRY_BAND = (0.26, 0.34) _DEFAULT_TURN_EXIT_BAND = (0.18, 0.26) # Heavy turn variant: legs from the 2026-07-25 canonical-anchor design @@ -557,11 +558,12 @@ def _make_turn_task(env: "PyBulletDominoComposedEnv", # finite (entry, exit) cells - the data for any future retuning. direction = _planning_mismatch_direction() if CFG.domino_min_block_turn_entry_lo is not None: - # Explicitly configured bands - probe with - # scripts/domino_debug/probe_min_block_bands.py when the - # friction pair changes (the differentiating cells move with the - # frictions). The shipping under_reach arm's bands live in - # scripts/configs/predicatorv3/envs/all.yaml. + # Explicitly configured bands - re-probe them when the friction + # pair changes (the differentiating cells move with the + # frictions), with scripts/domino_debug/probe_min_block_bands.py + # from tag iclr-empiric-submission. The benchmark's bands live in + # the domino_high_friction_turn entry of + # scripts/configs/empiric/envs.yaml. assert CFG.domino_min_block_turn_entry_hi is not None assert CFG.domino_min_block_turn_exit_lo is not None assert CFG.domino_min_block_turn_exit_hi is not None @@ -580,9 +582,11 @@ def _make_turn_task(env: "PyBulletDominoComposedEnv", # which LLM planner arms cannot discover (retune 2026-07-12). raise ValueError( "under_reach turn tasks require explicit " - "domino_min_block_turn_{entry,exit}_{lo,hi} flags (probe with " - "scripts/domino_debug/probe_min_block_bands.py; see the " - "domino_high_friction block in envs/all.yaml).") + "domino_min_block_turn_{entry,exit}_{lo,hi} flags (see the " + "domino_high_friction_turn entry of " + "scripts/configs/empiric/envs.yaml; probe new bands with " + "scripts/domino_debug/probe_min_block_bands.py from tag " + "iclr-empiric-submission).") else: entry_leg = round(float(rng.uniform(*_DEFAULT_TURN_ENTRY_BAND)), 2) exit_leg = round(float(rng.uniform(*_DEFAULT_TURN_EXIT_BAND)), 2) diff --git a/predicators/explorers/fixed_plan_explorer.py b/predicators/explorers/fixed_plan_explorer.py index 85c8ae9b44..2ad08d0164 100644 --- a/predicators/explorers/fixed_plan_explorer.py +++ b/predicators/explorers/fixed_plan_explorer.py @@ -28,7 +28,7 @@ from predicators.settings import CFG from predicators.structs import ExplorationStrategy, Object, State, _Option -# Same grammar as scripts/domino_debug/replay_plan.py. +# One option per line: ``Name(obj, ...) [param, ...]``. _LINE = re.compile(r"^\s*(\w+)\s*\(([^)]*)\)\s*\[([^\]]*)\]") diff --git a/predicators/settings.py b/predicators/settings.py index a069d68386..c1339787a4 100644 --- a/predicators/settings.py +++ b/predicators/settings.py @@ -1097,9 +1097,10 @@ class GlobalSettings: # from [lo, hi] per attempt. None (default) = the legacy over_reach band # hardcoded in _make_turn_task (the low-friction arm); under_reach arms # MUST set these explicitly - the legacy under_reach band shipped - # agent-intractable pair-corner tasks. Probe candidate bands with - # scripts/domino_debug/probe_min_block_bands.py whenever the friction - # pair changes (the differentiating cells move with the frictions). + # agent-intractable pair-corner tasks. Probe candidate bands whenever + # the friction pair changes (the differentiating cells move with the + # frictions), with scripts/domino_debug/probe_min_block_bands.py from + # tag iclr-empiric-submission. domino_min_block_turn_entry_lo: Optional[float] = None domino_min_block_turn_entry_hi: Optional[float] = None domino_min_block_turn_exit_lo: Optional[float] = None @@ -1166,8 +1167,8 @@ class GlobalSettings: fan_train_num_pos_y = 3 # The historical 6 x 6 uniform test split. The loc bounds in # pybullet_fan.py admit at most 10 x 9 cells at the 8 cm pitch, which - # fills the arena up to the fan rows; the maze split in - # scripts/configs/predicatorv3/envs/all.yaml uses that full grid. + # fills the arena up to the fan rows; the historical maze split used + # that full grid. fan_test_num_pos_x = 6 fan_test_num_pos_y = 6 fan_train_num_walls_per_task = [1] @@ -2353,9 +2354,8 @@ class GlobalSettings: # agent adds design margin in-session. Runs only when the approach # installs a fresh-env scope (perturbing the shared env would leak) # and a fit with nonzero posterior width has been applied. Default - # False so existing arms keep their behavior; the main arm - # (approaches/all.yaml sim_predicator) - # turns it on. + # False; no benchmark arm turns it on (the retired phased + # sim_predicator arm did). agent_plan_validation_physics_margin = False # Number of grid points the margin gate (and the sim.run physics # sweep) spreads evenly across the +-1-sigma range, endpoints diff --git a/scripts/cluster_utils.py b/scripts/cluster_utils.py index 5b982f0144..52ac17eef1 100644 --- a/scripts/cluster_utils.py +++ b/scripts/cluster_utils.py @@ -2,10 +2,11 @@ import copy import os +import re import shlex import subprocess from dataclasses import dataclass -from typing import Any, Dict, Iterator, List, Optional, Tuple +from typing import Any, Collection, Dict, Iterator, List, Optional, Tuple import yaml @@ -131,30 +132,36 @@ def _resolve_config(config_filepath: str) -> Dict[str, Any]: return merged -def _resolve_extends(section: Dict[str, Any]) -> Dict[str, Any]: - """Expand ``EXTENDS`` entries of an ENVS or APPROACHES section. +def parse_seed_range(text: str) -> Tuple[int, int]: + """Parse a seed range as (start, count): "3" is seed 3 alone, "2-4" is + seeds 2, 3 and 4.""" + match = re.fullmatch(r"(\d+)(?:-(\d+))?", text.strip()) + if match is None: + raise ValueError(f"seed range {text!r} is not N or N-M") + start = int(match.group(1)) + stop = int(match.group(2)) if match.group(2) is not None else start + if stop < start: + raise ValueError(f"seed range {text!r} ends before it starts") + return start, stop - start + 1 - An entry ``{EXTENDS: base, FLAGS: {...}}`` is the ``base`` entry of - the same section with the entry's own keys deep-merged on top, and - it is un-parked (``SKIP: False``) unless it says otherwise. This is - how a launcher gives a menu arm a round-specific experiment id: the - id is the entry's key, the definition stays in the menu. + +def _select(section: Dict[str, Any], keys: Optional[Collection[str]], + kind: str) -> Dict[str, Any]: + """The entries of an ENVS or APPROACHES section that a launch runs: + + the ``keys`` named on the command line, whether or not the config + parks them, else every entry the config does not SKIP. """ - resolved: Dict[str, Any] = {} - for key, entry in section.items(): - base_key = entry.get("EXTENDS") - if base_key is None: - resolved[key] = entry - continue - if base_key not in section: - raise ValueError(f"{key} EXTENDS unknown entry {base_key}") - if "EXTENDS" in section[base_key]: - raise ValueError(f"{key} EXTENDS {base_key}, which itself " - "EXTENDS another entry; extend menu entries") - derived = {k: v for k, v in entry.items() if k != "EXTENDS"} - merged = _deep_merge(section[base_key], {"SKIP": False}) - resolved[key] = _deep_merge(merged, derived) - return resolved + if keys is None: + return { + key: entry + for key, entry in section.items() if not entry.get("SKIP", False) + } + unknown = sorted(set(keys) - set(section)) + if unknown: + raise ValueError(f"unknown {kind} {unknown}; the config defines " + f"{sorted(section)}") + return {key: entry for key, entry in section.items() if key in keys} def parse_configs(config_filename: str) -> Iterator[Dict[str, Any]]: @@ -177,11 +184,25 @@ def parse_configs(config_filename: str) -> Iterator[Dict[str, Any]]: def generate_run_configs(config_filename: str, - batch_seeds: bool = False) -> Iterator[RunConfig]: - """Generate run configs from a (local path) config file.""" + batch_seeds: bool = False, + round_name: Optional[str] = None, + envs: Optional[Collection[str]] = None, + approaches: Optional[Collection[str]] = None, + seeds: Optional[Tuple[int, int]] = None, + require_round: bool = False) -> Iterator[RunConfig]: + """Generate run configs from a (local path) config file. + + The experiment id is ``-``, suffixed with + ``_`` when the launch names a round: ``round_name``, else the + config's ROUND key. ``envs`` and ``approaches`` pick entries by key + and ``seeds`` is a (start, count) pair that overrides START_SEED and + NUM_SEEDS. With ``require_round``, a continual-protocol run without + a round is an error: its runs auto-resume from their run folders, so + an unnamed relaunch would resume the previous launch. + """ for config in parse_configs(config_filename): - start_seed = config["START_SEED"] - num_seeds = config["NUM_SEEDS"] + start_seed, num_seeds = seeds or (config["START_SEED"], + config["NUM_SEEDS"]) args = config["ARGS"] flags = config["FLAGS"] if "USE_GPU" in config.keys(): @@ -196,20 +217,23 @@ def generate_run_configs(config_filename: str, train_refinement_estimator = config["TRAIN_REFINEMENT_ESTIMATOR"] else: train_refinement_estimator = False - approaches = _resolve_extends(config["APPROACHES"]) - envs = _resolve_extends(config["ENVS"]) + launch_round = round_name or config.get("ROUND") + if launch_round is not None and not re.fullmatch( + r"[A-Za-z0-9][A-Za-z0-9_]*", str(launch_round)): + raise ValueError(f"round {launch_round!r} must be letters, " + "digits and underscores") + suffix = f"_{launch_round}" if launch_round is not None else "" + selected_approaches = _select(config["APPROACHES"], approaches, + "approaches") + selected_envs = _select(config["ENVS"], envs, "envs") # Loop over approaches. - for approach_exp_id, approach_config in approaches.items(): - if approach_config.get("SKIP", False): - continue + for approach_exp_id, approach_config in selected_approaches.items(): approach = approach_config["NAME"] # Loop over envs. - for env_exp_id, env_config in envs.items(): - if env_config.get("SKIP", False): - continue + for env_exp_id, env_config in selected_envs.items(): env = env_config["NAME"] # Create the experiment ID, args, and flags. - experiment_id = f"{env_exp_id}-{approach_exp_id}" + experiment_id = f"{env_exp_id}-{approach_exp_id}{suffix}" run_args = list(args) if "ARGS" in approach_config: run_args.extend(approach_config["ARGS"]) @@ -220,6 +244,13 @@ def generate_run_configs(config_filename: str, run_flags.update(approach_config["FLAGS"]) if "FLAGS" in env_config: run_flags.update(env_config["FLAGS"]) + if (require_round and not suffix and + run_flags.get("experiment_protocol") == "continual"): + raise ValueError( + f"{experiment_id} runs the continual protocol, " + "whose runs auto-resume from their run folders: " + "name a round (--round or the config's ROUND) so " + "this launch cannot resume an earlier one") # Loop or batch over seeds. if batch_seeds: yield BatchSeedRunConfig(experiment_id, approach, env, diff --git a/scripts/configs/predicatorv3/random_actions_pybullet.yaml b/scripts/configs/ExoPredicator/random_actions_pybullet.yaml similarity index 96% rename from scripts/configs/predicatorv3/random_actions_pybullet.yaml rename to scripts/configs/ExoPredicator/random_actions_pybullet.yaml index 150d7fec43..5da4d1b9cb 100644 --- a/scripts/configs/predicatorv3/random_actions_pybullet.yaml +++ b/scripts/configs/ExoPredicator/random_actions_pybullet.yaml @@ -1,6 +1,6 @@ # Random actions agent on all PyBullet environments - generates test videos. # Usage: -# PYTHONPATH=. python scripts/local/launch_simp.py -c mara2/random_actions_pybullet.yaml +# PYTHONPATH=. python scripts/local/launch_simp.py -c ExoPredicator/random_actions_pybullet.yaml --- APPROACHES: random_actions_pybullet: diff --git a/scripts/configs/empiric/approaches.yaml b/scripts/configs/empiric/approaches.yaml new file mode 100644 index 0000000000..bd0f3d45cf --- /dev/null +++ b/scripts/configs/empiric/approaches.yaml @@ -0,0 +1,83 @@ +# The seven arms of the EMPIRIC benchmark. Each key is the approach half +# of the experiment id (-_). The shared flags in +# common.yaml give every arm the principled joint belief; No uncertainty +# turns it off. Arms gated on a deployable model on the test level set +# continual_require_model_on_test. +APPROACHES: + # EMPIRIC: the agent writes a model of the world's physics, the harness + # fits its parameters and keeps their uncertainty. + mb_opus: + NAME: agent_continual + FLAGS: + agent_sdk_model_name: claude-opus-5 + continual_require_model_on_test: true + # Model-free: the same agent with the model-side machinery switched + # off, so the two differ only in the model. + mf_opus: + NAME: agent_continual_model_free + FLAGS: &mf_flags + agent_sdk_model_name: claude-opus-5 + code_sim_learning_interval_belief: false + agent_explorer_info_seeking_noise_aware: false + code_sim_learning_rollout_noise_filter: false + code_sim_learning_carry_posterior: false + code_sim_learning_fit_evidence: false + continual_belief_frame: false + agent_model_repair: false + agent_planner_use_simulator: false + continual_uncertainty_decisions: false + # Model-free with the scene files an agentic real-to-sim agent would + # build from: the engine wrapper, the scene manifest and the URDF and + # mesh files as read-only references. Its prompt only asks it to solve + # the levels; nothing asks for a simulator or model. + mf_scene_package_opus: + NAME: agent_continual_model_free + FLAGS: + <<: *mf_flags + continual_provide_scene_package: true + # Standalone: the agent writes and owns a program world model, with no + # harness fitting. + standalone_opus: + NAME: agent_continual_program_world_model + FLAGS: + agent_sdk_model_name: claude-opus-5 + # Oracle dynamics: the ground-truth simulator stands in for the learned + # model, with no automatic execution gate or shadow audit (see + # docs/comparisons/oracle-repair-notes.md). + oracle_dynamics_opus: + NAME: agent_continual_oracle_dynamics + FLAGS: + agent_sdk_model_name: claude-opus-5 + continual_skill_preflight: false + continual_validation_audit: false + # No fitting: EMPIRIC with only its declared parameters and no harness + # parameter fitting (the agent may fit in its own sandbox code). + no_fitting_opus: + NAME: agent_continual_no_fitting + FLAGS: + agent_sdk_model_name: claude-opus-5 + continual_require_model_on_test: true + agent_sim_learn_declared_params_only: true + # No uncertainty: EMPIRIC fitting raw noisy observations, with no state + # smoothing, uncertainty margins, information seeking or joint belief. + no_uncertainty_opus: + NAME: agent_continual_no_uncertainty + FLAGS: + agent_sdk_model_name: claude-opus-5 + continual_require_model_on_test: true + code_sim_learning_rollout_noise_filter: false + continual_belief_frame: false + continual_uncertainty_decisions: false + agent_sim_learn_param_uncertainty: false + agent_plan_validation_rule_param_margin: false + agent_plan_validation_physics_margin: false + agent_explorer_info_seeking: false + agent_explorer_info_seeking_adaptive: false + agent_explorer_info_seeking_noise_aware: false + code_sim_learning_interval_belief: false + code_sim_learning_carry_posterior: false + belief_joint_draws: 0 + # Nothing about the observation noise is declared to this arm: no + # noise section, no [noise] line, and the harness fit models none + # of it. + continual_obs_noise_declared: false diff --git a/scripts/configs/empiric/benchmark.yaml b/scripts/configs/empiric/benchmark.yaml new file mode 100644 index 0000000000..2e03f5f637 --- /dev/null +++ b/scripts/configs/empiric/benchmark.yaml @@ -0,0 +1,29 @@ +# The EMPIRIC benchmark: the seven arms (approaches.yaml) on the five +# settings (envs.yaml), five seeds each, with the principled joint belief +# (common.yaml). +# +# Every launch names a round, which suffixes the experiment ids +# (-_, the run folders under continual_runs_dir), so a +# new round never auto-resumes an old one: +# +# python scripts/engaging/launch.py -c empiric/benchmark.yaml \ +# --round r2 --partition mit_preemptable --accounts b,c,d +# +# --envs, --approaches and --seeds launch a subset: +# +# python scripts/engaging/launch.py -c empiric/benchmark.yaml \ +# --round fan_fix_r1 --envs fan --approaches mb_opus --seeds 2-4 +# +# A launch worth keeping on record gets its own file next to this one, +# which includes this file and sets ROUND (and SKIP entries or seeds): +# +# includes: [benchmark.yaml] +# ROUND: fan_fix_r1 +# START_SEED: 2 +# NUM_SEEDS: 3 +includes: + - common.yaml + - envs.yaml + - approaches.yaml +START_SEED: 0 +NUM_SEEDS: 5 diff --git a/scripts/configs/predicatorv3/continual_common.yaml b/scripts/configs/empiric/common.yaml similarity index 72% rename from scripts/configs/predicatorv3/continual_common.yaml rename to scripts/configs/empiric/common.yaml index 2e29ed87c7..ce700bf27c 100644 --- a/scripts/configs/predicatorv3/continual_common.yaml +++ b/scripts/configs/empiric/common.yaml @@ -1,23 +1,9 @@ -# Shared base of every continual-protocol launch (the agent plays levels -# of one environment, train then test, through the sandboxed skills). -# A launcher includes this file plus the menus envs/continual.yaml and -# approaches/continual.yaml, un-parks one env and the arms it compares, -# and gives each arm a round-specific experiment id with EXTENDS: -# -# includes: [continual_common.yaml, envs/continual.yaml, -# approaches/continual.yaml] -# ENVS: -# balloons: {SKIP: False} -# APPROACHES: -# mb_opus_compose_r2: {EXTENDS: mb_opus} -# mf_opus_compose_r2: {EXTENDS: mf_opus} -# -# The experiment id is "-" (the run directory under -# continual_runs_dir). Env FLAGS override approach FLAGS, which override -# these. Launch with -# python scripts/engaging/launch.py -c predicatorv3/.yaml --partition mit_preemptable --accounts b,c -START_SEED: 0 -NUM_SEEDS: 1 +# Arguments and flags every EMPIRIC run shares: the continual protocol +# (the agent plays the levels of one environment, train then test, +# through the sandboxed skills), observation noise, and the principled +# joint belief that every arm but No uncertainty carries. +# benchmark.yaml includes this file with envs.yaml and approaches.yaml. +# Env FLAGS override approach FLAGS, which override these. ARGS: - debug - make_failure_videos diff --git a/scripts/configs/empiric/envs.yaml b/scripts/configs/empiric/envs.yaml new file mode 100644 index 0000000000..7584e1bafd --- /dev/null +++ b/scripts/configs/empiric/envs.yaml @@ -0,0 +1,136 @@ +# The five benchmark settings. Each key is the env half of the experiment +# id (-_). Env FLAGS override the approach's and the +# common ones, so per-domain noise levels and level budgets live here. +ENVS: + # Balloons: composition test levels (every test level composes lifts + # the agent measured on training racks; the weakest-first release + # bursts) with the 25-step goal dwell. + balloons: + NAME: pybullet_balloons + FLAGS: + max_initial_demos: 0 + horizon: 1500 + sesame_check_expected_atoms: false + pybullet_birrt_path_subsample_ratio: 2 + balloons_require_jam_decoy: false + balloons_goal_dwell_steps: 25 + continual_obs_noise_position: 0.01 + continual_obs_noise_orientation: 0.02 + continual_obs_noise_scalar: 0.0 + continual_steps_per_level: 5000 + num_train_tasks: 2 + # Bridge: a three-span training row and a four-span test row, with the + # Sept 16 four-span repairs (rigid grasp, lift-first transit, a + # certificate that waits for the robot to withdraw). + bridge: + NAME: pybullet_bridge + FLAGS: + max_initial_demos: 0 + horizon: 3000 + max_num_steps_interaction_request: 2000 + skill_place_settle_preload_force: 3.0 + process_planning_heuristic_weight: 10.0 + process_planning_max_execution_replans: 3 + wait_option_max_steps: 120 + pybullet_birrt_contact_margin: -0.005 + pybullet_pin_held_weld_assemblies: true + continual_obs_noise_position: 0.005 + continual_obs_noise_orientation: 0.02 + continual_obs_noise_scalar: 0.0 + continual_steps_per_level: 10000 + num_train_tasks: 1 + bridge_train_span_blocks: 3 + bridge_test_span_blocks: 4 + pybullet_grasp_max_force: 10000.0 + bridge_lift_before_transit: true + bridge_goal_robot_clearance: 0.01 + # Boil: one jug in training, two jugs on the test level, with the 5 cm + # faucet tolerance. + boil: + NAME: pybullet_boil + FLAGS: + max_initial_demos: 0 + excluded_objects_in_state_str: switch + max_num_steps_option_rollout: 100 + horizon: 500 + boil_goal: simple + boil_require_jug_full_to_heatup: true + script_option_file_name: boil.txt + boil_water_fill_speed: 0.0015 + pybullet_birrt_path_subsample_ratio: 2 + boil_num_jugs_train: [1] + boil_num_jugs_test: [2] + boil_num_burner_train: [1] + boil_num_burner_test: [1] + boil_faucet_align_threshold: 0.05 + continual_obs_noise_position: 0.0125 + continual_obs_noise_orientation: 0.05 + continual_obs_noise_scalar: 0.07 + continual_steps_per_level: 5000 + num_train_tasks: 1 + # Fan: uniform training positions, and a test level that turns on the + # exposed, inertial and ramp transfers (a 3 mm ramp with a longer + # landing). + fan: + NAME: pybullet_fan + FLAGS: + max_initial_demos: 0 + excluded_objects_in_state_str: switch + terminate_on_goal_reached: true + process_planning_heuristic_weight: 10.0 + horizon: 500 + pybullet_birrt_path_subsample_ratio: 2 + process_planning_max_execution_replans: 3 + fan_train_num_pos_x: 3 + fan_train_num_pos_y: 3 + fan_exposed_transfer: true + fan_inertial_transfer: true + fan_ramp_transfer: true + fan_ramp_rise: 0.003 + fan_ramp_landing_extension: 0.10 + fan_train_num_walls_per_task: [0] + fan_test_num_walls_per_task: [0] + fan_test_num_pos_x: 3 + fan_test_num_pos_y: 3 + fan_train_task_generation: uniform + fan_test_task_generation: uniform + continual_obs_noise_position: 0.005 + continual_obs_noise_orientation: 0.02 + continual_obs_noise_scalar: 0.0 + continual_steps_per_level: 5000 + num_train_tasks: 1 + # Domino: min-block friction tasks with a turn on the test level, true + # friction above the planner's belief. + domino_high_friction_turn: + NAME: pybullet_domino + FLAGS: + max_initial_demos: 0 + excluded_objects_in_state_str: loc,rot,angle,direction + horizon: 500 + domino_initialize_at_finished_state: false + domino_use_domino_blocks_as_target: true + domino_use_continuous_place: true + process_planning_heuristic_weight: 2.0 + domino_has_glued_dominos: false + keep_failed_demos: true + predicate_invent_invent_derived_predicates: true + pybullet_birrt_extend_num_interp: 20 + pybullet_birrt_path_subsample_ratio: 2 + domino_min_block_tasks: true + domino_true_friction: 0.5 + domino_planning_friction: 0.1 + domino_min_block_span_lo: 0.29 + domino_min_block_span_hi: 0.31 + domino_min_block_turn_entry_lo: 0.21 + domino_min_block_turn_entry_hi: 0.24 + domino_min_block_turn_exit_lo: 0.17 + domino_min_block_turn_exit_hi: 0.2 + domino_min_block_num_blues: 4 + domino_block_cost: 0.1 + domino_test_turn_ratio: 1.0 + online_learning_early_stopping_ignore_reward_bar: true + continual_obs_noise_position: 0.01 + continual_obs_noise_orientation: 0.04 + continual_obs_noise_scalar: 0.0 + continual_steps_per_level: 5000 + num_train_tasks: 1 diff --git a/scripts/configs/predicatorv3/approaches/all.yaml b/scripts/configs/predicatorv3/approaches/all.yaml deleted file mode 100644 index 97b0f0748c..0000000000 --- a/scripts/configs/predicatorv3/approaches/all.yaml +++ /dev/null @@ -1,388 +0,0 @@ -# Canonical arm menu (paper ids on each entry, C7 dropped); all parked, the -# exp_*.yaml launchers un-skip. List-valued FLAGS live only here (merges concat). -APPROACHES: - - # ---- OURS ---- - - # C1: PO, no GT; learns predicates, hybrid sim and params, with the - # physics-margin gate. Its FLAGS block is the anchor the ABLATIONS merge. - sim_predicator: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: &ours_flags - demonstrator: "oracle_process_planning" - explorer: "agent_model_based" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - agent_sim_learn_kept_predicates_names: ["Holding"] - partially_observable: True - agent_explorer_info_seeking: True - execution_monitor: "subgoal_annotations" - # Open-loop execution: suffix replanning never recovered a bridge - # divergence (0/4, 2026-09-01). 0 also disarms the divergence monitor. - agent_bilevel_max_execution_replans: 0 - agent_bilevel_use_llm_initial_params: True # LLM proposes params - agent_sdk_max_agent_turns_per_iteration: 200 - agent_sdk_image_max_px: 900 - agent_solve_max_attempts: 1 - agent_solve_attempt_wall_clock: 2700 - agent_solve_fresh_context: True - agent_solve_use_journal: True - agent_plan_validation_physics_margin: True - agent_plan_validation_rule_param_margin: True - agent_explorer_info_mcmc_steps: 0 - agent_validation_parallel_workers: 6 - bilevel_plan_without_sim: True # for the demonstrator - code_sim_learning_rollout_min_posterior_width: 0.1 - - # C1 in policy mode: the solve deliverable is a per-task closed-loop - # policy.py that owns recovery, so the replan machinery is off. - sim_predicator_policy: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - demonstrator: "oracle_process_planning" - explorer: "agent_model_based" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - agent_sim_learn_kept_predicates_names: ["Holding"] - partially_observable: True - agent_explorer_info_seeking: True - execution_monitor: "subgoal_annotations" - agent_bilevel_max_execution_replans: 0 - agent_solve_policy_mode: True - agent_bilevel_use_llm_initial_params: True # LLM proposes params - agent_sdk_max_agent_turns_per_iteration: 200 - agent_sdk_image_max_px: 900 - agent_solve_max_attempts: 1 - agent_solve_attempt_wall_clock: 2700 - agent_solve_fresh_context: True - agent_solve_use_journal: True - agent_plan_validation_physics_margin: True - agent_plan_validation_rule_param_margin: True - agent_explorer_info_mcmc_steps: 0 - agent_validation_parallel_workers: 6 - bilevel_plan_without_sim: True # for the demonstrator - code_sim_learning_rollout_min_posterior_width: 0.1 - - # ---- ORACLES (given a GT world model, ordered from most GT handed in to least) ---- - - # GT monolithic sim + GT predicates; the agent plans against it with - # submit_plan / sim.refine and the sketch scaffolding. - agent_oracle_mono_sim: - NAME: "agent_model_based" - SKIP: True - FLAGS: - explorer: "agent_model_free" - demonstrator: "oracle_process_planning" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - option_model_use_gui: True - agent_bilevel_log_state: False - agent_bilevel_plan_sketch_file: "tests/approaches/test_data/boil_plan_sketch.txt" - # U1 (upper bound): GT hybrid sim + true physical params, as if learning had - # succeeded; same solve budget as C1 (5 time-boxed attempts + journal). - agent_oracle_hybrid_sim: - NAME: "agent_sim_learning" - SKIP: True - FLAGS: - demonstrator: "oracle_process_planning" - bilevel_plan_without_sim: True # for the demonstrator - explorer: "agent_model_based" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - option_model_use_gui: False - agent_bilevel_log_state: False - agent_sim_learn_oracle_sim_program: True - agent_sim_learn_oracle_sim_params: True - agent_bilevel_use_llm_initial_params: True - num_online_learning_cycles: 0 - execution_monitor: "subgoal_annotations" - agent_bilevel_max_execution_replans: 2 - agent_sdk_max_agent_turns_per_iteration: 200 - agent_sdk_image_max_px: 900 - agent_solve_max_attempts: 1 - agent_solve_attempt_wall_clock: 2700 - agent_solve_fresh_context: True - agent_solve_use_journal: True - agent_plan_validation_rule_param_margin: True - agent_explorer_info_mcmc_steps: 0 - agent_validation_parallel_workers: 6 - # GT hybrid sim + GT predicates; learn only the params. - agent_param_learning: - NAME: "agent_sim_learning" - SKIP: True - FLAGS: - explorer: "agent_model_based" - demonstrator: "oracle_process_planning" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - option_model_use_gui: True - agent_bilevel_log_state: False - agent_bilevel_plan_sketch_file: "tests/approaches/test_data/boil_plan_sketch.txt" - agent_sim_learn_oracle_sim_program: True - agent_sim_learn_oracle_sim_params: False - agent_sim_learn_oracle_sim_param_noise_scale: 1.0 # 0.8 gives a satisficing plan - code_sim_learning_num_mcmc_steps: 0 - # GT predicates; learn the hybrid sim and its params. - agent_sim_learning: - NAME: "agent_sim_learning" - SKIP: True - FLAGS: - explorer: "agent_model_based" - demonstrator: "oracle_process_planning" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - option_model_use_gui: True - agent_bilevel_log_state: False - agent_bilevel_plan_sketch_file: "tests/approaches/test_data/boil_plan_sketch.txt" - agent_sim_learn_oracle_sim_program: False - agent_sim_learn_oracle_sim_params: False - code_sim_learning_num_mcmc_steps: 0 - # Fully-observable OURS: full state, no GT models. - agent_fo_predicate_invention: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - explorer: "agent_model_based" - demonstrator: "oracle_process_planning" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - option_model_use_gui: False - agent_bilevel_log_state: False - online_learning_early_stopping: True - agent_sim_learn_oracle_sim_program: False - agent_sim_learn_oracle_sim_params: False - code_sim_learning_num_mcmc_steps: 0 - agent_sim_learn_kept_predicates_names: ["Holding"] - - # ---- BASELINES (lower bounds / alternative learners) ---- - - # C2: the same agent with no simulator (agent_planner_use_simulator False); - # memory carries across trials via the session, notes.md and the journal. - # Same observation as OURS (partially_observable): the tier-1 C2 runs - # of 2026-09-03 predate this line and saw the hidden features. - agent_model_free_planning: - NAME: "agent_model_free" - SKIP: True - FLAGS: - explorer: "agent_model_free" - demonstrator: "oracle_process_planning" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - partially_observable: True - agent_planner_use_simulator: False - agent_planner_use_scratchpad: True - agent_solve_use_journal: True - bilevel_plan_without_sim: True # for the demonstrator - agent_solve_max_attempts: 1 - agent_solve_attempt_wall_clock: 2700 - agent_sdk_max_agent_turns_per_iteration: 200 - agent_sdk_image_max_px: 900 - # C6: NSRT learning with oracle predicates, samplers and full observability - # (flags from configs/ExoPredicator/causal_predicator_baselines.yaml). - operator_learning: - NAME: "online_nsrt_learning" - SKIP: True - FLAGS: - demonstrator: "oracle_process_planning" - bilevel_plan_without_sim: True - explorer: "exploit_planning" - terminate_on_goal_reached_and_option_terminated: True - bilevel_planning_explorer_enumerate_plans: True - exploit_bilevel_planning_explorer_fallback_explorer: "RandomNSRTs" - online_learning_assert_no_exclude_pred: False - disable_harmlessness_check: True - sampler_learner: "oracle" - option_learner: "no_learning" - clustering_learner_check_effect_equality: False - # C3: OURS' loop, but the model is a text document (world_model.md) the - # agent writes and reasons over; no code, no simulator, same inputs as OURS. - nl_world_model: - NAME: "agent_nl_world_model" - SKIP: True - FLAGS: - demonstrator: "oracle_process_planning" - explorer: "agent_model_free" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - agent_planner_use_simulator: False - agent_planner_use_scratchpad: True - agent_sim_learn_kept_predicates_names: ["Holding"] - partially_observable: True - agent_sdk_max_agent_turns_per_iteration: 200 - agent_sdk_image_max_px: 900 - agent_solve_max_attempts: 1 - agent_solve_attempt_wall_clock: 2700 - agent_solve_fresh_context: True - agent_solve_use_journal: True - bilevel_plan_without_sim: True # for the demonstrator - # C4 (Pinductor form): OURS' loop, but the model is an option-level program - # with a hidden state, scored by sim.score; belief particles drive the gate. - code_world_model: - NAME: "agent_program_world_model" - SKIP: True - FLAGS: - <<: *ours_flags - agent_explorer_info_seeking: False - agent_plan_validation_physics_margin: False - # C5: a GNN transition model over the object features (history-conditioned), - # same interaction budget as the agent arms, planned by shooting. - gnn_dynamics: - NAME: "gnn_dynamics_shooting" - SKIP: True - FLAGS: - demonstrator: "oracle_process_planning" - bilevel_plan_without_sim: True - explorer: "random_options" - terminate_on_goal_reached_and_option_terminated: True - partially_observable: True # same observation as OURS - gnn_num_epochs: 5000 - gnn_use_validation_set: True - gnn_do_normalization: True - gnn_dynamics_history_len: 2 - # C8: MAPLE-Q with oracle operators and samplers, same interaction budget as - # the agent arms (ExoPredicator flags minus its data sizing). - maple_q: - NAME: "maple_q_with_process" - SKIP: True - FLAGS: - demonstrator: "oracle_process_planning" - bilevel_plan_without_sim: True - explorer: "maple_q" - strips_learner: "oracle" - sampler_learner: "oracle" - only_learn_exogenous_processes: True - online_learning_assert_no_exclude_pred: False - maple_q_same_hla_option_param_space: False - mlp_regressor_max_itr: 640000 - active_sampler_learning_batch_size: 512 - # Oracle residual program with LLM-guessed params, no learning (not a paper arm). - agent_base_sim_no_learning: - NAME: "agent_sim_learning" - SKIP: True - FLAGS: - demonstrator: "oracle_process_planning" - bilevel_plan_without_sim: True # for the demonstrator - explorer: "agent_model_based" - terminate_on_goal_reached_and_option_terminated: True - agent_sdk_use_local_sandbox: True - option_model_terminate_on_repeat: False - option_model_use_gui: False - agent_bilevel_log_state: False - agent_sim_learn_oracle_sim_program: True - agent_sim_learn_oracle_sim_params: False - agent_bilevel_use_llm_initial_params: True - num_online_learning_cycles: 0 - execution_monitor: "subgoal_annotations" - agent_bilevel_max_execution_replans: 2 - - # ---- ABLATIONS (OURS minus one component, paper A1-A8; merge *ours_flags) ---- - - # A1: no learning; plans are made, checked and run on the base engine. - # The sweeps measure it as C1's pre-loop test; parked here for standalone reruns. - sim_predicator_no_learning: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - <<: *ours_flags - num_online_learning_cycles: 0 - # A2: one data-free synthesis session, then solve (no online learning). - sim_predicator_zero_shot: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - <<: *ours_flags - agent_sim_learn_zero_shot: True - num_online_learning_cycles: 0 - # A3: scripted random-options exploration instead of the agent explorer. - sim_predicator_undirected_explore: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - <<: *ours_flags - explorer: "random_options" - agent_explorer_info_seeking: False - # A4: no parameter fitting; declared inits are the estimate, [lo, hi] the - # interval the margins and the ensemble sample from. - sim_predicator_no_param_fit: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - <<: *ours_flags - agent_sim_learn_declared_params_only: True - # A5: execute the first goal-reaching capture (1 rollout, no margins). - sim_predicator_no_validation: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - <<: *ours_flags - agent_plan_validation_rollouts: 1 - agent_plan_validation_rollouts_after_flaky: 1 - agent_plan_validation_physics_margin: False - agent_plan_validation_rule_param_margin: False - # A6: experiments under the point estimate (no ensemble-disagreement - # probe scoring); the validation gate keeps its margins. - sim_predicator_explore_no_disagreement: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - <<: *ours_flags - agent_explorer_info_seeking: False - # A7: validation under the point estimate (no perturbed rollouts, no - # ensemble members); experiments keep the ensemble. - sim_predicator_validation_no_uncertainty: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - <<: *ours_flags - agent_plan_validation_physics_margin: False - agent_plan_validation_rule_param_margin: False - # A6+A7 combined: point estimates only; every consumer of a posterior is off. - sim_predicator_no_uncertainty: - NAME: "agent_sim_predicate_invention" - SKIP: True - FLAGS: - <<: *ours_flags - agent_sim_learn_param_uncertainty: False - agent_plan_validation_physics_margin: False - agent_plan_validation_rule_param_margin: False - agent_explorer_info_seeking: False - # A8: no invented predicates (the non-inventing class, Holding + goal_nl). - sim_predicator_no_predicates: - NAME: "agent_sim_learning" - SKIP: True - FLAGS: - <<: *ours_flags - - # ---- DEMONSTRATOR / SCRIPTED (the demo source every agent arm consumes) ---- - - # Oracle process planning + bilevel refinement with GT models; oracle.yaml - # runs it standalone. - oracle: - NAME: "oracle_process_planning" - SKIP: True - FLAGS: - demonstrator: "oracle_process_planning" - terminate_on_goal_reached_and_option_terminated: True - sesame_check_expected_atoms: False - bilevel_plan_without_sim: True - # Scripted human-in-the-loop option control (manual baseline). - human_interaction: - NAME: "human_interaction" - SKIP: True - FLAGS: - human_option_control_approach_use_scripted_option: True - human_option_control_approach_use_all_options: True - scripted_option_dir: "scripted_option_policies" - skill_phase_use_motion_planning: True - terminate_on_goal_reached_and_option_terminated: True diff --git a/scripts/configs/predicatorv3/approaches/continual.yaml b/scripts/configs/predicatorv3/approaches/continual.yaml deleted file mode 100644 index b7664ecaac..0000000000 --- a/scripts/configs/predicatorv3/approaches/continual.yaml +++ /dev/null @@ -1,203 +0,0 @@ -# Continual-protocol arm menu. Every entry is parked (SKIP); a launcher -# un-parks arms, usually under a round-specific key with EXTENDS (see -# continual_common.yaml). MB arms are gated on a deployable model on the -# test level (continual_require_model_on_test). MF arms switch off the -# model-side machinery so the two differ only in the model. The skill -# preflight (continual_skill_preflight) is off by default since Sept 18, -# 2026; a launcher that wants the Sept 17 Sonnet setting sets it to true. -APPROACHES: - from_assets_opus: - NAME: agent_continual_from_assets - SKIP: true - FLAGS: - agent_sdk_model_name: claude-opus-5 - continual_require_model_on_test: true - agent_sim_learn_declared_params_only: false - continual_uncertainty_decisions: true - mb_opus: - NAME: agent_continual - SKIP: true - FLAGS: - agent_sdk_model_name: claude-opus-5 - continual_require_model_on_test: true - mb_sonnet: - NAME: agent_continual - SKIP: true - FLAGS: - agent_sdk_model_name: claude-sonnet-5 - continual_require_model_on_test: true - # EMPIRIC with what the agentic real-to-sim arm receives (Sept 18, - # 2026): the engine wrapper, the scene manifest and the URDF and mesh - # files, plus the twin's own core module where the domain declares one - # (Fan and Balloons). The domain twin still backs the model. - mb_scene_package_opus: - NAME: agent_continual - SKIP: true - FLAGS: - agent_sdk_model_name: claude-opus-5 - continual_require_model_on_test: true - agent_sim_provide_base_sim_source: true - continual_provide_scene_package: true - mf_opus: - NAME: agent_continual_model_free - SKIP: true - FLAGS: &mf_flags - agent_sdk_model_name: claude-opus-5 - code_sim_learning_interval_belief: false - agent_explorer_info_seeking_noise_aware: false - code_sim_learning_rollout_noise_filter: false - code_sim_learning_carry_posterior: false - code_sim_learning_fit_evidence: false - continual_belief_frame: false - agent_model_repair: false - agent_planner_use_simulator: false - continual_uncertainty_decisions: false - mf_sonnet: - NAME: agent_continual_model_free - SKIP: true - FLAGS: - <<: *mf_flags - agent_sdk_model_name: claude-sonnet-5 - # The direct agent with the scene files the agentic real-to-sim arm - # builds from (Sept 19, 2026): the engine wrapper, the scene manifest - # and the URDF and mesh files as read-only references. Its prompt only - # asks it to solve the levels; nothing asks for a simulator or model. - mf_scene_package_opus: - NAME: agent_continual_model_free - SKIP: true - FLAGS: - <<: *mf_flags - continual_provide_scene_package: true - # ---- The other six arms of the eight-agent comparison - # (continual_eight_agent_noisy_sweep.yaml). Their approach classes - # came over from the continual-comparisons line in the Sept 17 merge - # (agent_continual_program_approach.py, agent_continual_frozen_approach.py, - # agent_continual_ablation_approach.py). Ids follow the mb_/mf_ - # pattern: _. - # Standalone: the agent writes and owns a program world model, no harness - # fitting. - standalone_opus: - NAME: agent_continual_program_world_model - SKIP: true - FLAGS: - agent_sdk_model_name: claude-opus-5 - standalone_sonnet: - NAME: agent_continual_program_world_model - SKIP: true - FLAGS: - agent_sdk_model_name: claude-sonnet-5 - # Oracle dynamics: the ground-truth simulator stands in for the learned - # model. Use the repaired r2 runtime with no automatic execution gate - # or shadow audit. See docs/comparisons/oracle-repair-notes.md. - oracle_dynamics_opus: - NAME: agent_continual_oracle_dynamics - SKIP: true - FLAGS: - agent_sdk_model_name: claude-opus-5 - continual_skill_preflight: false - continual_validation_audit: false - oracle_dynamics_sonnet: - NAME: agent_continual_oracle_dynamics - SKIP: true - FLAGS: - agent_sdk_model_name: claude-sonnet-5 - continual_skill_preflight: false - continual_validation_audit: false - # Scene-only: the exact scene twin with corrected base calibration and - # no mechanism code, frozen for the run (formerly "oracle scene"). - scene_only_opus: - NAME: agent_continual_scene_only - SKIP: true - FLAGS: - agent_sdk_model_name: claude-opus-5 - scene_only_sonnet: - NAME: agent_continual_scene_only - SKIP: true - FLAGS: - agent_sdk_model_name: claude-sonnet-5 - # Zero shot: no training level, the test level is played cold. - zero_shot_opus: - NAME: agent_continual_zero_shot - SKIP: true - FLAGS: - agent_sdk_model_name: claude-opus-5 - zero_shot_sonnet: - NAME: agent_continual_zero_shot - SKIP: true - FLAGS: - agent_sdk_model_name: claude-sonnet-5 - # No fitting: the MB agent with only its declared parameters, no - # harness parameter fitting (the agent may fit in its own sandbox - # code). Gated like the MB arms. - no_fitting_opus: - NAME: agent_continual_no_fitting - SKIP: true - FLAGS: &no_fitting_flags - agent_sdk_model_name: claude-opus-5 - continual_require_model_on_test: true - agent_sim_learn_declared_params_only: true - no_fitting_sonnet: - NAME: agent_continual_no_fitting - SKIP: true - FLAGS: - <<: *no_fitting_flags - agent_sdk_model_name: claude-sonnet-5 - # No uncertainty handling: fit raw noisy observations, with no state - # smoothing, uncertainty margins or information seeking. Gated like MB. - no_uncertainty_opus: - NAME: agent_continual_no_uncertainty - SKIP: true - FLAGS: &no_uncertainty_flags - agent_sdk_model_name: claude-opus-5 - continual_require_model_on_test: true - code_sim_learning_rollout_noise_filter: false - continual_belief_frame: false - continual_uncertainty_decisions: false - agent_sim_learn_param_uncertainty: false - agent_plan_validation_rule_param_margin: false - agent_plan_validation_physics_margin: false - agent_explorer_info_seeking: false - agent_explorer_info_seeking_adaptive: false - agent_explorer_info_seeking_noise_aware: false - code_sim_learning_interval_belief: false - code_sim_learning_carry_posterior: false - belief_joint_draws: 0 - # Nothing about the observation noise is declared to this arm: no - # noise section, no [noise] line, and the harness fit models none - # of it (Sept 18, 2026). - continual_obs_noise_declared: false - no_uncertainty_sonnet: - NAME: agent_continual_no_uncertainty - SKIP: true - FLAGS: - <<: *no_uncertainty_flags - agent_sdk_model_name: claude-sonnet-5 - # Agentic real-to-sim baseline (Sept 18, 2026): no domain twin. The - # agent gets the generic PyBulletEnv, a domain-agnostic SceneBase, the - # scene manifest and the asset files, and builds its own simulator; - # the harness fits nothing and runs no uncertainty machinery. Gated - # like the MB arms. Runs with the same skill library as the other - # arms (composite by default); set skill_library: primitive for the - # robot-stack variant. - real_to_sim_opus: - NAME: agent_continual_real_to_sim - SKIP: true - FLAGS: &real_to_sim_flags - agent_sdk_model_name: claude-opus-5 - continual_require_model_on_test: true - agent_sim_learn_declared_params_only: true - continual_uncertainty_decisions: false - agent_sim_learn_param_uncertainty: false - agent_plan_validation_rule_param_margin: false - agent_plan_validation_physics_margin: false - agent_explorer_info_seeking: false - agent_explorer_info_seeking_adaptive: false - agent_explorer_info_seeking_noise_aware: false - code_sim_learning_interval_belief: false - code_sim_learning_carry_posterior: false - real_to_sim_sonnet: - NAME: agent_continual_real_to_sim - SKIP: true - FLAGS: - <<: *real_to_sim_flags - agent_sdk_model_name: claude-sonnet-5 diff --git a/scripts/configs/predicatorv3/common.yaml b/scripts/configs/predicatorv3/common.yaml deleted file mode 100644 index 7e130607fd..0000000000 --- a/scripts/configs/predicatorv3/common.yaml +++ /dev/null @@ -1,35 +0,0 @@ -ARGS: - - "debug" - # - "use_gui" - - "make_failure_videos" - - "make_test_videos" - - "make_interaction_videos" - # - "make_demo_videos" - # - "make_demo_images" # support images - # - "make_failure_images" # query images - # - "make_test_images" # query images - # - "save_atoms" -FLAGS: - num_online_learning_cycles: 5 - online_learning_early_stopping: True - online_learning_early_stopping_require_all_attempts: True - online_learning_early_stopping_skip_redundant_test: True - online_nsrt_learning_requests_per_cycle: 2 - skill_phase_use_motion_planning: True - max_num_steps_interaction_request: 500 - pretrained_model_service_provider: "openrouter" - llm_model_name: "google/gemini-2.5-pro" - llm_openai_max_response_tokens: 1e6 - terminate_on_goal_reached: False - pybullet_ik_validate: False - num_train_tasks: 1 - num_test_tasks: 1 - video_fps: 20 - pybullet_camera_height: 900 - pybullet_camera_width: 900 - planning_filter_unreachable_nsrt: False - timeout: 600 - log: 'logs/' - no_repeated_arguments_in_grounding: True -START_SEED: 0 -NUM_SEEDS: 1 \ No newline at end of file diff --git a/scripts/configs/predicatorv3/continual_benchmark_five_seeds.yaml b/scripts/configs/predicatorv3/continual_benchmark_five_seeds.yaml deleted file mode 100644 index 10ce8bc357..0000000000 --- a/scripts/configs/predicatorv3/continual_benchmark_five_seeds.yaml +++ /dev/null @@ -1,38 +0,0 @@ -# Extend the paper's six comparison arms with seeds 3 and 4. -# Freeze from f2ed37aef: repaired runtime, original benchmark geometry. -# Do not use the later Balloons layout change for this cohort. -# Launch with --partition mit_preemptable --requeue --accounts a,b,c,d. -# Account dat (the user's "e") is backup only, outside normal rotation. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 3 -NUM_SEEDS: 2 -FLAGS: - continual_skill_preflight: false - continual_validation_audit: false -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - oracle_dynamics_opus_benchmark_r2: - EXTENDS: oracle_dynamics_opus - mf_opus_benchmark_r2: - EXTENDS: mf_opus - mf_scene_package_opus_benchmark_r1: - EXTENDS: mf_scene_package_opus - standalone_opus_benchmark_r2: - EXTENDS: standalone_opus - no_fitting_opus_benchmark_r1: - EXTENDS: no_fitting_opus - no_uncertainty_opus_benchmark_r2: - EXTENDS: no_uncertainty_opus diff --git a/scripts/configs/predicatorv3/continual_direct_scene_files_benchmark_r1.yaml b/scripts/configs/predicatorv3/continual_direct_scene_files_benchmark_r1.yaml deleted file mode 100644 index df14befdd2..0000000000 --- a/scripts/configs/predicatorv3/continual_direct_scene_files_benchmark_r1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -# The direct agent with the scene files (Sept 19, 2026): the agentic -# real-to-sim arm's engine wrapper, scene manifest and URDF and mesh -# files as read-only references, and a prompt that only asks it to solve -# the levels (mf_scene_package_opus). Opus, three seeds each: 5 domains x -# 3 seeds = 15 runs, so the logs land under -# logs/agent_continual_model_free/-mf_scene_package_opus_benchmark_r1/seed. -# -# python scripts/engaging/launch.py -c predicatorv3/continual_direct_scene_files_benchmark_r1.yaml --partition mit_preemptable,mit_normal --requeue --accounts b,d -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 3 -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - mf_scene_package_opus_benchmark_r1: - EXTENDS: mf_scene_package_opus diff --git a/scripts/configs/predicatorv3/continual_eight_agent_noisy_sweep.yaml b/scripts/configs/predicatorv3/continual_eight_agent_noisy_sweep.yaml deleted file mode 100644 index 0aad4ddd63..0000000000 --- a/scripts/configs/predicatorv3/continual_eight_agent_noisy_sweep.yaml +++ /dev/null @@ -1,55 +0,0 @@ -# The eight-agent benchmark sweep (Sept 18, 2026): the two main agents, -# model-based (EMPIRIC, agent_continual) and model-free (the direct -# agent, agent_continual_model_free), plus the six ablation arms -# (standalone program world model, oracle dynamics, scene-only, zero -# shot, no harness fitting, no explicit uncertainty), all on Opus, on -# the five benchmark test settings, three seeds each: 8 approaches x 5 -# domains x 3 seeds = 120 runs. Envs and arms come from the menus -# (envs/continual.yaml, approaches/continual.yaml): Balloons composition -# test levels with the 25-step dwell, Bridge four-span test row, Boil -# two-jug test level, Fan maze test level, Domino high-friction turn. -# The skill preflight is off (the runtime default). The round keys keep -# these logs apart from the Sept 17 rounds -# (logs//-/seed). -# -# The Sept 12-14 version of this sweep ran the same eight arms on the -# Sept 14 settings (three-span Bridge, uniform Fan, one-jug Boil, the -# original Balloons distribution) under runtime 091d8c5db. -# -# Forty Slurm arrays of three seeds; the Opus arms saturate one Claude -# account's session limit in about an hour, so spread the accounts: -# python scripts/engaging/launch.py -c predicatorv3/continual_eight_agent_noisy_sweep.yaml --partition mit_preemptable,mit_normal --requeue --accounts b,c -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 3 -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - mb_opus_benchmark_r1: - EXTENDS: mb_opus - mf_opus_benchmark_r1: - EXTENDS: mf_opus - standalone_opus_benchmark_r1: - EXTENDS: standalone_opus - oracle_dynamics_opus_benchmark_r1: - EXTENDS: oracle_dynamics_opus - scene_only_opus_benchmark_r1: - EXTENDS: scene_only_opus - zero_shot_opus_benchmark_r1: - EXTENDS: zero_shot_opus - no_fitting_opus_benchmark_r1: - EXTENDS: no_fitting_opus - no_uncertainty_opus_benchmark_r1: - EXTENDS: no_uncertainty_opus diff --git a/scripts/configs/predicatorv3/continual_empiric_benchmark_r2.yaml b/scripts/configs/predicatorv3/continual_empiric_benchmark_r2.yaml deleted file mode 100644 index e9985a865a..0000000000 --- a/scripts/configs/predicatorv3/continual_empiric_benchmark_r2.yaml +++ /dev/null @@ -1,26 +0,0 @@ -# Two prospective seeds per domain, with bounded nonblocking validation. -# Freeze this runtime before submitting; never resume historical EMPIRIC. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 3 -NUM_SEEDS: 2 -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - mb_opus_benchmark_r2: - EXTENDS: mb_opus - FLAGS: - continual_skill_preflight: false - continual_validation_audit: true - continual_validation_audit_seconds: 600.0 diff --git a/scripts/configs/predicatorv3/continual_empiric_scene_package_benchmark_r1.yaml b/scripts/configs/predicatorv3/continual_empiric_scene_package_benchmark_r1.yaml deleted file mode 100644 index 04111f0f08..0000000000 --- a/scripts/configs/predicatorv3/continual_empiric_scene_package_benchmark_r1.yaml +++ /dev/null @@ -1,29 +0,0 @@ -# EMPIRIC rerun on the benchmark runtime (Sept 18, 2026), with what the -# agentic real-to-sim arm receives: the engine wrapper, the scene -# manifest and the asset files, plus the twin's core module on Fan and -# Balloons (mb_scene_package_opus). Opus, three seeds each: 5 domains x -# 3 seeds = 15 runs, on the same menus as the other benchmark arms, so -# the logs land under -# logs/agent_continual/-mb_scene_package_opus_benchmark_r1/seed. -# -# python scripts/engaging/launch.py -c predicatorv3/continual_empiric_scene_package_benchmark_r1.yaml --partition mit_preemptable,mit_normal --requeue --accounts b,d -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 3 -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - mb_scene_package_opus_benchmark_r1: - EXTENDS: mb_scene_package_opus diff --git a/scripts/configs/predicatorv3/continual_fan_inertial_baselines_r1.yaml b/scripts/configs/predicatorv3/continual_fan_inertial_baselines_r1.yaml deleted file mode 100644 index e51f4ff1ef..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_inertial_baselines_r1.yaml +++ /dev/null @@ -1,20 +0,0 @@ -# Five additional arms on the original frozen, no-ramp inertial domain. -includes: - - /home/ycliang/predicators/logs/fan-inertial-baselines-runtime-20260921/scripts/configs/predicatorv3/continual_fan_inertial_pilot_r1.yaml -START_SEED: 0 -NUM_SEEDS: 5 -APPROACHES: - mb_opus_inertial_pilot_r1: - SKIP: true - mf_opus_inertial_pilot_r1: - SKIP: true - oracle_dynamics_opus_inertial_r1: - EXTENDS: oracle_dynamics_opus - mf_scene_package_opus_inertial_r1: - EXTENDS: mf_scene_package_opus - standalone_opus_inertial_r1: - EXTENDS: standalone_opus - no_fitting_opus_inertial_r1: - EXTENDS: no_fitting_opus - no_uncertainty_opus_inertial_r1: - EXTENDS: no_uncertainty_opus diff --git a/scripts/configs/predicatorv3/continual_fan_inertial_confirmation_r1.yaml b/scripts/configs/predicatorv3/continual_fan_inertial_confirmation_r1.yaml deleted file mode 100644 index 108eb42fcd..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_inertial_confirmation_r1.yaml +++ /dev/null @@ -1,15 +0,0 @@ -# Fresh confirmation seeds; launch only after the pilot screening rule passes. -# Execute from the frozen inertial runtime, not a revised environment. -includes: - - continual_fan_inertial_pilot_r1.yaml -START_SEED: 2 -NUM_SEEDS: 3 -APPROACHES: - mb_opus_inertial_pilot_r1: - SKIP: true - mf_opus_inertial_pilot_r1: - SKIP: true - mb_opus_inertial_confirmation_r1: - EXTENDS: mb_opus - mf_opus_inertial_confirmation_r1: - EXTENDS: mf_opus diff --git a/scripts/configs/predicatorv3/continual_fan_inertial_pilot_r1.yaml b/scripts/configs/predicatorv3/continual_fan_inertial_pilot_r1.yaml deleted file mode 100644 index 1287c7c63d..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_inertial_pilot_r1.yaml +++ /dev/null @@ -1,27 +0,0 @@ -# Separate candidate, not a replacement for either existing Fan cohort. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 2 -FLAGS: - continual_skill_preflight: false - continual_validation_audit: false -ENVS: - fan_inertial: - EXTENDS: fan_maze - FLAGS: - fan_exposed_transfer: true - fan_inertial_transfer: true - fan_train_num_walls_per_task: "[0]" - fan_test_num_walls_per_task: "[0]" - fan_test_num_pos_x: 3 - fan_test_num_pos_y: 3 - fan_train_task_generation: uniform - fan_test_task_generation: uniform -APPROACHES: - mb_opus_inertial_pilot_r1: - EXTENDS: mb_opus - mf_opus_inertial_pilot_r1: - EXTENDS: mf_opus diff --git a/scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2.yaml b/scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2.yaml deleted file mode 100644 index 34c7f8a337..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# Fresh matched seeds with aligned state/timing/discrepancy guidance. -# Fan is the reviewed 3 mm ramp with the extended landing. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 5 -ENVS: - fan: - SKIP: false -APPROACHES: - oracle_dynamics_opus_fan_prompt_r2: - EXTENDS: oracle_dynamics_opus - FLAGS: - continual_skill_preflight: false - continual_validation_audit: false diff --git a/scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2_retry_seed0.yaml b/scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2_retry_seed0.yaml deleted file mode 100644 index 3d4b8ce09e..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2_retry_seed0.yaml +++ /dev/null @@ -1,5 +0,0 @@ -# Retry after account a rejected Claude subscription access before doing work. -includes: - - continual_fan_oracle_prompt_r2.yaml -START_SEED: 0 -NUM_SEEDS: 1 diff --git a/scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2_retry_seed3.yaml b/scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2_retry_seed3.yaml deleted file mode 100644 index e7e3e11e8c..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_oracle_prompt_r2_retry_seed3.yaml +++ /dev/null @@ -1,5 +0,0 @@ -# Retry after account a rejected Claude subscription access before doing work. -includes: - - continual_fan_oracle_prompt_r2.yaml -START_SEED: 3 -NUM_SEEDS: 1 diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_confirmation_r1.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_confirmation_r1.yaml deleted file mode 100644 index 4630f1a11e..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_confirmation_r1.yaml +++ /dev/null @@ -1,15 +0,0 @@ -# Fresh seeds after the ramp pilot screening gate passes. -# Execute from the unchanged frozen ramp runtime. -includes: - - continual_fan_ramp_pilot_r1.yaml -START_SEED: 2 -NUM_SEEDS: 3 -APPROACHES: - mb_opus_ramp_pilot_r1: - SKIP: true - mf_opus_ramp_pilot_r1: - SKIP: true - mb_opus_ramp_confirmation_r1: - EXTENDS: mb_opus - mf_opus_ramp_confirmation_r1: - EXTENDS: mf_opus diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_long_landing_r1.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_long_landing_r1.yaml deleted file mode 100644 index 7e5df47da0..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_long_landing_r1.yaml +++ /dev/null @@ -1,16 +0,0 @@ -# Five development seeds on the visually reviewed 10 cm landing extension. -includes: - - continual_fan_ramp_pilot_r1.yaml -START_SEED: 0 -NUM_SEEDS: 5 -ENVS: - fan_ramp: - FLAGS: - fan_ramp_landing_extension: 0.10 -APPROACHES: - mb_opus_ramp_pilot_r1: - SKIP: true - mf_opus_ramp_pilot_r1: - SKIP: true - mb_opus_ramp_long_landing_r1: - EXTENDS: mb_opus diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_low_drop_confirmation_r1.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_low_drop_confirmation_r1.yaml deleted file mode 100644 index e3f5984148..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_low_drop_confirmation_r1.yaml +++ /dev/null @@ -1,5 +0,0 @@ -# Additional EMPIRIC seeds on the unchanged lower-drop runtime. -includes: - - /home/ycliang/predicators/logs/fan-ramp-low-drop-runtime-20260921/scripts/configs/predicatorv3/continual_fan_ramp_low_drop_r1.yaml -START_SEED: 2 -NUM_SEEDS: 3 diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_low_drop_r1.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_low_drop_r1.yaml deleted file mode 100644 index 83bf1ea7bb..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_low_drop_r1.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# Visually reviewed 3 mm ramp, 10 cm longer landing, EMPIRIC screening first. -includes: - - continual_fan_ramp_pilot_r1.yaml -START_SEED: 0 -NUM_SEEDS: 2 -ENVS: - fan_ramp: - FLAGS: - fan_ramp_landing_extension: 0.10 - fan_ramp_rise: 0.003 -APPROACHES: - mb_opus_ramp_pilot_r1: - SKIP: true - mf_opus_ramp_pilot_r1: - SKIP: true - mb_opus_ramp_low_drop_r1: - EXTENDS: mb_opus diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r1.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r1.yaml deleted file mode 100644 index 6275901a45..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r1.yaml +++ /dev/null @@ -1,10 +0,0 @@ -# Two additional seeds, preserving the existing frozen Oracle ramp setup. -includes: - - /home/ycliang/predicators/logs/fan-ramp-skill-repair-runtime-20260921/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml -START_SEED: 5 -NUM_SEEDS: 2 -APPROACHES: - mb_opus_ramp_skill_repair_r1: - SKIP: true - oracle_dynamics_opus_ramp_skill_repair_r1: - EXTENDS: oracle_dynamics_opus diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r2.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r2.yaml deleted file mode 100644 index f9b0b2e97a..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r2.yaml +++ /dev/null @@ -1,10 +0,0 @@ -# Two further seeds, preserving the frozen Oracle Dynamics Fan + Ramp setup. -includes: - - /home/ycliang/predicators/logs/fan-ramp-skill-repair-runtime-20260921/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml -START_SEED: 7 -NUM_SEEDS: 2 -APPROACHES: - mb_opus_ramp_skill_repair_r1: - SKIP: true - oracle_dynamics_opus_ramp_skill_repair_r1: - EXTENDS: oracle_dynamics_opus diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r3.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r3.yaml deleted file mode 100644 index ed4f65f528..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_oracle_extra_r3.yaml +++ /dev/null @@ -1,10 +0,0 @@ -# Two further seeds of the unchanged frozen Oracle Fan + Ramp cohort. -includes: - - /home/ycliang/predicators/logs/fan-ramp-skill-repair-runtime-20260921/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml -START_SEED: 9 -NUM_SEEDS: 2 -APPROACHES: - mb_opus_ramp_skill_repair_r1: - SKIP: true - oracle_dynamics_opus_ramp_skill_repair_r1: - EXTENDS: oracle_dynamics_opus diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_pilot_r1.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_pilot_r1.yaml deleted file mode 100644 index 92a8b26f04..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_pilot_r1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -# User approved two seeds per agent. Separate from previous Fan cohorts. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 2 -FLAGS: - continual_skill_preflight: false - continual_validation_audit: false -ENVS: - fan_ramp: - EXTENDS: fan_maze - FLAGS: - fan_exposed_transfer: true - fan_inertial_transfer: true - fan_ramp_transfer: true - fan_train_num_walls_per_task: "[0]" - fan_test_num_walls_per_task: "[0]" - fan_test_num_pos_x: 3 - fan_test_num_pos_y: 3 - fan_train_task_generation: uniform - fan_test_task_generation: uniform -APPROACHES: - mb_opus_ramp_pilot_r1: - EXTENDS: mb_opus - mf_opus_ramp_pilot_r1: - EXTENDS: mf_opus diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_baselines_r1.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_baselines_r1.yaml deleted file mode 100644 index f51a824177..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_baselines_r1.yaml +++ /dev/null @@ -1,20 +0,0 @@ -# Six comparison arms on exactly the repaired EMPIRIC ramp runtime. -includes: - - /home/ycliang/predicators/logs/fan-ramp-skill-repair-runtime-20260921/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml -START_SEED: 0 -NUM_SEEDS: 5 -APPROACHES: - mb_opus_ramp_skill_repair_r1: - SKIP: true - oracle_dynamics_opus_ramp_skill_repair_r1: - EXTENDS: oracle_dynamics_opus - mf_opus_ramp_skill_repair_r1: - EXTENDS: mf_opus - mf_scene_package_opus_ramp_skill_repair_r1: - EXTENDS: mf_scene_package_opus - standalone_opus_ramp_skill_repair_r1: - EXTENDS: standalone_opus - no_fitting_opus_ramp_skill_repair_r1: - EXTENDS: no_fitting_opus - no_uncertainty_opus_ramp_skill_repair_r1: - EXTENDS: no_uncertainty_opus diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml deleted file mode 100644 index 020b931104..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml +++ /dev/null @@ -1,11 +0,0 @@ -# Five matched development seeds on the reviewed 3 mm ramp / longer landing. -# Freeze the original domain and EMPIRIC configuration; only skills change. -includes: - - /home/ycliang/predicators/logs/fan-ramp-skill-repair-runtime-20260921/scripts/configs/predicatorv3/continual_fan_ramp_low_drop_r1.yaml -START_SEED: 0 -NUM_SEEDS: 5 -APPROACHES: - mb_opus_ramp_low_drop_r1: - SKIP: true - mb_opus_ramp_skill_repair_r1: - EXTENDS: mb_opus diff --git a/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_seed0_resume.yaml b/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_seed0_resume.yaml deleted file mode 100644 index c015e1a5de..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_seed0_resume.yaml +++ /dev/null @@ -1,5 +0,0 @@ -# Resume only the interrupted seed, preserving its cohort and runtime. -includes: - - /home/ycliang/predicators/logs/fan-ramp-skill-repair-runtime-20260921/scripts/configs/predicatorv3/continual_fan_ramp_skill_repair_r1.yaml -START_SEED: 0 -NUM_SEEDS: 1 diff --git a/scripts/configs/predicatorv3/continual_fan_transfer_pilot_r1.yaml b/scripts/configs/predicatorv3/continual_fan_transfer_pilot_r1.yaml deleted file mode 100644 index 058e2b3f59..0000000000 --- a/scripts/configs/predicatorv3/continual_fan_transfer_pilot_r1.yaml +++ /dev/null @@ -1,27 +0,0 @@ -# Separate pilot, not a replacement for the paper's Fan maze results. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 2 -FLAGS: - continual_skill_preflight: false - continual_validation_audit: false -ENVS: - fan_transfer: - EXTENDS: fan_maze - FLAGS: - fan_exposed_transfer: true - # Quoted CLI literals replace the inherited lists; YAML lists append. - fan_train_num_walls_per_task: "[0]" - fan_test_num_walls_per_task: "[0]" - fan_test_num_pos_x: 3 - fan_test_num_pos_y: 3 - fan_train_task_generation: uniform - fan_test_task_generation: uniform -APPROACHES: - mb_opus_transfer_pilot_r1: - EXTENDS: mb_opus - mf_opus_transfer_pilot_r1: - EXTENDS: mf_opus diff --git a/scripts/configs/predicatorv3/continual_five_ablations_benchmark_r1.yaml b/scripts/configs/predicatorv3/continual_five_ablations_benchmark_r1.yaml deleted file mode 100644 index 566a16693c..0000000000 --- a/scripts/configs/predicatorv3/continual_five_ablations_benchmark_r1.yaml +++ /dev/null @@ -1,42 +0,0 @@ -# The five remaining comparison arms of the eight-agent benchmark sweep -# (Sept 18, 2026), launched after the scene-only round: standalone -# program world model, oracle dynamics, zero shot, no harness fitting, -# no explicit uncertainty, all on Opus, on the five benchmark settings, -# three seeds each: 5 arms x 5 domains x 3 seeds = 75 runs. Envs and -# arms come from the menus (envs/continual.yaml, approaches/ -# continual.yaml); the round keys match the eight-agent sweep -# (continual_eight_agent_noisy_sweep.yaml), so the logs land under -# logs//-_opus_benchmark_r1/seed either way. -# The Opus MB, MF and scene-only rounds ran earlier under the same keys. -# -# Launched from a frozen worktree so later edits to the main tree do not -# reach running or requeued jobs: -# python scripts/engaging/launch.py -c predicatorv3/continual_five_ablations_benchmark_r1.yaml --partition mit_preemptable,mit_normal --requeue --accounts a,b,c,d -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 3 -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - standalone_opus_benchmark_r1: - EXTENDS: standalone_opus - oracle_dynamics_opus_benchmark_r1: - EXTENDS: oracle_dynamics_opus - zero_shot_opus_benchmark_r1: - EXTENDS: zero_shot_opus - no_fitting_opus_benchmark_r1: - EXTENDS: no_fitting_opus - no_uncertainty_opus_benchmark_r1: - EXTENDS: no_uncertainty_opus diff --git a/scripts/configs/predicatorv3/continual_from_assets_expansion_r1.yaml b/scripts/configs/predicatorv3/continual_from_assets_expansion_r1.yaml deleted file mode 100644 index 666575ff7f..0000000000 --- a/scripts/configs/predicatorv3/continual_from_assets_expansion_r1.yaml +++ /dev/null @@ -1,29 +0,0 @@ -# Only the eight new runs: Bridge is already running, Fan maze is cancelled. -# Pin menus to the tested runtime rather than the mutable working checkout. -includes: - - /home/ycliang/predicators/logs/empiric-from-assets-runtime-20260921/scripts/configs/predicatorv3/continual_from_assets_pilot_r1.yaml -ENVS: - fan: - SKIP: true - bridge: - SKIP: true - fan_ramp: - EXTENDS: fan - FLAGS: - fan_exposed_transfer: true - fan_inertial_transfer: true - fan_ramp_transfer: true - fan_ramp_landing_extension: 0.10 - fan_ramp_rise: 0.003 - fan_train_num_walls_per_task: "[0]" - fan_test_num_walls_per_task: "[0]" - fan_test_num_pos_x: 3 - fan_test_num_pos_y: 3 - fan_train_task_generation: uniform - fan_test_task_generation: uniform - domino_high_friction_turn: - SKIP: false - balloons: - SKIP: false - boil: - SKIP: false diff --git a/scripts/configs/predicatorv3/continual_from_assets_pilot_r1.yaml b/scripts/configs/predicatorv3/continual_from_assets_pilot_r1.yaml deleted file mode 100644 index 56e420ec6b..0000000000 --- a/scripts/configs/predicatorv3/continual_from_assets_pilot_r1.yaml +++ /dev/null @@ -1,37 +0,0 @@ -# Two seeds each on Fan + ramp, Bridge, Domino, Balloons and Boil. -# No environment changes, no mandatory preflight, separate result cohort. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 2 -FLAGS: - continual_skill_preflight: false - continual_validation_audit: false -ENVS: - fan_ramp: - EXTENDS: fan - FLAGS: - fan_exposed_transfer: true - fan_inertial_transfer: true - fan_ramp_transfer: true - fan_ramp_landing_extension: 0.10 - fan_ramp_rise: 0.003 - fan_train_num_walls_per_task: "[0]" - fan_test_num_walls_per_task: "[0]" - fan_test_num_pos_x: 3 - fan_test_num_pos_y: 3 - fan_train_task_generation: uniform - fan_test_task_generation: uniform - bridge: - SKIP: false - domino_high_friction_turn: - SKIP: false - balloons: - SKIP: false - boil: - SKIP: false -APPROACHES: - from_assets_opus_pilot_r1: - EXTENDS: from_assets_opus diff --git a/scripts/configs/predicatorv3/continual_no_uncertainty_raw_obs_seed0_r1.yaml b/scripts/configs/predicatorv3/continual_no_uncertainty_raw_obs_seed0_r1.yaml deleted file mode 100644 index df101ac6e6..0000000000 --- a/scripts/configs/predicatorv3/continual_no_uncertainty_raw_obs_seed0_r1.yaml +++ /dev/null @@ -1,23 +0,0 @@ -# Updated no-explicit-uncertainty ablation: one seed across the five benchmark -# domains. This round fits and plans from raw noisy observations, with state -# smoothing and uncertainty-aware decisions disabled. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 1 -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - no_uncertainty_raw_obs_opus_r1: - EXTENDS: no_uncertainty_opus diff --git a/scripts/configs/predicatorv3/continual_oracle_validation_r2.yaml b/scripts/configs/predicatorv3/continual_oracle_validation_r2.yaml deleted file mode 100644 index 0a9ec8b201..0000000000 --- a/scripts/configs/predicatorv3/continual_oracle_validation_r2.yaml +++ /dev/null @@ -1,19 +0,0 @@ -# Repair pilot only. Retain r1 recordings; never resume them with new code. -# Launch from a frozen worktree after compute-node validation: -# python scripts/engaging/launch.py -c predicatorv3/continual_oracle_validation_r2.yaml --partition mit_preemptable --requeue --accounts a,b,c,d -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 3 -ENVS: - domino_high_friction_turn: - SKIP: false - bridge: - SKIP: false -APPROACHES: - oracle_dynamics_opus_benchmark_r2: - EXTENDS: oracle_dynamics_opus - FLAGS: - continual_skill_preflight: false diff --git a/scripts/configs/predicatorv3/continual_principled_belief_r1.yaml b/scripts/configs/predicatorv3/continual_principled_belief_r1.yaml deleted file mode 100644 index 417073b6c8..0000000000 --- a/scripts/configs/predicatorv3/continual_principled_belief_r1.yaml +++ /dev/null @@ -1,44 +0,0 @@ -# The paper's seven arms rerun on the principled joint belief -# (principled-belief branch, Sept 26, 2026): five benchmark settings, -# five seeds, current environments (the Balloons chute aligned with the -# rack since Sept 19, the ramp Fan, Boil's Place fix). Every arm but No -# uncertainty carries 16 joint draws (continual_common.yaml). Launch -# from a frozen worktree with --partition mit_preemptable --requeue -# --accounts b,c,d (account a's organization blocks Claude Code; dat is -# backup only). Seeds 0-1 go first; override START_SEED and NUM_SEEDS in -# a second launch for seeds 2-4. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 2 -FLAGS: - continual_skill_preflight: false - continual_validation_audit: false -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - mb_opus_joint_r1: - EXTENDS: mb_opus - mf_opus_joint_r1: - EXTENDS: mf_opus - mf_scene_package_opus_joint_r1: - EXTENDS: mf_scene_package_opus - standalone_opus_joint_r1: - EXTENDS: standalone_opus - oracle_dynamics_opus_joint_r1: - EXTENDS: oracle_dynamics_opus - no_fitting_opus_joint_r1: - EXTENDS: no_fitting_opus - no_uncertainty_opus_joint_r1: - EXTENDS: no_uncertainty_opus diff --git a/scripts/configs/predicatorv3/continual_principled_belief_smoke_r1.yaml b/scripts/configs/predicatorv3/continual_principled_belief_smoke_r1.yaml deleted file mode 100644 index 0733350090..0000000000 --- a/scripts/configs/predicatorv3/continual_principled_belief_smoke_r1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -# Smoke runs of the principled joint belief (principled-belief branch, -# Sept 25, 2026): EMPIRIC on the five benchmark settings, one seed each, -# before the matched comparison. The joint belief is on through -# continual_common.yaml (belief_joint_draws 16). Launch from a frozen -# worktree with --partition mit_preemptable --requeue --accounts a,b,c,d. -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 1 -FLAGS: - continual_skill_preflight: false - continual_validation_audit: false -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - mb_opus_joint_smoke_r1: - EXTENDS: mb_opus diff --git a/scripts/configs/predicatorv3/continual_real_to_sim_benchmark_r1.yaml b/scripts/configs/predicatorv3/continual_real_to_sim_benchmark_r1.yaml deleted file mode 100644 index 7b514acdf0..0000000000 --- a/scripts/configs/predicatorv3/continual_real_to_sim_benchmark_r1.yaml +++ /dev/null @@ -1,29 +0,0 @@ -# Agentic real-to-sim baseline on the five benchmark settings (Sept 18, -# 2026): the agent builds its own PyBullet scene from the engine, the -# scene manifest and the asset files (agent_continual_real_to_sim), on -# Opus, three seeds each: 5 domains x 3 seeds = 15 runs. Envs and the -# arm come from the menus (envs/continual.yaml, approaches/continual.yaml); -# the round key follows the eight-agent sweep, so the logs land under -# logs/agent_continual_real_to_sim/-real_to_sim_opus_benchmark_r1/seed. -# -# python scripts/engaging/launch.py -c predicatorv3/continual_real_to_sim_benchmark_r1.yaml --partition mit_preemptable,mit_normal --requeue --accounts a,b,c,d -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 3 -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - real_to_sim_opus_benchmark_r1: - EXTENDS: real_to_sim_opus diff --git a/scripts/configs/predicatorv3/continual_scene_only_benchmark_r1.yaml b/scripts/configs/predicatorv3/continual_scene_only_benchmark_r1.yaml deleted file mode 100644 index fe4ccc7453..0000000000 --- a/scripts/configs/predicatorv3/continual_scene_only_benchmark_r1.yaml +++ /dev/null @@ -1,30 +0,0 @@ -# Scene-only ablation on the five benchmark settings (Sept 18, 2026): -# the exact scene twin with corrected base calibration, no mechanism -# code, frozen for the run (agent_continual_scene_only, formerly -# "oracle scene"), on Opus, three seeds each: 5 domains x 3 seeds = 15 -# runs. Envs and the arm come from the menus (envs/continual.yaml, -# approaches/continual.yaml); the round key matches the eight-agent -# sweep so the logs land under logs/agent_continual_scene_only/ -# -scene_only_opus_benchmark_r1/seed either way. -# -# python scripts/engaging/launch.py -c predicatorv3/continual_scene_only_benchmark_r1.yaml --partition mit_preemptable,mit_normal --requeue --accounts c,d -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 3 -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - scene_only_opus_benchmark_r1: - EXTENDS: scene_only_opus diff --git a/scripts/configs/predicatorv3/continual_standalone_no_uncertainty_r2.yaml b/scripts/configs/predicatorv3/continual_standalone_no_uncertainty_r2.yaml deleted file mode 100644 index f934a3a5f4..0000000000 --- a/scripts/configs/predicatorv3/continual_standalone_no_uncertainty_r2.yaml +++ /dev/null @@ -1,37 +0,0 @@ -# Relaunch of two comparison arms on the five benchmark settings, seed 0 -# only (Sept 18, 2026); more seeds follow once these look right. -# -# - Standalone sim, closer to WorldCoder: the run_python probe scores the -# agent's world_model.py on the recorded data and rolls a plan through -# it once; plan search (sim.refine), repeated-trial rollouts, predicate -# scoring and engine renders of predicted states are withheld. -# - No explicit uncertainty, with the observation noise undeclared: no -# noise section, no [noise] line, and the harness fit models none of it. -# -# The r1 rounds of these arms had the fuller probe and the declared noise; -# their logs stay under the _benchmark_r1 and _raw_obs_opus_r1 keys. -# -# Launched from a frozen worktree: -# python scripts/engaging/launch.py -c predicatorv3/continual_standalone_no_uncertainty_r2.yaml --partition mit_preemptable,mit_normal --requeue --accounts a,b,c,d -includes: - - continual_common.yaml - - envs/continual.yaml - - approaches/continual.yaml -START_SEED: 0 -NUM_SEEDS: 1 -ENVS: - balloons: - SKIP: false - bridge: - SKIP: false - boil: - SKIP: false - fan: - SKIP: false - domino_high_friction_turn: - SKIP: false -APPROACHES: - standalone_opus_benchmark_r2: - EXTENDS: standalone_opus - no_uncertainty_opus_benchmark_r2: - EXTENDS: no_uncertainty_opus diff --git a/scripts/configs/predicatorv3/envs/all.yaml b/scripts/configs/predicatorv3/envs/all.yaml deleted file mode 100644 index 3ad8e0dcac..0000000000 --- a/scripts/configs/predicatorv3/envs/all.yaml +++ /dev/null @@ -1,352 +0,0 @@ -ENVS: - domino: - NAME: "pybullet_domino" - SKIP: True - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: "loc,rot,angle,direction" - horizon: 500 - domino_initialize_at_finished_state: False - domino_use_domino_blocks_as_target: True - domino_use_continuous_place: True - process_planning_heuristic_weight: 2.0 - domino_has_glued_dominos: False - keep_failed_demos: True - predicate_invent_invent_derived_predicates: True - pybullet_birrt_extend_num_interp: 20 - pybullet_birrt_path_subsample_ratio: 2 - domino_turns: - NAME: "pybullet_domino" - SKIP: True - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: "loc,rot,angle,direction" - horizon: 500 - domino_initialize_at_finished_state: False - domino_use_domino_blocks_as_target: True - domino_use_continuous_place: True - process_planning_heuristic_weight: 2.0 - domino_has_glued_dominos: False - keep_failed_demos: True - predicate_invent_invent_derived_predicates: True - pybullet_birrt_extend_num_interp: 20 - pybullet_birrt_path_subsample_ratio: 2 - domino_test_turn_ratio: 1.0 - # Min-block friction-sysID tasks (reward = toppled - block_cost * blues); one - # block per mismatch direction. Forward: true friction below the planner's. - domino_low_friction: - NAME: "pybullet_domino" - SKIP: True - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: "loc,rot,angle,direction" - horizon: 500 - domino_initialize_at_finished_state: False - domino_use_domino_blocks_as_target: True - domino_use_continuous_place: True - process_planning_heuristic_weight: 2.0 - domino_has_glued_dominos: False - keep_failed_demos: True - predicate_invent_invent_derived_predicates: True - pybullet_birrt_extend_num_interp: 20 # raise to avoid placement collisions - pybullet_birrt_path_subsample_ratio: 2 - domino_min_block_tasks: True - domino_true_friction: 0.1 - domino_planning_friction: 0.5 - domino_min_block_span_lo: 0.13 - domino_min_block_span_hi: 0.30 - domino_min_block_num_blues: 4 - # Reverse: true friction above the planner's, so it over-builds and pays - # per-block cost. Geometry retuned 2026-07-12 (probe_min_block_bands.py). - domino_high_friction: - NAME: "pybullet_domino" - SKIP: True - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: "loc,rot,angle,direction" - horizon: 500 - domino_initialize_at_finished_state: False - domino_use_domino_blocks_as_target: True - domino_use_continuous_place: True - process_planning_heuristic_weight: 2.0 - domino_has_glued_dominos: False - keep_failed_demos: True - predicate_invent_invent_derived_predicates: True - pybullet_birrt_extend_num_interp: 20 # raise to avoid placement collisions - pybullet_birrt_path_subsample_ratio: 2 - domino_min_block_tasks: True - # Spans / turn legs: true K* 1 (2 on turns) vs believed 2 (3); four - # staged blues give the believed build a spare. - domino_true_friction: 0.5 - domino_planning_friction: 0.1 - domino_min_block_span_lo: 0.29 - domino_min_block_span_hi: 0.31 - domino_min_block_turn_entry_lo: 0.21 - domino_min_block_turn_entry_hi: 0.24 - domino_min_block_turn_exit_lo: 0.17 - domino_min_block_turn_exit_hi: 0.20 - domino_min_block_num_blues: 4 - domino_block_cost: 0.1 # doubled so one extra blue shows as a 0.1 gap - online_learning_early_stopping_ignore_reward_bar: True - # domino_high_friction with all-turn test tasks; exp_domino.yaml's env. - domino_high_friction_turn: - NAME: "pybullet_domino" - SKIP: True - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: "loc,rot,angle,direction" - horizon: 500 - domino_initialize_at_finished_state: False - domino_use_domino_blocks_as_target: True - domino_use_continuous_place: True - process_planning_heuristic_weight: 2.0 - domino_has_glued_dominos: False - keep_failed_demos: True - predicate_invent_invent_derived_predicates: True - pybullet_birrt_extend_num_interp: 20 # raise to avoid placement collisions - pybullet_birrt_path_subsample_ratio: 2 - domino_min_block_tasks: True - domino_true_friction: 0.5 - domino_planning_friction: 0.1 - domino_min_block_span_lo: 0.29 - domino_min_block_span_hi: 0.31 - domino_min_block_turn_entry_lo: 0.21 - domino_min_block_turn_entry_hi: 0.24 - domino_min_block_turn_exit_lo: 0.17 - domino_min_block_turn_exit_hi: 0.20 - domino_min_block_num_blues: 4 - domino_block_cost: 0.1 - domino_test_turn_ratio: 1.0 - online_learning_early_stopping_ignore_reward_bar: True - # domino_high_friction_turn on the real scene's setup (Panda + table tile); - # only the robot and table change, so differences are attributable to them. - domino_high_friction_turn_real: - NAME: "pybullet_domino_real_geometry" - SKIP: True - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: "loc,rot,angle,direction" - horizon: 500 - domino_initialize_at_finished_state: False - domino_use_domino_blocks_as_target: True - domino_use_continuous_place: True - process_planning_heuristic_weight: 2.0 - domino_has_glued_dominos: False - keep_failed_demos: True - predicate_invent_invent_derived_predicates: True - pybullet_birrt_extend_num_interp: 20 - pybullet_birrt_path_subsample_ratio: 2 - domino_min_block_tasks: True - domino_true_friction: 0.5 - domino_planning_friction: 0.1 - domino_min_block_span_lo: 0.29 - domino_min_block_span_hi: 0.31 - # Turn legs are the Fetch values and do NOT certify on the Panda (test - # split empty); recalibration is parked, see CALIBRATION_NOTES.md. - domino_min_block_turn_entry_lo: 0.21 - domino_min_block_turn_entry_hi: 0.24 - domino_min_block_turn_exit_lo: 0.17 - domino_min_block_turn_exit_hi: 0.20 - domino_min_block_num_blues: 4 - domino_block_cost: 0.1 - domino_test_turn_ratio: 1.0 - online_learning_early_stopping_ignore_reward_bar: True - pybullet_robot: "panda" - # Required for the Panda: rests the closed fingers on the 0.015 m - # domino's faces (grasp detection tolerance 0.0005). - pybullet_closed_fingers: 0.008 - # Real-world domino: learn and explore in sim on the reconstructed scene, - # test on the real Franka. The scene JSON sizes the env; set no counts here. - domino_real: - NAME: "pybullet_domino_real" - SKIP: True - # store_true flags go in ARGS (bare --flag), never in FLAGS. - ARGS: - - "make_test_images" - - "make_failure_images" - FLAGS: - num_test_tasks: 1 # one reconstructed scene - max_initial_demos: 0 # the grid oracle cannot solve the real scene - excluded_objects_in_state_str: "loc,rot,angle,direction" - horizon: 400 - domino_initialize_at_finished_state: False - domino_use_domino_blocks_as_target: True - domino_use_continuous_place: True - process_planning_heuristic_weight: 2.0 - domino_has_glued_dominos: False - keep_failed_demos: True - predicate_invent_invent_derived_predicates: True - pybullet_birrt_extend_num_interp: 20 - pybullet_birrt_path_subsample_ratio: 2 - option_model_use_gui: False - agent_bilevel_log_state: False - agent_sim_learn_oracle_sim_program: False - agent_sim_learn_oracle_sim_params: False - code_sim_learning_num_mcmc_steps: 0 - pybullet_robot: "panda" - domino_use_skill_factories: True - domino_real_scene: "/home/amberli/babyrobot/BabyRobotPredicator/scenes/domino_straight.json" - # Roles by id (green start, purple target); ignored if the scene has 'role'. - domino_real_start_id: 6 - domino_real_target_id: 5 - pybullet_closed_fingers: 0.015 # fingers on the 0.029 m real domino's faces - real_robot_execute: False - # Heavy-block (immovable obstacle) tasks, mass-only mismatch; exp_domino_heavy.yaml. - domino_heavy: - NAME: "pybullet_domino" - SKIP: True - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: "loc,rot,angle,direction" - horizon: 500 - domino_initialize_at_finished_state: False - domino_use_domino_blocks_as_target: True - domino_use_continuous_place: True - process_planning_heuristic_weight: 2.0 - domino_has_glued_dominos: False - keep_failed_demos: True - predicate_invent_invent_derived_predicates: True - pybullet_birrt_extend_num_interp: 20 # raise to avoid placement collisions - pybullet_birrt_path_subsample_ratio: 2 - domino_heavy_block_tasks: True - domino_min_block_num_blues: 4 - domino_test_turn_ratio: 1.0 - # Flags from the boil oracle test; boil has a PO simulator. exp_boil_sweep.yaml. - boil: - NAME: "pybullet_boil" - SKIP: True - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: "switch" - max_num_steps_option_rollout: 100 - horizon: 500 - boil_goal: "simple" - boil_require_jug_full_to_heatup: True - script_option_file_name: "boil.txt" - boil_water_fill_speed: 0.0015 - pybullet_birrt_path_subsample_ratio: 2 - boil_num_jugs_train: [1] - boil_num_jugs_test: [2] - boil_num_burner_train: [1] - boil_num_burner_test: [1] - fan: - NAME: "pybullet_fan" - SKIP: True - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: "switch" - terminate_on_goal_reached: True - process_planning_heuristic_weight: 10.0 - horizon: 500 - pybullet_birrt_path_subsample_ratio: 2 - process_planning_max_execution_replans: 3 # safety net for learned models - # Test tasks: maze layouts on the full 10 x 9 arena grid (see - # fan_test_task_generation in settings.py). - fan_train_num_pos_x: 3 - fan_train_num_pos_y: 3 - fan_test_num_pos_x: 10 - fan_test_num_pos_y: 9 - fan_train_num_walls_per_task: [1] - fan_test_num_walls_per_task: [16, 20, 24] - fan_train_task_generation: "uniform" - fan_test_task_generation: "maze" - fan_maze_min_segments: 4 - fan_maze_min_path_len: 10 - fan_maze_max_segment_len: 4 - bridge: - NAME: "pybullet_bridge" - SKIP: True - FLAGS: - max_initial_demos: 0 - horizon: 3000 - # Real episodes take 600-950 steps; 2000 leaves headroom for a 20-25 - # option plan while bounding a looping policy's data. - max_num_steps_interaction_request: 2000 - # Press 3 N into the support before release (post-release drift 4.5 -> 0.2 mm). - skill_place_settle_preload_force: 3.0 - process_planning_heuristic_weight: 10.0 # unweighted skeleton search is slow - # Each Wait ends on the first atom change, so multi-cure tails need a - # cheap replan-from-current-state. - process_planning_max_execution_replans: 3 - # The packed staging grid grazes by 2-3 mm; the default 1 mm margin - # makes those unrecoverable BiRRT rejections. - pybullet_birrt_contact_margin: -0.005 - # Kinematic pin of welded assemblies while held (carried bonded beams - # drift ~9-12 deg unpinned, 0 pinned). - pybullet_pin_held_weld_assemblies: True - busyboard: - NAME: "pybullet_busyboard" - # Parked by default; exp_busyboard.yaml un-skips it. - SKIP: True - FLAGS: - max_initial_demos: 0 - # A press is ~22 low-level steps and a lamp needs ~48 driven ones - # to light, so a three-press plan runs past the common 500 cap and - # every refinement would be rejected on the horizon check. - horizon: 2000 - # REQUIRED here, for the same reason pybullet_bridge needs it. A - # lamp's lighting is a delayed effect of a drive condition, so the - # tick it lands on depends on how many low-level steps the - # surrounding options happen to take; the symbolic delay places it - # on one tick and physics may deliver it on the neighbouring one, - # and the per-step atom check then rejects a plan that reaches the - # goal. Measured 2026-09-01 on oracle_process_planning over 10 test - # tasks: 8/10 with the check on, 10/10 with it off, at either - # candidate delay value. - sesame_check_expected_atoms: False - # Every button press that lights nothing is free on this board, so - # nothing in the goal stops a plan from latching extra buttons. The - # necessity gate refuses a capture that still reaches the goal with - # a step removed (run_20260902_152811 pressed three of four buttons - # for a two-press goal). - agent_plan_validation_necessity: True - icerink: - NAME: "pybullet_icerink" - # Parked by default; a launcher built on continual_common.yaml un-skips it. - SKIP: True - FLAGS: - max_initial_demos: 0 - # A push is ~60 low-level steps and a slide settles within ~40 - # more; a three-push plan plus a wait runs past the common 500 cap. - horizon: 2000 - # A slide lands on its target a few ticks after the push option - # ends, so the per-step atom check would reject a plan that - # reaches the goal (same reason as pybullet_busyboard). - sesame_check_expected_atoms: False - pybullet_birrt_path_subsample_ratio: 2 - launcher: - NAME: "pybullet_launcher" - SKIP: True - FLAGS: - max_initial_demos: 0 - horizon: 1500 - # The top block topples a few ticks after the launch option ends. - sesame_check_expected_atoms: False - pybullet_birrt_path_subsample_ratio: 2 - magnets: - NAME: "pybullet_magnets" - SKIP: True - FLAGS: - max_initial_demos: 0 - horizon: 1500 - # A pulled piece settles under the tip a few ticks after the move. - sesame_check_expected_atoms: False - pybullet_birrt_path_subsample_ratio: 2 - balloons: - NAME: "pybullet_balloons" - SKIP: True - FLAGS: - max_initial_demos: 0 - horizon: 1500 - # The box rises to the band a few ticks after the last tie. - sesame_check_expected_atoms: False - pybullet_birrt_path_subsample_ratio: 2 - crane: - NAME: "pybullet_crane" - SKIP: True - FLAGS: - max_initial_demos: 0 - horizon: 1500 - # The crate settles on the bin some ticks after the pull option ends. - sesame_check_expected_atoms: False - pybullet_birrt_path_subsample_ratio: 2 diff --git a/scripts/configs/predicatorv3/envs/continual.yaml b/scripts/configs/predicatorv3/envs/continual.yaml deleted file mode 100644 index 9a44c0d52d..0000000000 --- a/scripts/configs/predicatorv3/envs/continual.yaml +++ /dev/null @@ -1,266 +0,0 @@ -# Continual-protocol environment menu. Every entry is parked (SKIP); a -# launcher un-parks the one it runs (see continual_common.yaml). Env FLAGS -# override the approach's and the common ones, so per-domain noise levels -# and level budgets live here. Keys are the env half of the experiment id. -# -# The benchmark settings (Sept 18, 2026) are balloons (composition test -# levels), bridge (four-span test row), boil (two-jug test level), fan -# (maze test level) and domino_high_friction_turn; the other entries are -# the earlier or alternative splits they replaced. -ENVS: - # Balloons: the main tree's benchmark, composition test levels (every test - # level composes lifts the - # agent measured on training racks; the weakest-first release bursts) and - # the 25-step goal dwell. Round 1 (Sept 16, 2026) ran with dwell 1, which - # let a swinging box win at a turning point; not comparable. - balloons: - NAME: pybullet_balloons - SKIP: true - FLAGS: - max_initial_demos: 0 - horizon: 1500 - sesame_check_expected_atoms: false - pybullet_birrt_path_subsample_ratio: 2 - balloons_require_jam_decoy: false - balloons_goal_dwell_steps: 25 - continual_obs_noise_position: 0.01 - continual_obs_noise_orientation: 0.02 - continual_obs_noise_scalar: 0.0 - continual_steps_per_level: 5000 - num_train_tasks: 2 - # Balloons, bundle test levels (Sept 17, 2026): the test rack ties its - # balloons into bundles of two, one clip per bundle, so no cut is a - # small trim and the arithmetic answer (the bundle resting in band) - # bursts on its first-cut overshoot; see the pybullet_balloons module - # doc. Training racks are the usual singles. - balloons_bundles: - NAME: pybullet_balloons - SKIP: true - FLAGS: - max_initial_demos: 0 - horizon: 1500 - sesame_check_expected_atoms: false - pybullet_birrt_path_subsample_ratio: 2 - balloons_require_jam_decoy: false - balloons_goal_dwell_steps: 25 - balloons_test_bundle_sizes: [2, 2, 2, 2] - balloons_max_sampling_attempts: 60 - continual_obs_noise_position: 0.01 - continual_obs_noise_orientation: 0.02 - continual_obs_noise_scalar: 0.0 - continual_steps_per_level: 5000 - num_train_tasks: 2 - # Bridge, three spans on both splits (the Sept 14 comparison setting). - bridge_three_span: - NAME: pybullet_bridge - SKIP: true - FLAGS: &bridge_three_span_flags - max_initial_demos: 0 - horizon: 3000 - max_num_steps_interaction_request: 2000 - skill_place_settle_preload_force: 3.0 - process_planning_heuristic_weight: 10.0 - process_planning_max_execution_replans: 3 - wait_option_max_steps: 120 - pybullet_birrt_contact_margin: -0.005 - pybullet_pin_held_weld_assemblies: true - continual_obs_noise_position: 0.005 - continual_obs_noise_orientation: 0.02 - continual_obs_noise_scalar: 0.0 - continual_steps_per_level: 10000 - num_train_tasks: 1 - bridge_train_span_blocks: 3 - bridge_test_span_blocks: 3 - # Bridge, the benchmark setting: a three-span training row, a four-span - # test row, with the Sept 16 four-span repairs (rigid grasp, lift-first - # transit, certificate that waits for the robot to withdraw). Not - # comparable with the span_transfer_r1 cohort, which ran without the - # repairs. Opus MB 3/3 at 3852 steps against MF 3/3 at 6134 (Sept 17-18). - bridge: - NAME: pybullet_bridge - SKIP: true - FLAGS: &bridge_flags - <<: *bridge_three_span_flags - bridge_train_span_blocks: 3 - bridge_test_span_blocks: 4 - pybullet_grasp_max_force: 10000.0 - bridge_lift_before_transit: true - bridge_goal_robot_clearance: 0.01 - # The same setting under the key the Sept 17 span-transfer rounds ran - # under (their logs live at bridge_span_transfer-). - bridge_span_transfer: - NAME: pybullet_bridge - SKIP: true - FLAGS: *bridge_flags - # Boil: one jug in training, two jugs on the test level, the 5 cm faucet - # tolerance (the code default since d3ac34f02, pinned here; the old 10 cm - # rows were dropped on Sept 17). - boil: - NAME: pybullet_boil - SKIP: true - FLAGS: &boil_flags - max_initial_demos: 0 - excluded_objects_in_state_str: switch - max_num_steps_option_rollout: 100 - horizon: 500 - boil_goal: simple - boil_require_jug_full_to_heatup: true - script_option_file_name: boil.txt - boil_water_fill_speed: 0.0015 - pybullet_birrt_path_subsample_ratio: 2 - boil_num_jugs_train: [1] - boil_num_jugs_test: [2] - boil_num_burner_train: [1] - boil_num_burner_test: [1] - boil_faucet_align_threshold: 0.05 - continual_obs_noise_position: 0.0125 - continual_obs_noise_orientation: 0.05 - continual_obs_noise_scalar: 0.07 - continual_steps_per_level: 5000 - num_train_tasks: 1 - # Historical maze setting, retained for reproducible development configs. - fan_maze: - NAME: pybullet_fan - SKIP: true - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: switch - terminate_on_goal_reached: true - process_planning_heuristic_weight: 10.0 - horizon: 500 - pybullet_birrt_path_subsample_ratio: 2 - process_planning_max_execution_replans: 3 - fan_train_num_pos_x: 3 - fan_train_num_pos_y: 3 - fan_test_num_pos_x: 10 - fan_test_num_pos_y: 9 - fan_train_num_walls_per_task: [1] - fan_test_num_walls_per_task: [16, 20, 24] - fan_train_task_generation: uniform - fan_test_task_generation: maze - fan_maze_min_segments: 4 - fan_maze_min_path_len: 10 - fan_maze_max_segment_len: 4 - continual_obs_noise_position: 0.005 - continual_obs_noise_orientation: 0.02 - continual_obs_noise_scalar: 0.0 - continual_steps_per_level: 5000 - num_train_tasks: 1 - # Default benchmark Fan: reviewed 3 mm ramp with longer landing. - fan: - NAME: pybullet_fan - SKIP: true - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: switch - terminate_on_goal_reached: true - process_planning_heuristic_weight: 10.0 - horizon: 500 - pybullet_birrt_path_subsample_ratio: 2 - process_planning_max_execution_replans: 3 - fan_train_num_pos_x: 3 - fan_train_num_pos_y: 3 - fan_exposed_transfer: true - fan_inertial_transfer: true - fan_ramp_transfer: true - fan_ramp_rise: 0.003 - fan_ramp_landing_extension: 0.10 - fan_train_num_walls_per_task: [0] - fan_test_num_walls_per_task: [0] - fan_test_num_pos_x: 3 - fan_test_num_pos_y: 3 - fan_train_task_generation: uniform - fan_test_task_generation: uniform - continual_obs_noise_position: 0.005 - continual_obs_noise_orientation: 0.02 - continual_obs_noise_scalar: 0.0 - continual_steps_per_level: 5000 - num_train_tasks: 1 - # Domino: min-block friction tasks with a turn, true friction above the - # planner's belief. - domino_high_friction_turn: - NAME: pybullet_domino - SKIP: true - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: loc,rot,angle,direction - horizon: 500 - domino_initialize_at_finished_state: false - domino_use_domino_blocks_as_target: true - domino_use_continuous_place: true - process_planning_heuristic_weight: 2.0 - domino_has_glued_dominos: false - keep_failed_demos: true - predicate_invent_invent_derived_predicates: true - pybullet_birrt_extend_num_interp: 20 - pybullet_birrt_path_subsample_ratio: 2 - domino_min_block_tasks: true - domino_true_friction: 0.5 - domino_planning_friction: 0.1 - domino_min_block_span_lo: 0.29 - domino_min_block_span_hi: 0.31 - domino_min_block_turn_entry_lo: 0.21 - domino_min_block_turn_entry_hi: 0.24 - domino_min_block_turn_exit_lo: 0.17 - domino_min_block_turn_exit_hi: 0.2 - domino_min_block_num_blues: 4 - domino_block_cost: 0.1 - domino_test_turn_ratio: 1.0 - online_learning_early_stopping_ignore_reward_bar: true - continual_obs_noise_position: 0.01 - continual_obs_noise_orientation: 0.04 - continual_obs_noise_scalar: 0.0 - continual_steps_per_level: 5000 - num_train_tasks: 1 - # Fan, the original test split: uniform 6 x 6 test positions with two or - # three walls (the Sept 16 Fan row; those were the runtime defaults then, - # pinned here because the maze entry above overrides them). - fan_uniform: - NAME: pybullet_fan - SKIP: true - FLAGS: - max_initial_demos: 0 - excluded_objects_in_state_str: switch - terminate_on_goal_reached: true - process_planning_heuristic_weight: 10.0 - horizon: 500 - pybullet_birrt_path_subsample_ratio: 2 - process_planning_max_execution_replans: 3 - fan_test_num_pos_x: 6 - fan_test_num_pos_y: 6 - fan_test_num_walls_per_task: [2, 3] - fan_test_task_generation: uniform - continual_obs_noise_position: 0.005 - continual_obs_noise_orientation: 0.02 - continual_obs_noise_scalar: 0.0 - continual_steps_per_level: 5000 - num_train_tasks: 1 - # Boil, one test jug (Sonnet MB 2/2 in 1240 vs MF 2/2 in 3379 on Sept 16). - boil_one_jug: - NAME: pybullet_boil - SKIP: true - FLAGS: - <<: *boil_flags - boil_num_jugs_test: [1] - # Balloons, the original Sept 16 pilot setting (the table's Balloons row): - # chute scene, original task distribution, jam decoy, dwell 1. Its - # balloons_scene and balloons_task_generation flags come from the - # continual-comparisons line (frozen worktrees), not the main tree, until - # that merge lands. - balloons_original: - NAME: pybullet_balloons - SKIP: true - FLAGS: - max_initial_demos: 0 - horizon: 1500 - sesame_check_expected_atoms: false - pybullet_birrt_path_subsample_ratio: 2 - balloons_scene: chute - balloons_task_generation: original - balloons_require_jam_decoy: true - balloons_goal_dwell_steps: 1 - continual_obs_noise_position: 0.01 - continual_obs_noise_orientation: 0.02 - continual_obs_noise_scalar: 0.0 - continual_steps_per_level: 5000 - num_train_tasks: 2 diff --git a/scripts/configs/predicatorv3/exp_boil_sweep.yaml b/scripts/configs/predicatorv3/exp_boil_sweep.yaml deleted file mode 100644 index ff29763cbb..0000000000 --- a/scripts/configs/predicatorv3/exp_boil_sweep.yaml +++ /dev/null @@ -1,119 +0,0 @@ -# BOIL sweep over every paper arm (C7 dropped), 3 seeds each. -# Usage: python scripts/engaging/launch.py -c predicatorv3/exp_boil_sweep.yaml ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -NUM_SEEDS: 3 -ENVS: - boil: - SKIP: False - FLAGS: - num_test_tasks: 5 -APPROACHES: -# Every arm is tested after every learning cycle. Learning arms skip the pre-loop -# test; U1 and A2 (no cycles) are evaluated by it, and C1 keeps it as A1 (no learning). - # C1 - sim_predicator: - SKIP: False - ARGS: - - auto_resume - # C1 in policy mode (parked; the paper's C1 is plan mode). - sim_predicator_policy: - SKIP: True - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C2 - agent_model_free_planning: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C3 - nl_world_model: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C4 - code_world_model: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C5 - gnn_dynamics: - SKIP: False - FLAGS: - skip_initial_test: True - # C6 - operator_learning: - SKIP: False - FLAGS: - skip_initial_test: True - # C8 - maple_q: - SKIP: False - FLAGS: - skip_initial_test: True - # U1 (PO and open-loop, like C1) - agent_oracle_hybrid_sim: - SKIP: False - FLAGS: - partially_observable: True - agent_bilevel_max_execution_replans: 0 - ARGS: - - auto_resume - # A2 - sim_predicator_zero_shot: - SKIP: False - ARGS: - - auto_resume - # A3 - sim_predicator_undirected_explore: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A4 (the rule-param margin samples the declared intervals here) - sim_predicator_no_param_fit: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A5 - sim_predicator_no_validation: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A6 - sim_predicator_explore_no_disagreement: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A7 - sim_predicator_validation_no_uncertainty: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A8 - sim_predicator_no_predicates: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume diff --git a/scripts/configs/predicatorv3/exp_bridge.yaml b/scripts/configs/predicatorv3/exp_bridge.yaml deleted file mode 100644 index e981ab3fdf..0000000000 --- a/scripts/configs/predicatorv3/exp_bridge.yaml +++ /dev/null @@ -1,33 +0,0 @@ -# Bridge launcher: un-skips the bridge env and the OURS arm(s). -# Usage: python scripts/local/launch_simp.py -c predicatorv3/exp_bridge.yaml --parallel ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -# Seed-3 rerun on the 2026-09-02 execution fixes (stall completion -# 621190a1b, retreat lift-off 7ac637ddd, partial-open descent 4dcc7121e, -# clearance-aware certification); prior checkpoints parked in -# saved_approaches/backup_pre_clearance_relaunch_20260902/ so auto_resume -# starts fresh. Top-level scalars override common.yaml here only. -START_SEED: 3 -NUM_SEEDS: 1 -ENVS: - bridge: - SKIP: False -APPROACHES: - agent_oracle_hybrid_sim: - SKIP: True # regression passed 2026-08-24 (1/1) - sim_predicator: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume # resume the latest per-cycle checkpoint on requeue - # Same learner; the solve deliverable is a per-task closed-loop policy.py. - sim_predicator_policy: - SKIP: True # plan arm only for the seed-3 rerun - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume diff --git a/scripts/configs/predicatorv3/exp_bridge_sweep.yaml b/scripts/configs/predicatorv3/exp_bridge_sweep.yaml deleted file mode 100644 index 4054e755cb..0000000000 --- a/scripts/configs/predicatorv3/exp_bridge_sweep.yaml +++ /dev/null @@ -1,119 +0,0 @@ -# BRIDGE sweep over every paper arm (C7 dropped), 3 seeds each. -# Usage: python scripts/engaging/launch.py -c predicatorv3/exp_bridge_sweep.yaml ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -NUM_SEEDS: 3 -ENVS: - bridge: - SKIP: False - FLAGS: - num_test_tasks: 5 -APPROACHES: -# Every arm is tested after every learning cycle. Learning arms skip the pre-loop -# test; U1 and A2 (no cycles) are evaluated by it, and C1 keeps it as A1 (no learning). - # C1 - sim_predicator: - SKIP: False - ARGS: - - auto_resume - # C1 in policy mode (parked; the paper's C1 is plan mode). - sim_predicator_policy: - SKIP: True - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C2 (same per-attempt solve budget as C1) - agent_model_free_planning: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C3 - nl_world_model: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C4 - code_world_model: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C5 - gnn_dynamics: - SKIP: False - FLAGS: - skip_initial_test: True - # C6 - operator_learning: - SKIP: False - FLAGS: - skip_initial_test: True - # C8 (same interaction budget as the agent arms) - maple_q: - SKIP: False - FLAGS: - skip_initial_test: True - # U1 (PO and open-loop, like C1) - agent_oracle_hybrid_sim: - SKIP: False - FLAGS: - partially_observable: True - agent_bilevel_max_execution_replans: 0 - ARGS: - - auto_resume - # A2 - sim_predicator_zero_shot: - SKIP: False - ARGS: - - auto_resume - # A3 - sim_predicator_undirected_explore: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A4 (the rule-param margin samples the declared intervals here) - sim_predicator_no_param_fit: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A5 - sim_predicator_no_validation: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A6 - sim_predicator_explore_no_disagreement: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A7 - sim_predicator_validation_no_uncertainty: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A8 - sim_predicator_no_predicates: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume diff --git a/scripts/configs/predicatorv3/exp_busyboard.yaml b/scripts/configs/predicatorv3/exp_busyboard.yaml deleted file mode 100644 index 0c02090b53..0000000000 --- a/scripts/configs/predicatorv3/exp_busyboard.yaml +++ /dev/null @@ -1,30 +0,0 @@ -# Thin launcher: run the busyboard experiment (one arm at a time). -# Usage: python scripts/local/launch_simp.py -c predicatorv3/exp_busyboard.yaml --parallel -# Env definitions live in envs/all.yaml and approach definitions in -# approaches/all.yaml (all parked by default); this file only un-skips the -# env + arm(s) it runs. To run a baseline sweep, flip additional arms' -# SKIP to False here. -# -# The env block carries --sesame_check_expected_atoms False, which this -# domain requires; see envs/all.yaml for why. The demonstrator -# (oracle_process_planning) solves 10/10 test tasks with it. ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -ENVS: - busyboard: - SKIP: False - # busyboard has no agent-specific excluded_predicates. The wiring - # helper predicates (SoleDriver / JointDrivers) are injected for the - # oracle only, so they are not in env.predicates and an agent already - # runs over the physical vocabulary without them. -APPROACHES: - # Oracle hybrid-sim arm: solved 1/1 with the minimal plan (job 21840975). - agent_oracle_hybrid_sim: - SKIP: True - sim_predicator: - SKIP: False - ARGS: - - auto_resume # resume the latest per-cycle checkpoint on requeue diff --git a/scripts/configs/predicatorv3/exp_domino.yaml b/scripts/configs/predicatorv3/exp_domino.yaml deleted file mode 100644 index 99c51ce1c2..0000000000 --- a/scripts/configs/predicatorv3/exp_domino.yaml +++ /dev/null @@ -1,34 +0,0 @@ -# Thin launcher: run the domino friction-sysID experiment ("ours" arm). -# Usage: python scripts/local/launch_simp.py -c predicatorv3/exp_domino.yaml --parallel -# Env definitions live in envs/all.yaml and approach definitions in -# approaches/all.yaml (all parked by default); this file only un-skips the -# env + arm(s) it runs and sets agent-specific ENVS overrides. ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -ENVS: - # NOTE: launch_simp runs the ENVS x APPROACHES cross-product - park - # arms here before launching if the full matrix is not wanted. - domino: - SKIP: True - domino_turns: - SKIP: True - # Forward (over-reach) arm: true friction 0.1, planner believes 0.5. - domino_low_friction: - SKIP: True - # Reverse (under-reach) arm: true friction 0.5, planner believes 0.1. - domino_high_friction: - SKIP: True - domino_high_friction_turn: - SKIP: False -APPROACHES: - agent_oracle_hybrid_sim: - SKIP: True - FLAGS: - agent_sim_learn_kept_predicates_names: ["Holding", "HandEmpty"] - sim_predicator: - SKIP: False - FLAGS: - skip_initial_test: True diff --git a/scripts/configs/predicatorv3/exp_domino_heavy.yaml b/scripts/configs/predicatorv3/exp_domino_heavy.yaml deleted file mode 100644 index 90d715629c..0000000000 --- a/scripts/configs/predicatorv3/exp_domino_heavy.yaml +++ /dev/null @@ -1,20 +0,0 @@ -# Thin launcher: validate the heavy-block (mass-only mismatch) domino tasks -# with the oracle-sim upper-bound arm (GT hybrid sim + GT physical params, -# i.e. the planner knows the gray block's true heavy mass). -# Usage: python scripts/local/launch_simp.py -c predicatorv3/exp_domino_heavy.yaml --parallel -# Env definitions live in envs/all.yaml and approach definitions in -# approaches/all.yaml (all parked by default); this file only un-skips the -# env + arm(s) it runs. ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -ENVS: - domino_heavy: - SKIP: False -APPROACHES: - agent_oracle_hybrid_sim: - SKIP: False - FLAGS: - agent_sim_learn_kept_predicates_names: ["Holding", "HandEmpty"] diff --git a/scripts/configs/predicatorv3/exp_domino_real.yaml b/scripts/configs/predicatorv3/exp_domino_real.yaml deleted file mode 100644 index ae214c2028..0000000000 --- a/scripts/configs/predicatorv3/exp_domino_real.yaml +++ /dev/null @@ -1,196 +0,0 @@ -# Thin launcher: real-world domino testing through the Predicators -# pipeline. -# Usage: python scripts/local/launch_simp.py -c predicatorv3/exp_domino_real.yaml ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -FLAGS: - # One cycle for the first integration run: the point is to get one episode - # all the way through record -> post-process -> fit. Raise it once that - # path is known to work; with human_reset off there is no prompt between - # episodes, so a multi-cycle run would start the second one on whatever the - # first left behind. - num_online_learning_cycles: 1 - wait_option_max_steps: 200 - # One exploration episode per cycle. common.yaml asks for 2, but the - # explorer replays the same fixed plan every time, so the second episode - # costs a full run of the hardware (and a human scene reset) to collect a - # near-duplicate of the first. - online_nsrt_learning_requests_per_cycle: 1 -NUM_SEEDS: 1 -# Agent-only env overrides: deep-merged on top of envs/all.yaml. These -# excluded_predicates are dropped for "ours" runs (matches exp_domino.yaml; -# oracle.yaml would keep them in). -ENVS: - domino_real: - SKIP: False - FLAGS: - excluded_predicates: "InitialBlock,MovableBlock,Tilting,Upright,InFront" - real_robot_execute: True - # LIVE: the arm moves and the cameras look. Learning a friction from - # real observations needs a real cascade to observe, so neither half - # can be faked -- a dry arm leaves the scene untouched and a blind run - # has nothing to report. (For the no-motion rung instead, set - # real_robot_dry True with perception "none" and the two flags below - # False; "none" with the look on raises at executor construction.) - real_robot_dry: False - # -- open-loop execution, recorded end to end -------------------------- - # The whole episode's motion ships in one batch once it has been - # simulated, so the arm runs the plan as one contiguous stroke instead - # of idling through the next option's motion planning. Mutually - # exclusive with the boundary look, which is asserted at construction: - # a look has to happen BETWEEN two options and batching leaves no such - # moment. - real_robot_open_loop_episode: True - real_robot_observe_at_option_boundary: False - # Start the take in front of Push rather than in front of the whole - # batch. Only the cascade is scored: on run_20260818_092302 the first - # onset was 107 s into a 131 s track, so the pick-and-place before it - # was ~80% of the video and none of the evidence. The arm still runs - # the bridge as one batch, then pauses once while the take opens. - # - # This also lines the track's frame 0 up with the arrangement the push - # acts on, which is what the id matching compares against -- with the - # take starting at the reset, that run could not match 2 of 4 dominoes - # and logged 122 warnings. - real_robot_record_from_option: "Push" - # The cameras record the whole execution; the poses come out of - # post-processing afterwards. NOT "zed": that is the marker pipeline, - # whose 20mm tags do not resolve at this camera distance (1 of ~7 on - # one camera, 0 on the other), and it would also fight the recorder - # for the cameras. "scene_file" replays the captured layout, which is - # what a fixed-plan replay wants -- the plan names specific objects and - # a rebuild could renumber them. - real_robot_perception: "scene_file" - real_robot_human_reset: False - real_robot_record_episodes: True - real_robot_process_takes: True - # Fit the poses from 30264679. Markerless is single-camera and the two - # are not interchangeable: on hand-measured ground truth this one is 6x - # better on orientation (1.03 deg median against 6.29), which is what - # the topple onsets are read off. The other tracks more frames (99.9% - # against 82%), so revisit if coverage turns out to matter more. - real_robot_track_camera: "30264679" - # Trim the still lead-in before SAM-2 sees it. The take starts at the - # reset and the twin then simulates every option with the arm parked: - # on run_20260817_162250 that was 152 s of a static scene out of 420 s - # recorded, ~6.5k of 18.1k frames. Needs BabyRobotPredicator's - # --trim-motion; an older driver ignores the request rather than - # failing, and test_the_driver_honours_the_trim_request_once_it_can - # says which of the two you have. - real_robot_trim_still_frames: True - # Stage 2 needs one box per domino. Draw them once at the start of the - # run, in a drag window, while a human is still at the bench -- rather - # than producing a boxes.json out of band beforehand, or having a window - # open mid-run. Valid here because the fixed plan trains and tests on - # one arrangement, so the boxes drawn on the scene as it stands are the - # right ones for every take. Needs a display (X forwarding over SSH). - # Set real_robot_snapshot_boxes_json to an earlier run's boxes.json - # instead to skip the window entirely. The 2026-08-17 capture shipped - # one, drawn on this very arrangement -- scenes/ - # domino_row_20260817_markerless/boxes.json, four boxes at 1280x720 on - # camera 30264679 -- so it is reusable only if the run records on the - # same camera at the same resolution. Check before trusting it: boxes - # from another camera land on empty table and stage 2 fits nothing. - real_robot_pick_boxes_at_start: True - real_robot_snapshot_boxes_json: "" - # -- post-processing speed --------------------------------------------- - # run_20260818_092302 took 1008 s to turn a 138 s take into a track and - # missed the fit's 900 s deadline by 108 s, so the episode it had just - # recorded was skipped. These two take roughly 5 minutes off that. - # - # 30 of this machine's 32 cores for stage 4, which ran 16.0 cores busy - # for its whole 364 s. Sized for THIS box -- lower it on a smaller one, - # and remember the pipeline runs in the background while the next - # episode drives the robot. - real_robot_track_jobs: 30 - # Skip masks_overlay.mp4: 163 s, rendered before stage 4 and so paid - # straight out of time-to-track. Turn it back on when the tracks look - # wrong -- it is how id swaps are spotted. - real_robot_track_viz: False - real_robot_divergence_atol: 0.02 - # No learning for now, just hybrid sim. - # agent_sim_learn_oracle_sim_program: True - # agent_sim_learn_oracle_sim_params: True - # -- the mismatch ------------------------------------------------------ - # Applied only to sims built with skip_process_dynamics=True (the - # approach's base env and option models), so the twin keeps the - # settings.py default 0.5 and goes on modelling the real table. - domino_planning_friction: 0.1 - # -- the scene, and the roles it does not carry ------------------------- - # The 2026-08-17 capture: four dominoes, all standing, on an arc rather - # than a row. Its records are in id order, so capture id N lands in slot - # N and is named domino_N. - domino_real_scene: "/home/amberli/babyrobot/BabyRobotPredicator/scenes/domino_row_20260817.json" - # This capture has no per-domino 'role' field, so the env reads the roles - # off these ids -- green (the one Push acts on) is capture id 3, purple - # (the goal) is capture id 0, and ids 1 and 2 are the movables the plan - # bridges with. envs/all.yaml's 6 / 5 are domino_straight.json's ids and - # appear nowhere in this scene; left in place the task has no target at - # all and _task_from_perceived asserts on it. - domino_real_start_id: 3 - domino_real_target_id: 0 - # -- learning the friction from the recording -------------------------- - # Score the free-running rollout against the markerless pose track - # instead of against every recorded state. Under open-loop nothing - # corrects the twin, so those states ARE the twin's own simulation and - # scoring them recovers the twin's friction by construction -- the - # defect this experiment exists to fix. With this off, turning the two - # flags above on makes the fit worse, not better. - code_sim_learning_rollout_score_observed_only: True - # The run manifest the recorder writes, naming each episode's track. - code_sim_learning_rollout_track_path: "logs/zed_tracks/tracks.json" - # Drop the commanded arm and the non-kinematic features from the scored - # scope. The arm reproduces at every candidate friction so it can only - # dilute -- and with it in scope nothing in the episode ever rests, so - # rest-point segmentation can never cut. - code_sim_learning_rollout_scope_types: ["domino"] - # The fit blocks this long for tracks the manifest promised. - # Post-processing runs about 3x the length of a take, and the loop fits - # as soon as an episode ends; without the wait the fit finds nothing and - # falls back to the per-step scoring above. - code_sim_learning_track_wait_s: 900.0 - # The markerless pipeline emits poses in the ROBOT BASE frame; a twin - # state is in the env's world frame. For this env the two differ by a - # quarter turn about z plus (0.75, 0.72) -- exactly - # pybullet_domino.real_geometry.base_to_world_transform, whose - # constants these mirror. Matching absorbs a translation by voting over - # candidate offsets, but not a rotation: unset, every track/twin pair - # lands 144-307 mm apart against a 40 mm tolerance and no domino is - # matched at all. - code_sim_learning_track_frame_yaw: 1.5707963267948966 - code_sim_learning_track_frame_xy: [0.75, 0.72] -APPROACHES: - agent_oracle_hybrid_sim: - SKIP: True - sim_predicator: - SKIP: False - FLAGS: - # Replay one fixed plan every episode instead of planning. What is - # being tested is whether perception feeds the learner well enough to - # move the friction belief, so the exploration half should be a - # constant: a run that goes wrong is then the loop's fault and not the - # planner's, and every episode is directly comparable to the last. - # (explorer is only ever set in approaches/all.yaml, never in - # envs/all.yaml, so this override is not shadowed by the env block.) - explorer: "fixed_plan" - # Matched to domino_real_scene above: the plan names specific objects and - # specific world coordinates, so the two move together or the replay - # places dominoes into empty table. The sketch's header carries the - # geometry it was derived from. One alternate for this same scene sits - # beside it -- _release055, the same plan with the placement drop cut - # from 29 mm to 9 mm, for if the real placements bounce or land tilted. - fixed_plan_explorer_path: "scripts/plan_sketches/domino_row_20260817_bridge2.txt" - # The agent's own plan-testing simulator is built with - # skip_process_dynamics=agent_planner_use_base_simulator, and only a - # skip_process_dynamics=True env picks up domino_planning_friction. - # Left False (the default) the agent tests its plans against the TRUE - # friction -- its belief is not mismatched at all, and there is - # nothing for the sysID to discover. True is what makes the agent - # actually believe 0.1. - agent_planner_use_base_simulator: True - # The pre-loop test is a second full episode on the hardware before - # anything is learned; the experiment is about the cycle. - skip_initial_test: True diff --git a/scripts/configs/predicatorv3/exp_domino_real_geometry.yaml b/scripts/configs/predicatorv3/exp_domino_real_geometry.yaml deleted file mode 100644 index 79184f8153..0000000000 --- a/scripts/configs/predicatorv3/exp_domino_real_geometry.yaml +++ /dev/null @@ -1,58 +0,0 @@ -# Thin launcher: the domino_high_friction_turn system-ID experiment, run on -# the real scene's physical setup -- Franka Panda on its short pedestal and -# the extended table tile -- instead of the Fetch on a flat base. -# -# Usage: -# python scripts/local/launch_simp.py -c predicatorv3/exp_domino_real_geometry.yaml -# -# This is exp_domino.yaml with one env swapped. Tasks, friction mismatch -# (true 0.5 / believed 0.1), span and turn-leg bands, turn ratio, blue budget -# and reward semantics are identical to the domino_high_friction_turn arm, and -# the dominoes are the same simulated blocks. The robot and the table are the -# only things that move, so the two runs answer "does this result survive the -# real robot's kinematics?" and nothing else. -# -# Because the blocks did not change, the 2026-07-12 band calibration carries -# over unchanged -- those bands are a property of the blocks and the friction -# pair. Re-probe only if you change one of those: -# python scripts/domino_debug/probe_min_block_bands.py reach \ -# --frictions 0.1 0.5 --env pybullet_domino_real_geometry --robot panda -# -# This is NOT exp_domino_real.yaml. That one runs the pybullet_domino_real -# env, which discards the generated tasks and rebuilds a single task from a -# perceived scene JSON; the min-block machinery never runs there. This one -# keeps every generated task and changes only the physical setup. ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -FLAGS: - num_online_learning_cycles: 3 -ENVS: - # launch_simp runs the ENVS x APPROACHES cross-product - park arms here - # before launching if the full matrix is not wanted. - domino: - SKIP: True - domino_turns: - SKIP: True - domino_low_friction: - SKIP: True - domino_high_friction: - SKIP: True - # The simulated-setup arm this one is paired against. Un-skip it to run - # both and compare; parked by default so a launch is the real-setup arm - # alone. - domino_high_friction_turn: - SKIP: True - domino_high_friction_turn_real: - SKIP: False -APPROACHES: - agent_oracle_hybrid_sim: - SKIP: True - FLAGS: - agent_sim_learn_kept_predicates_names: ["Holding", "HandEmpty"] - sim_predicator: - SKIP: False - FLAGS: - skip_initial_test: True diff --git a/scripts/configs/predicatorv3/exp_domino_real_replay.yaml b/scripts/configs/predicatorv3/exp_domino_real_replay.yaml deleted file mode 100644 index 0f3093b46a..0000000000 --- a/scripts/configs/predicatorv3/exp_domino_real_replay.yaml +++ /dev/null @@ -1,65 +0,0 @@ -# Replay: the domino friction fit, scored against an ALREADY-RECORDED track. -# Usage: -# python scripts/local/launch_simp.py -c predicatorv3/exp_domino_real_replay.yaml -# -# Same experiment as exp_domino_real.yaml, with the two slow halves removed: -# the arm does not move and the cameras do not look. Everything downstream of -# the track -- id matching, interval residuals, the parameter sweep, and the -# agent's decision about what to declare -- runs exactly as it does live. -# -# WHY THIS EXISTS. One live episode costs a scene reset, ~110 s of arm motion -# and ~3 min of markerless post-processing, and it produces the same track -# every time the layout is the same. The objective is what has been changing, -# not the data, so iterating on the objective against a known-good recording -# is the loop that matters. run_20260820_123606 is that recording: four -# dominoes cascaded 3 -> 2 -> 1 -> 0, the twin reproduced all four, and the -# ids match 5 of 5 offline. -# -# WHAT STILL RUNS. The twin simulates the fixed plan, which is where the -# recorded trajectory comes from -- that half is pure PyBullet and was never -# the slow part. The arm is dry, so the plan's motion is a no-op at the -# hardware boundary, and perception is the captured scene file rather than a -# camera. The trajectory this produces is the same computation the live run -# performed, because under open-loop nothing corrects the twin mid-episode. -# -# WHAT THIS CANNOT TELL YOU. Whether the real world would have cascaded -# differently at a different friction. The track is fixed, so the replay -# answers "what does the fit do with this evidence", never "is the evidence -# right". Re-record when the scene or the plan changes. ---- -includes: - - exp_domino_real.yaml -ENVS: - domino_real: - FLAGS: - # -- the arm and the cameras, both off --------------------------------- - # Dry: no arm is built and arm calls are no-ops, so the plan is - # simulated and then dropped at the hardware boundary. The executor is - # still attached (real_robot_execute stays True) so the same shipping - # and batching path runs -- it just ships into nothing. - real_robot_dry: True - # "scene_file" replays domino_real_scene: cameraless, and it reports the - # captured layout, which is what the twin has to start from for its - # trajectory to match the recorded run's. - real_robot_perception: "scene_file" - # No takes, so no ZED session, no SVOs and no markerless pipeline. This - # also stops tracks.json being rewritten, which is what makes the frozen - # manifest below safe to point at. - real_robot_record_episodes: False - real_robot_process_takes: False - # Nothing to reset between episodes when nothing moved, and nothing to - # draw boxes on when no camera looked. Both of these BLOCK on a human - # (a terminal prompt and an OpenCV drag window), which would defeat the - # point of a replay. - real_robot_human_reset: False - real_robot_pick_boxes_at_start: False - real_robot_snapshot_rebuild: False - # -- the evidence ------------------------------------------------------ - # The frozen copy, NOT logs/zed_tracks/tracks.json: that file is - # rewritten by every live run, so a replay pointed at it would silently - # start scoring whatever was recorded most recently. - code_sim_learning_rollout_track_path: "logs/zed_tracks/replay_20260820_124013.json" - # The track is already on disk and complete, so there is nothing to wait - # for. Left long enough to be a real error rather than a hang if the - # manifest ever points somewhere wrong. - code_sim_learning_track_wait_s: 30 diff --git a/scripts/configs/predicatorv3/exp_domino_sweep.yaml b/scripts/configs/predicatorv3/exp_domino_sweep.yaml deleted file mode 100644 index fae7b1b2ea..0000000000 --- a/scripts/configs/predicatorv3/exp_domino_sweep.yaml +++ /dev/null @@ -1,118 +0,0 @@ -# DOMINO sweep over every paper arm (C7 dropped), 3 seeds each. -# Usage: python scripts/engaging/launch.py -c predicatorv3/exp_domino_sweep.yaml ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -NUM_SEEDS: 3 -ENVS: - domino_high_friction_turn: # exp_domino.yaml's env - SKIP: False - FLAGS: - num_test_tasks: 5 -APPROACHES: -# Every arm is tested after every learning cycle. Learning arms skip the pre-loop -# test; U1 and A2 (no cycles) are evaluated by it, and C1 keeps it as A1 (no learning). - # C1 - sim_predicator: - SKIP: False - ARGS: - - auto_resume - # C1 in policy mode (parked; the paper's C1 is plan mode). - sim_predicator_policy: - SKIP: True - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C2 - agent_model_free_planning: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C3 - nl_world_model: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C4 - code_world_model: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C5 - gnn_dynamics: - SKIP: False - FLAGS: - skip_initial_test: True - # C6 - operator_learning: - SKIP: False - FLAGS: - skip_initial_test: True - # C8 - maple_q: - SKIP: False - FLAGS: - skip_initial_test: True - # U1 (keeps HandEmpty like exp_domino.yaml) - agent_oracle_hybrid_sim: - SKIP: False - FLAGS: - agent_sim_learn_kept_predicates_names: ["Holding", "HandEmpty"] - ARGS: - - auto_resume - # A2 - sim_predicator_zero_shot: - SKIP: False - ARGS: - - auto_resume - # A3 - sim_predicator_undirected_explore: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A4 (the rule-param margin samples the declared intervals here) - sim_predicator_no_param_fit: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A5 - sim_predicator_no_validation: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A6 - sim_predicator_explore_no_disagreement: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A7 - sim_predicator_validation_no_uncertainty: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # A8 - sim_predicator_no_predicates: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume diff --git a/scripts/configs/predicatorv3/exp_fan.yaml b/scripts/configs/predicatorv3/exp_fan.yaml deleted file mode 100644 index 9764670636..0000000000 --- a/scripts/configs/predicatorv3/exp_fan.yaml +++ /dev/null @@ -1,30 +0,0 @@ -# Thin launcher: run the fan experiment (agent_oracle_hybrid_sim arm). -# Usage: python scripts/local/launch_simp.py -c predicatorv3/exp_fan.yaml --parallel -# Env definitions live in envs/all.yaml and approach definitions in -# approaches/all.yaml (all parked by default); this file only un-skips the -# env + arm(s) it runs. To run a baseline sweep, flip additional arms' -# SKIP to False here. ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -ENVS: - fan: - SKIP: False - # fan has no agent-specific excluded_predicates. The grid predicates - # (BallAtLoc / ClearLoc / SideOf / FanFacingSide / OppositeFan) are - # helper-only (injected for the oracle), so they are not in env.predicates - # and the agent already runs grid-free over the physical vocabulary. -APPROACHES: - agent_oracle_hybrid_sim: - SKIP: True - sim_predicator: - SKIP: False - FLAGS: - skip_initial_test: True - # Ablation axis: surface the fan env's base-sim source - # (pybullet_fan_base.py + pybullet_env.py) in the agent sandbox's - # ./reference/base_sim/. The wind dynamics, task generation, and - # goal semantics live in pybullet_fan.py, which is never provided. - agent_sim_provide_base_sim_source: True diff --git a/scripts/configs/predicatorv3/exp_fan_sweep.yaml b/scripts/configs/predicatorv3/exp_fan_sweep.yaml deleted file mode 100644 index 180a958887..0000000000 --- a/scripts/configs/predicatorv3/exp_fan_sweep.yaml +++ /dev/null @@ -1,129 +0,0 @@ -# FAN sweep over every paper arm (C7 dropped), 3 seeds each. -# Usage: python scripts/engaging/launch.py -c predicatorv3/exp_fan_sweep.yaml ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -NUM_SEEDS: 3 -ENVS: - fan: - SKIP: False - FLAGS: - num_test_tasks: 5 -APPROACHES: -# Every arm is tested after every learning cycle. Learning arms skip the pre-loop -# test; U1 and A2 (no cycles) are evaluated by it, and C1 keeps it as A1 (no learning). - # C1 (base-sim source surfaced, as in exp_fan.yaml, on every arm with a base sim) - sim_predicator: - SKIP: False - FLAGS: - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume - # C1 in policy mode (parked; the paper's C1 is plan mode). - sim_predicator_policy: - SKIP: True - FLAGS: - skip_initial_test: True - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume - # C2 - agent_model_free_planning: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C3 - nl_world_model: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C4 - code_world_model: - SKIP: False - FLAGS: - skip_initial_test: True - ARGS: - - auto_resume - # C5 - gnn_dynamics: - SKIP: False - FLAGS: - skip_initial_test: True - # C6 - operator_learning: - SKIP: False - FLAGS: - skip_initial_test: True - # C8 - maple_q: - SKIP: False - FLAGS: - skip_initial_test: True - # U1 - agent_oracle_hybrid_sim: - SKIP: False - FLAGS: - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume - # A2 - sim_predicator_zero_shot: - SKIP: False - FLAGS: - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume - # A3 - sim_predicator_undirected_explore: - SKIP: False - FLAGS: - skip_initial_test: True - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume - # A4 (the rule-param margin samples the declared intervals here) - sim_predicator_no_param_fit: - SKIP: False - FLAGS: - skip_initial_test: True - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume - # A5 - sim_predicator_no_validation: - SKIP: False - FLAGS: - skip_initial_test: True - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume - # A6 - sim_predicator_explore_no_disagreement: - SKIP: False - FLAGS: - skip_initial_test: True - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume - # A7 - sim_predicator_validation_no_uncertainty: - SKIP: False - FLAGS: - skip_initial_test: True - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume - # A8 - sim_predicator_no_predicates: - SKIP: False - FLAGS: - skip_initial_test: True - agent_sim_provide_base_sim_source: True - ARGS: - - auto_resume diff --git a/scripts/configs/predicatorv3/oracle.yaml b/scripts/configs/predicatorv3/oracle.yaml deleted file mode 100644 index f6a2c94000..0000000000 --- a/scripts/configs/predicatorv3/oracle.yaml +++ /dev/null @@ -1,30 +0,0 @@ -# Thin launcher: run the oracle arm. -# Usage: python scripts/local/launch_simp.py -c predicatorv3/oracle.yaml -# Approach definitions live in approaches/all.yaml and env definitions in -# envs/all.yaml (all parked by default); this file only un-skips the env + -# arm(s) it runs and sets oracle-specific ENVS overrides. ---- -includes: - - common.yaml - - envs/all.yaml - - approaches/all.yaml -ENVS: - # Pick the env for the oracle run: fan is active; flip SKIPs to run - # domino instead. - fan: - SKIP: False - # Oracle keeps restricted push (target inferred from state). The agent - # configs rely on the codebase default (False) so the LLM can name the - # trigger domino explicitly via Push(robot, domino). - domino: - FLAGS: - domino_restricted_push: True - domino_low_friction: - FLAGS: - domino_restricted_push: True - domino_high_friction: - FLAGS: - domino_restricted_push: True -APPROACHES: - oracle: - SKIP: False diff --git a/scripts/domino_debug/__init__.py b/scripts/domino_debug/__init__.py deleted file mode 100644 index e69de29bb2..0000000000 diff --git a/scripts/domino_debug/count_turns.py b/scripts/domino_debug/count_turns.py deleted file mode 100644 index d05a91c4c1..0000000000 --- a/scripts/domino_debug/count_turns.py +++ /dev/null @@ -1,91 +0,0 @@ -"""Count % of generated domino test tasks that contain a turn. - -Uses whatever DominoTaskGenerator is currently installed on disk, so it -can be run across git versions of the generator by swapping the file in -place. -""" -import numpy as np - -from predicators import utils -from predicators.envs.pybullet_domino.components.domino_component import \ - DominoComponent -from predicators.envs.pybullet_domino.env import PyBulletDominoEnv -from predicators.envs.pybullet_domino.task_generators import \ - domino_task_generator as dtg -from predicators.settings import CFG -from predicators.structs import Task - -N_TASKS = 40 -SEED = 0 - - -def ang_diff(a: float, b: float) -> float: - """Return the smallest unsigned angle between a and b (mod pi).""" - d = (a - b) % np.pi - return min(d, np.pi - d) - - -def is_turn(task: Task, comp: DominoComponent) -> bool: - """Return True if the task's start/target dominoes differ in yaw.""" - st = task.init - sy = ty = None - for d in st.get_objects(comp.domino_type): - r, g, b = st.get(d, "r"), st.get(d, "g"), st.get(d, "b") - if abs(r - comp.start_domino_color[0]) < 1e-2 and \ - abs(g - comp.start_domino_color[1]) < 1e-2: - sy = st.get(d, "yaw") - elif abs(r - comp.target_domino_color[0]) < 1e-2 and \ - abs(b - comp.target_domino_color[2]) < 1e-2: - ty = st.get(d, "yaw") - return sy is not None and ty is not None and \ - ang_diff(sy, ty) > np.deg2rad(30) - - -def main() -> None: - """Print the percentage of generated test tasks with a turn.""" - utils.reset_config({ - "env": "pybullet_domino", - "seed": SEED, - "num_train_tasks": 0, - "num_test_tasks": N_TASKS, - "domino_initialize_at_finished_state": False, - "domino_use_domino_blocks_as_target": True, - "domino_has_glued_dominos": False, - "domino_test_num_dominos": [3], - "domino_test_num_targets": [1, 2], - "domino_test_num_pivots": [0], - }) - env = PyBulletDominoEnv(use_gui=False) - comp = env._domino_component # pylint: disable=protected-access - assert comp is not None - robot_init_state = { - "x": env.robot_init_x, - "y": env.robot_init_y, - "z": env.robot_init_z, - "fingers": env.open_fingers, - "roll": env.robot_init_roll, - "tilt": env.robot_init_tilt, - "wrist": env.robot_init_wrist, - } - gen = dtg.DominoTaskGenerator( - domino_component=comp, - robot=env._robot, # pylint: disable=protected-access - robot_init_state=robot_init_state, - additional_components=[]) - # Reproduce the env's per-seed test path: seeds 0-4, 5 tasks each. - turns = total = 0 - for seed in range(5): - rng = np.random.default_rng(seed + 10000) - tasks = gen.generate_tasks( - num_tasks=5, - rng=rng, - possible_num_dominos=CFG.domino_test_num_dominos, - possible_num_targets=CFG.domino_test_num_targets, - possible_num_pivots=CFG.domino_test_num_pivots) - turns += sum(is_turn(t.task, comp) for t in tasks) - total += len(tasks) - print(f"RESULT turns={turns}/{total} = {100.0 * turns / total:.1f}%") - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/measure_turn_diversity.py b/scripts/domino_debug/measure_turn_diversity.py deleted file mode 100644 index 28311c1dec..0000000000 --- a/scripts/domino_debug/measure_turn_diversity.py +++ /dev/null @@ -1,185 +0,0 @@ -"""Render the first 5 test tasks for seeds 0-4 (the exact tasks the agent run -used) and measure turn % with the grasp-clearance staging check (commit -50d56e940) ON ("after") vs OFF ("before"). - -Reproduces the env's generation path exactly: for seed N the test rng is -np.random.default_rng(N + CFG.test_env_seed_offset), 5 tasks per seed. - -A task is a TURN if the purple target block's yaw differs from the green start -block's yaw by > 30 deg (a straight chain shares one yaw mod pi; a turn90 chain -ends ~90 deg rotated). -""" -from typing import Any, List, Tuple - -import numpy as np -from matplotlib import pyplot as plt -from matplotlib.patches import Rectangle -from matplotlib.transforms import Affine2D - -from predicators import utils -from predicators.envs.pybullet_domino.env import PyBulletDominoEnv -from predicators.envs.pybullet_domino.task_generators import \ - domino_task_generator as dtg -from predicators.settings import CFG -from predicators.structs import EnvironmentTask - -SEEDS = [0, 1, 2, 3, 4] -TASKS_PER_SEED = 5 -OFFSET = 10000 # CFG.test_env_seed_offset - -DominoEntry = Tuple[float, float, float, str] - - -def ang_diff(a: float, b: float) -> float: - """Smallest angular difference modulo pi (radians).""" - d = (a - b) % np.pi - return min(d, np.pi - d) - - -def classify(task: EnvironmentTask, - comp: Any) -> Tuple[bool, List[DominoEntry]]: - """Return (is_turn, dominoes) for a task using start/target yaw.""" - st = task.init - dominoes, sy, ty = [], None, None - for d in st.get_objects(comp.domino_type): - r, g, b = st.get(d, "r"), st.get(d, "g"), st.get(d, "b") - x, y, yaw = st.get(d, "x"), st.get(d, "y"), st.get(d, "yaw") - if abs(r - comp.start_domino_color[0]) < 1e-2 and \ - abs(g - comp.start_domino_color[1]) < 1e-2: - role, sy = "start", yaw - elif abs(r - comp.target_domino_color[0]) < 1e-2 and \ - abs(b - comp.target_domino_color[2]) < 1e-2: - role, ty = "target", yaw - else: - role = "movable" - dominoes.append((x, y, yaw, role)) - is_turn = (sy is not None and ty is not None - and ang_diff(sy, ty) > np.deg2rad(30)) - return is_turn, dominoes - - -def build_generator(env: PyBulletDominoEnv) -> dtg.DominoTaskGenerator: - """Build a DominoTaskGenerator matching the env's config.""" - ris = { - "x": env.robot_init_x, - "y": env.robot_init_y, - "z": env.robot_init_z, - "fingers": env.open_fingers, - "roll": env.robot_init_roll, - "tilt": env.robot_init_tilt, - "wrist": env.robot_init_wrist, - } - # pylint: disable=protected-access - return dtg.DominoTaskGenerator( - domino_component=env._domino_component, # type: ignore[arg-type] - robot=env._robot, - robot_init_state=ris, - additional_components=[]) - - -def _no_block(*_a: Any, **_k: Any) -> bool: - """Stub replacement that never blocks grasp clearance.""" - return False - - -def gen_all(env: PyBulletDominoEnv, - disable_grasp: bool) -> List[Tuple[int, int, EnvironmentTask]]: - """Generate all (seed, task_idx, task) entries, optionally with the grasp- - clearance staging check disabled.""" - gen = build_generator(env) - cls = dtg.DominoTaskGenerator - # pylint: disable=protected-access - orig = cls._grasp_clearance_blocked - if disable_grasp: - cls._grasp_clearance_blocked = _no_block # type: ignore[method-assign] - try: - out: List[Tuple[int, int, EnvironmentTask]] = [] - for seed in SEEDS: - rng = np.random.default_rng(seed + OFFSET) - tasks = gen.generate_tasks( - num_tasks=TASKS_PER_SEED, - rng=rng, - possible_num_dominos=CFG.domino_test_num_dominos, - possible_num_targets=CFG.domino_test_num_targets, - possible_num_pivots=CFG.domino_test_num_pivots) - for ti, t in enumerate(tasks): - out.append((seed, ti, t)) - finally: - cls._grasp_clearance_blocked = orig # type: ignore[method-assign] - return out - - -def render(entries: List[Tuple[int, int, EnvironmentTask]], comp: Any, - title: str, path: str) -> None: - """Render a grid of task scenes and report the turn percentage.""" - nrows, ncols = len(SEEDS), TASKS_PER_SEED - fig, axes = plt.subplots(nrows, ncols, figsize=(2.4 * ncols, 2.4 * nrows)) - w, dpth = comp.domino_width, comp.domino_depth - cmap = {"start": "#2ca02c", "movable": "#6699ff", "target": "#cc66cc"} - by_key = {(s, ti): t for (s, ti, t) in entries} - turns = 0 - for r, seed in enumerate(SEEDS): - for c in range(TASKS_PER_SEED): - ax = axes[r][c] - t = by_key.get((seed, c)) - if t is None: - ax.axis("off") - continue - is_turn, dominoes = classify(t, comp) - turns += is_turn - for (x, y, yaw, role) in dominoes: - rect = Rectangle((-w / 2, -dpth / 2), - w, - dpth, - color=cmap[role]) - rect.set_transform(Affine2D().rotate(yaw).translate(x, y) + - ax.transData) - ax.add_patch(rect) - ax.set_xlim(comp.domino_x_lb - 0.05, comp.domino_x_ub + 0.05) - ax.set_ylim(comp.domino_y_lb - 0.05, comp.domino_y_ub + 0.05) - ax.set_aspect("equal") - ax.set_xticks([]) - ax.set_yticks([]) - ax.set_title( - f"seed{seed} t{c}: {'TURN' if is_turn else 'straight'}", - fontsize=9, - color="red" if is_turn else "black") - n = len(entries) - fig.suptitle( - f"{title} — {turns}/{n} turns ({100.0*turns/n:.0f}%)\n" - "green=start blue=movable(staged) purple=target", - fontsize=13) - fig.tight_layout(rect=[0, 0, 1, 0.97]) - fig.savefig(path, dpi=95) - plt.close(fig) - print(f" {title}: {turns}/{n} turns = {100.0*turns/n:.1f}% -> {path}") - - -def main() -> None: - """Generate, render, and compare turn % before/after the staging check.""" - utils.reset_config({ - "env": "pybullet_domino", - "seed": 0, - "num_train_tasks": 0, - "num_test_tasks": TASKS_PER_SEED, - "test_env_seed_offset": OFFSET, - "domino_initialize_at_finished_state": False, - "domino_use_domino_blocks_as_target": True, - "domino_has_glued_dominos": False, - "domino_test_num_dominos": [3], - "domino_test_num_targets": [1, 2], - "domino_test_num_pivots": [0], - }) - env = PyBulletDominoEnv(use_gui=False) - comp = env._domino_component # pylint: disable=protected-access - base = "/Users/ycliang/Code/predicators/scripts/domino_debug/" - for label, disable, fn in [ - ("AFTER (grasp-clearance ON = commit 50d56e940)", False, "after"), - ("BEFORE (grasp-clearance OFF = parent 50d56e940~1)", True, "before"), - ]: - entries = gen_all(env, disable) - render(entries, comp, label, f"{base}turn_diversity_{fn}.png") - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/probe_cascade.py b/scripts/domino_debug/probe_cascade.py deleted file mode 100644 index 26d7c0dff1..0000000000 --- a/scripts/domino_debug/probe_cascade.py +++ /dev/null @@ -1,113 +0,0 @@ -"""Locate where a domino cascade dies: run the recorded sketch -(Pick->Place->Push->Wait) through the REAL option model and log every domino's -roll (topple angle) after each step. - -Usage: - PYTHONPATH=. python scripts/domino_debug/probe_cascade.py \ - -""" -import logging -import sys -from typing import List - -import numpy as np - -from predicators import utils -from predicators.approaches import create_approach -from predicators.approaches.agent_sim_learning_approach import \ - AgentSimLearningApproach -from predicators.envs import get_or_create_env -from predicators.ground_truth_models import get_gt_options -from predicators.ground_truth_models.domino import processes as P -from predicators.structs import Object, State -from scripts.domino_debug.replay_domino_sketches import _FLAGS - -logging.disable(logging.CRITICAL) - - -def rolls(state: State, dominoes: List[Object]) -> str: - """Format each domino's roll angle for one-line logging.""" - return " ".join(f"{d.name}:r={state.get(d,'roll'):+.3f}" - for d in dominoes) - - -def main() -> None: - """Probe a recorded cascade step-by-step via the real option model.""" - seed, ti = int(sys.argv[1]), int(sys.argv[2]) - utils.reset_config(dict(_FLAGS, seed=seed)) - - env = get_or_create_env("pybullet_domino") - options = get_gt_options(env.get_name()) - preds, _ = utils.parse_config_excluded_predicates(env) - train_tasks = [t.task for t in env.get_train_tasks()] - approach = create_approach("agent_sim_learning", preds, options, env.types, - env.action_space, train_tasks) - assert isinstance(approach, AgentSimLearningApproach) - # pylint: disable=protected-access - approach._maybe_install_oracle_samplers() - om = approach._option_model - # pylint: enable=protected-access - assert om is not None - opt = {o.name: o for o in options} - InFront = [p for p in preds if p.name == "InFront"][0] - Toppled = [p for p in preds if p.name == "Toppled"][0] - - task = env.get_test_tasks()[ti].task - state = task.init - dominoes = sorted([o for o in state if o.type.name == "domino"], - key=lambda o: o.name) - robot = [o for o in state if o.type.name == "robot"][0] - fallen = env.fallen_threshold if hasattr(env, "fallen_threshold") else None - print(f"# seed{seed} env-task{ti}: {len(dominoes)} dominoes, " - f"fallen_threshold={fallen}") - for d in dominoes: - print(f" {d.name}: x={state.get(d,'x'):.4f} y={state.get(d,'y'):.4f} " - f"yaw={state.get(d,'yaw'):+.4f}") - print(f" goal: {sorted(str(a) for a in task.goal)}") - print(f"\ninit {rolls(state, dominoes)}") - - held = dominoes[1] - ref0, ref1 = dominoes[0], dominoes[3] - sub = { - utils.GroundAtom(InFront, [held, ref0]), - utils.GroundAtom(InFront, [ref1, held]) - } - - # pylint: disable=protected-access - # Pick(d1) - pick_params = P._pick_option_sampler(state, set(), - np.random.default_rng(0), - [robot, held]) - s = om.get_next_state_and_num_actions( - state, opt["Pick"].ground([robot, held], pick_params))[0] - # Place(d1) at generator-faithful pose for the two InFront subgoals - pp = P._place_option_sampler(s, sub, np.random.default_rng(3), [robot]) - s = om.get_next_state_and_num_actions(s, opt["Place"].ground([robot], - pp))[0] - print(f"after Place {rolls(s, dominoes)} " - f"(d1 placed at {pp[0]:.3f},{pp[1]:.3f},yaw={pp[3]:+.3f})") - print(" InFront subgoals holding: " + - str([str(a) for a in sub if a.holds(s)])) - - # Push - push = opt["Push"] - pparams = P._push_option_sampler(s, set(), np.random.default_rng(0), - [robot]) - # pylint: enable=protected-access - pg = push.ground([robot], pparams) if len(push.types) == 1 else \ - push.ground([robot, dominoes[0]], pparams) - s = om.get_next_state_and_num_actions(s, pg)[0] - print(f"after Push {rolls(s, dominoes)}") - - # Wait - wait = opt["Wait"] - wparams = np.zeros(wait.params_space.shape[0], dtype=np.float32) - s = om.get_next_state_and_num_actions(s, wait.ground([robot], wparams))[0] - print(f"after Wait {rolls(s, dominoes)}") - print("\nToppled after Wait: " + - str({d.name: Toppled.holds(s, [d]) - for d in dominoes})) - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/probe_infront_drift.py b/scripts/domino_debug/probe_infront_drift.py deleted file mode 100644 index 2bde9c28b9..0000000000 --- a/scripts/domino_debug/probe_infront_drift.py +++ /dev/null @@ -1,116 +0,0 @@ -"""Measure InFront settle-drift: place a domino at the oracle sampler's -generator-faithful nominal pose through the REAL option model (PyBullet forward -sim), then compare the settled pose to the nominal one and to the InFront -tolerance window. - -Usage: PYTHONPATH=. python \ - scripts/domino_debug/probe_infront_drift.py -""" -import logging -import sys - -import numpy as np - -from predicators import utils -from predicators.approaches import create_approach -from predicators.envs import get_or_create_env -from predicators.ground_truth_models import get_gt_options -from predicators.ground_truth_models.domino import processes as P - -logging.disable(logging.CRITICAL) - -# Same flags the replay uses so the option model + oracle samplers match. -# pylint: disable=wrong-import-position -from scripts.domino_debug.replay_domino_sketches import _FLAGS # noqa: E402 - - -def main() -> None: - """Probe InFront settle-drift for one seed/test-task index.""" - seed = int(sys.argv[1]) - ti = int(sys.argv[2]) - - utils.reset_config(dict(_FLAGS, seed=seed)) - - env = get_or_create_env("pybullet_domino") - options = get_gt_options(env.get_name()) - preds, _ = utils.parse_config_excluded_predicates(env) - train_tasks = [t.task for t in env.get_train_tasks()] - approach = create_approach("agent_sim_learning", preds, options, env.types, - env.action_space, train_tasks) - # pylint: disable=protected-access - approach._maybe_install_oracle_samplers() # type: ignore[attr-defined] - om = approach._option_model # type: ignore[attr-defined] - assert om is not None, "no option model" - - task = env.get_test_tasks()[ti].task - state = task.init - dominoes = sorted([o for o in state if o.type.name == "domino"], - key=lambda o: o.name) - robot = [o for o in state if o.type.name == "robot"][0] - print(f"# seed{seed} env-task{ti}: {len(dominoes)} dominoes") - for d in dominoes: - print(f" {d.name}: x={state.get(d,'x'):.4f} y={state.get(d,'y'):.4f} " - f"yaw={state.get(d,'yaw'):+.4f} roll={state.get(d,'roll'):+.4f}") - - InFront = [p for p in preds if p.name == "InFront"][0] - infront_holds = InFront.holds - pos_gap = 0.098 # PyBulletDominoEnv.pos_gap - pos_tol = pos_gap * 0.3 - print(f"\n# InFront window: pos_tol={pos_tol:.4f} m, ang_tol=15deg, " - f"pos_gap={pos_gap:.4f}") - - opt_by_name = {o.name: o for o in options} - Pick, Place = opt_by_name["Pick"], opt_by_name["Place"] - - # Place domino_1 to satisfy InFront(domino_1, domino_0). - held, ref = dominoes[1], dominoes[0] - sub = {utils.GroundAtom(InFront, [held, ref])} - - # 1) Pick the held domino. - # pylint: disable=protected-access - pick_params = P._pick_option_sampler(state, set(), - np.random.default_rng(0), - [robot, held]) - pick_opt = Pick.ground([robot, held], pick_params) - assert pick_opt.initiable(state), "pick not initiable" - s1, _ = om.get_next_state_and_num_actions(state, pick_opt) - print(f"\n# after Pick({held.name}): is_held={s1.get(held,'is_held'):.2f}") - - # 2) Sample the generator-faithful placement and run Place. - for trial in range(5): - rng = np.random.default_rng(100 + trial) - place_params = P._place_option_sampler(s1, sub, rng, [robot]) - nom_x, nom_y, _, nom_yaw = [float(v) for v in place_params] - place_opt = Place.ground([robot], place_params) - if not place_opt.initiable(s1): - print(f" trial{trial}: place not initiable") - continue - s2, _ = om.get_next_state_and_num_actions(s1, place_opt) - gx, gy, gyaw = (s2.get(held, "x"), s2.get(held, - "y"), s2.get(held, "yaw")) - groll = s2.get(held, "roll") - rx, ry, ryaw = (s2.get(ref, "x"), s2.get(ref, "y"), s2.get(ref, "yaw")) - infront = infront_holds(s2, [held, ref]) - # nominal InFront check (kinematic, roll=0, exactly at sampler pose) - snom = s2.copy() - snom.set(held, "x", nom_x) - snom.set(held, "y", nom_y) - snom.set(held, "yaw", nom_yaw) - snom.set(held, "roll", 0.0) - nom_infront = infront_holds(snom, [held, ref]) - print(f"\n trial{trial}: nominal place=({nom_x:.4f},{nom_y:.4f}," - f"yaw={nom_yaw:+.4f}) nominal_InFront={nom_infront}") - print(f" settled {held.name}=({gx:.4f},{gy:.4f},yaw={gyaw:+.4f}," - f"roll={groll:+.4f})") - print(f" drift dx={gx-nom_x:+.4f} dy={gy-nom_y:+.4f} " - f"dyaw={gyaw-nom_yaw:+.4f} (pos_tol={pos_tol:.4f})") - tol10 = np.sin(np.radians(10)) - is_cardinal = (abs(np.sin(ryaw)) < tol10 or abs(np.cos(ryaw)) < tol10) - cardinal = "yes" if is_cardinal else "NO" - print(f" {ref.name}=({rx:.4f},{ry:.4f},yaw={ryaw:+.4f}) " - f"cardinal={cardinal}") - print(f" => settled InFront({held.name},{ref.name}) = {infront}") - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/probe_min_block_bands.py b/scripts/domino_debug/probe_min_block_bands.py deleted file mode 100644 index d8bd6478f6..0000000000 --- a/scripts/domino_debug/probe_min_block_bands.py +++ /dev/null @@ -1,180 +0,0 @@ -"""Anchor probes for calibrating min-block task bands (spans / turn legs). - -The min-block differentiation bands are friction-pair specific: straight -spans need a window where the true friction's chain count is below the -planning friction's, and turn legs need cells where a natural corner -tops at the true friction while the believed side needs an extra blue. -This script measures both at the canonical probe anchor with the SAME -machinery task generation uses (memoized straight probes, the labeled -turn-layout family search, real Push rollouts), so its numbers transfer -to the generator's certificates. - -Used for the 2026-07-12 domino_high_friction short-leg retune (see the -env block comments in scripts/configs/predicatorv3/envs/all.yaml). Run -it whenever domino_true_friction / domino_planning_friction change: - - python scripts/domino_debug/probe_min_block_bands.py reach \ - --frictions 0.1 0.5 --span-lo 0.12 --span-hi 0.64 - python scripts/domino_debug/probe_min_block_bands.py turn \ - --frictions 0.1 0.5 --cells 0.22,0.18 0.23,0.19 --reps 3 - -Reading the output: - * reach: pick a span window where k(true) < k(planning) is stable - across --reps (repeats clear the probe memo, sampling solver-history - variance; a span whose count flips between rounds is knife-edge). - * turn: per (entry, exit) cell and friction, the first k with a - toppling layout plus per-family topple counts. "corner" (single - natural-yaw corner blue) is the agent-buildable style - a cell whose - only topplers are "pair" (the legacy 45-degree pair) is - agent-intractable and must NOT ship (that was the pre-retune - high_friction failure). On the planning-friction side, prefer cells - whose k ties are impossible: believed k should exceed the true k in - EVERY rep (single-rep believed reads flicker ~1/3 on knife-edge - cells, which is why the generator re-runs its believed certificates - twice post-staging). -""" -import argparse -import json -import time -from typing import Any, Dict, List, Sequence, Tuple - -import numpy as np - -from predicators import utils - - -def _build_env() -> Any: - """Env with the min-block flags that affect probe physics.""" - utils.reset_config({ - "env": "pybullet_domino", - "approach": "oracle", - "seed": 0, - "use_gui": False, - "domino_initialize_at_finished_state": False, - "domino_use_domino_blocks_as_target": True, - "domino_use_continuous_place": True, - "domino_has_glued_dominos": False, - "domino_use_skill_factories": True, - "skill_phase_use_motion_planning": True, - "pybullet_ik_validate": False, - "pybullet_birrt_extend_num_interp": 20, - "pybullet_birrt_path_subsample_ratio": 2, - "domino_min_block_tasks": True, - "horizon": 500, - }) - # pylint: disable-next=import-outside-toplevel - from predicators.envs.pybullet_domino.env import PyBulletDominoEnv - return PyBulletDominoEnv(use_gui=False) - - -def _turn_poses(mbu: Any, entry: float, exit_: float) -> Tuple[Any, Any]: - """Left-turn L at the canonical anchor, entry along +x (like the straight - probes).""" - sx, sy = mbu._PROBE_ANCHOR # pylint: disable=protected-access - syaw = np.pi / 2 - u = np.array([np.sin(syaw), np.cos(syaw)]) - p = np.array([-u[1], u[0]]) - t = np.array([sx, sy]) + entry * u + exit_ * p - tyaw = float(np.arctan2(-p[0], p[1])) - return (sx, sy, syaw), (float(t[0]), float(t[1]), tyaw) - - -def probe_reach(env: Any, mbu: Any, frictions: Sequence[float], - spans: Sequence[float], budget: int, - reps: int) -> Dict[str, List[Any]]: - """k = straight_span_k_star per (friction, span), ``reps`` rounds with - the memo cleared between rounds.""" - results: Dict[str, List[Any]] = {} - for rep in range(reps): - mbu._span_probe_memo.clear() # pylint: disable=protected-access - for f in frictions: - env.set_domino_physical_params(lateral_friction=f) - for span in spans: - k = mbu.straight_span_k_star(env, span, budget=budget) - results.setdefault(f"{f}|{span:.2f}", []).append(k) - print(f"reach rep{rep} f={f} span={span:.2f} -> k={k}", - flush=True) - return results - - -def probe_turn(env: Any, mbu: Any, frictions: Sequence[float], - cells: Sequence[Tuple[float, ...]], budget: int, - reps: int) -> Dict[str, List[Any]]: - """First k with a toppler per (friction, cell), with per-family topple - counts from full k-layer scans, ``reps`` times each.""" - comp = env._domino_component # pylint: disable=protected-access - results: Dict[str, List[Any]] = {} - for rep in range(reps): - for f in frictions: - env.set_domino_physical_params(lateral_friction=f) - for entry, exit_ in cells: - sp, tp = _turn_poses(mbu, entry, exit_) - push_opt = mbu._get_push_option(env) # pylint: disable=protected-access - t0 = time.time() - first_k, layers = None, [] - for k in range(budget + 1): - fams: Dict[str, int] = {} - # pylint: disable-next=protected-access - for fam, od, s, t in mbu._candidate_turn_layouts_labeled( - comp, k, sp, tp): - if mbu._layout_topples(env, od, s, t, push_opt): # pylint: disable=protected-access - fams[fam] = fams.get(fam, 0) + 1 - layers.append({"k": k, "topples": fams}) - if fams: - first_k = k - break # the K* layer is fully scanned; stop - results.setdefault(f"{f}|{entry}|{exit_}", []).append({ - "k": - first_k, - "layers": - layers, - }) - print( - f"turn rep{rep} f={f} legs=({entry},{exit_}) -> " - f"k={first_k} ({time.time() - t0:.1f}s) " - f"families={layers[-1]['topples']}", - flush=True) - return results - - -def _main() -> None: - parser = argparse.ArgumentParser(description=__doc__) - sub = parser.add_subparsers(dest="mode", required=True) - common = argparse.ArgumentParser(add_help=False) - common.add_argument("--frictions", type=float, nargs="+", required=True) - common.add_argument("--budget", type=int, default=5) - common.add_argument("--reps", type=int, default=1) - common.add_argument("--out", type=str, default="") - reach = sub.add_parser("reach", parents=[common]) - reach.add_argument("--span-lo", type=float, default=0.12) - reach.add_argument("--span-hi", type=float, default=0.64) - reach.add_argument("--span-step", type=float, default=0.04) - turn = sub.add_parser("turn", parents=[common]) - turn.add_argument("--cells", - type=str, - nargs="+", - required=True, - help="entry,exit leg pairs, e.g. 0.22,0.18") - args = parser.parse_args() - - env = _build_env() - # pylint: disable-next=import-outside-toplevel - from predicators.envs.pybullet_domino.task_generators import \ - min_block_utils as mbu - if args.mode == "reach": - n = int(round((args.span_hi - args.span_lo) / args.span_step)) + 1 - spans = [round(args.span_lo + i * args.span_step, 2) for i in range(n)] - results = probe_reach(env, mbu, args.frictions, spans, args.budget, - args.reps) - else: - cells = [tuple(float(v) for v in c.split(",")) for c in args.cells] - results = probe_turn(env, mbu, args.frictions, cells, args.budget, - args.reps) - if args.out: - with open(args.out, "w", encoding="utf-8") as fh: - json.dump(results, fh, indent=1) - print(f"saved {args.out}") - - -if __name__ == "__main__": - _main() diff --git a/scripts/domino_debug/probe_real_scene.py b/scripts/domino_debug/probe_real_scene.py deleted file mode 100644 index 37ec17534e..0000000000 --- a/scripts/domino_debug/probe_real_scene.py +++ /dev/null @@ -1,287 +0,0 @@ -"""Execute domino SKILLS on the real-bench scene IN SIMULATION and save an MP4. - -This is a sim-only debug harness. It reproduces the stock config -(``predicatorv3/exp_domino_real.yaml``) so the env, bench_setup patches, -geometry and scene match a real run exactly, grounds a hand-specified skill -sketch (Pick / Place / Push / Wait) with the domino oracle samplers, then -rolls it out through the env via ``env.step`` -- the same path -``run_testing`` uses to render its test videos -- capturing one frame per -low-level action into an MP4. - -Use it to watch a skill's arm motion + physics on the real scene and iterate on -skill geometry (grasp height, push pose, ...) without the LLM planner. - -Usage (from the predicators repo root, robot-ml env; set PYTHONHASHSEED=0): - PYTHONPATH=. python scripts/domino_debug/probe_real_scene.py - # custom sketch + a different scene + live GUI: - PYTHONPATH=. python scripts/domino_debug/probe_real_scene.py \ - --sketch "Pick:1 Place:1@6 Push:start Wait" --gui - -Sketch grammar (space-separated ``Skill[:obj[@ref]]`` tokens): - Push[:obj] topple a domino (obj: a domino id / "start" / "target"; - default "start"). Non-restricted Push grounds [robot, obj]. - Pick:obj grasp a domino. - Place:obj[@ref] place the held domino; @ref adds an InFront(obj, ref) - subgoal so the placer aims for it (else an empty goal). - Wait let the physics settle (the cascade). -Object refs: a domino id from the scene JSON, or "start"/"target" (by role). -""" -import argparse -import json -import logging -import os -from typing import Any, Callable, Dict, List, Optional, Tuple - -import numpy as np - -from predicators import utils -from predicators.envs import get_or_create_env -from predicators.ground_truth_models import get_gt_options -from predicators.ground_truth_models.domino import processes as P -from predicators.settings import CFG -from predicators.structs import Object, Predicate, State, _Option -from scripts.cluster_utils import generate_run_configs - -# This is a debug harness that deliberately pokes env / component / sampler -# internals to drive skills directly, so protected access is expected. -# pylint: disable=protected-access - - -def _load_config(config: str, scene: str | None, seed: int) -> None: - """reset_config from the stock launcher config (single source of truth), - optionally overriding the scene JSON.""" - rc = list(generate_run_configs(config))[0] - flags = dict(rc.flags) - flags.update({"env": rc.env, "approach": rc.approach, "seed": seed}) - flags.pop("log", None) # launcher arg, not a CFG flag - if scene is not None: - flags["domino_real_scene"] = scene - utils.reset_config(flags) - - -def _build_resolver( - env: Any, state: State -) -> Tuple[Callable[[str], Object], Object, Object, List[Object]]: - """id / 'start' / 'target' -> the predicators domino Object. - - Dominoes are placed in scene order, so scene index i -> object - ``domino_i``. - """ - comp = env._domino_component - with open(CFG.domino_real_scene, encoding="utf-8") as f: - scene_ids = [d["id"] for d in json.load(f)["dominoes"]] - id_to_obj = {sid: comp.dominos[i] for i, sid in enumerate(scene_ids)} - dominoes = sorted([o for o in state if o.type.name == "domino"], - key=lambda o: o.name) - start = next(d for d in dominoes if comp._StartBlock_holds(state, [d])) - target = next(d for d in dominoes if comp._TargetDomino_holds(state, [d])) - - def resolve(ref: str) -> Object: - if ref == "start": - return start - if ref == "target": - return target - return id_to_obj[int(ref)] - - return resolve, start, target, dominoes - - -def _ground_token(tok: str, state: State, opt: Dict[str, Any], - in_front: Optional[Predicate], robot: Object, - resolve: Callable[[str], Object], - rng: np.random.Generator) -> _Option: - """Ground one sketch token, sampling its parameters on ``state``. - - ``state`` is the state this option will actually start from, not the - episode's initial state -- see ``_lazy_option_policy``. - """ - name, _, rest = tok.partition(":") - objref, _, ref = rest.partition("@") - if name == "Pick": - d = resolve(objref) - params = P._pick_option_sampler(state, set(), rng, [robot, d]) - return opt["Pick"].ground([robot, d], params) - if name == "Place": - place_d = resolve(objref) if objref else None - goal = set() - if ref and in_front is not None and place_d is not None: - goal = {utils.GroundAtom(in_front, [place_d, resolve(ref)])} - params = P._place_option_sampler(state, goal, rng, [robot]) - return opt["Place"].ground([robot], params) - if name == "Push": - d = resolve(objref) if objref else resolve("start") - params = P._push_option_sampler(state, set(), rng, [robot]) - push = opt["Push"] - return (push.ground([robot], params) - if len(push.types) == 1 else push.ground([robot, d], params)) - if name == "Wait": - wait = opt["Wait"] - params = np.zeros(wait.params_space.shape[0], dtype=np.float32) - return wait.ground([robot], params) - raise ValueError(f"unknown skill in sketch: {name!r}") - - -def _lazy_option_policy(sketch: str, env: Any, robot: Object, - resolve: Callable[[str], Object], seed: int, - recorded: List[_Option]) -> Callable[[State], _Option]: - """Ground the sketch one token at a time, each against the live state.""" - tokens = sketch.split() - opt = {o.name: o for o in get_gt_options(env.get_name())} - in_front = next((p for p in env.predicates if p.name == "InFront"), None) - rng = np.random.default_rng(seed) - index = 0 - - def _option_policy(state: State) -> _Option: - nonlocal index - if index >= len(tokens): - # The rollout ends here rather than erroring - raise utils.OptionExecutionFailure("sketch exhausted") - tok = tokens[index] - index += 1 - ground = _ground_token(tok, state, opt, in_front, robot, resolve, rng) - print(f"# ground : {tok} -> {ground.simple_str()}" - f"[{', '.join(f'{float(p):.4f}' for p in ground.params)}]") - # Append the grounded option to the plan - recorded.append(ground) - return ground - - return _option_policy - - -def _dump_plan(path: str, plan: List[_Option], header: List[str]) -> None: - """Write the grounded plan in ``replay_plan.py``'s text format. - - The continuous parameters here came out of the oracle samplers, so this - file is the only record of the exact numbers that were just watched - working. ``replay_plan`` re-grounds them verbatim, which is the whole - point: the plan that reaches the Franka is the plan that was verified in - simulation, not a fresh sample that merely came from the same sampler. - - ``_Option.simple_str`` is deliberately parameter-free, so the line format - is built here. It has to satisfy ``replay_plan._LINE``, i.e. - ``Name(objs)[nums]``. - """ - lines = [f"# {h}" for h in header] - for opt in plan: - objs = ", ".join(f"{o.name}:{o.type.name}" for o in opt.objects) - params = ", ".join(f"{float(p):.6f}" for p in opt.params) - lines.append(f"{opt.name}({objs})[{params}]") - os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True) - with open(path, "w", encoding="utf-8") as f: - f.write("\n".join(lines) + "\n") - - -def main() -> None: - """Parse args, roll out the sketch on the real scene, and save the MP4.""" - ap = argparse.ArgumentParser( - description=__doc__, - formatter_class=argparse.RawDescriptionHelpFormatter) - ap.add_argument("--config", - default="predicatorv3/exp_domino_real.yaml", - help="launcher config to reproduce (env + flags + scene).") - ap.add_argument("--scene", - default=None, - help="override CFG.domino_real_scene (a capture JSON).") - ap.add_argument("--sketch", - default="Push:start Wait", - help="skill sequence to execute (see module docstring).") - ap.add_argument("--seed", type=int, default=0) - ap.add_argument("--gui", - action="store_true", - help="also open a live PyBullet window while rendering.") - ap.add_argument("--max-steps", - type=int, - default=1500, - help="max low-level env steps (Wait can be long).") - ap.add_argument("--frame-stride", - type=int, - default=2, - help="keep every Nth rendered frame in the MP4.") - ap.add_argument("--out", - default=None, - help="output mp4 path (default: " - "logs/probe_real_scene/_.mp4).") - ap.add_argument("--dump-plan", - default=None, - help="also write the grounded plan (with the sampled " - "parameters) in replay_plan.py's format, so this exact " - "rollout can be shipped to the real arm.") - args = ap.parse_args() - logging.basicConfig(level=logging.INFO, format="%(message)s") - - _load_config(args.config, args.scene, args.seed) - if args.gui: - CFG.use_gui = True - - env = get_or_create_env(CFG.env) - preds, _ = utils.parse_config_excluded_predicates(env) - task = env.get_test_tasks()[0].task - state = task.init - robot = next(o for o in state if o.type.name == "robot") - resolve, start, target, dominoes = _build_resolver(env, state) - Toppled = next(p for p in preds if p.name == "Toppled") - - print(f"# scene : {os.path.basename(CFG.domino_real_scene)} " - f"({len(dominoes)} dominoes)") - print(f"# roles : start={start.name} target={target.name}") - print(f"# sketch : {args.sketch}") - print(f"# goal : {sorted(str(a) for a in task.goal)}") - - # Filled in as the rollout grounds each token; empty until then, which is - # why the grounded plan is printed per option above rather than up front. - plan: List[_Option] = [] - # Wait terminates on an abstract-atom change (CFG.wait_option_terminate_ - # on_atom_change), so the policy needs an abstract function over the preds. - policy = utils.option_policy_to_policy( - _lazy_option_policy(args.sketch, env, robot, resolve, args.seed, plan), - abstract_function=lambda s: utils.abstract(s, preds)) - monitor = utils.VideoMonitor(env.render) - traj, _ = utils.run_policy( - policy, - env, - "test", - 0, - termination_function=lambda s: False, - max_num_steps=args.max_steps, - exceptions_to_break_on={utils.OptionExecutionFailure}, - monitor=monitor) - - final = traj.states[-1] - toppled = {d.name: bool(Toppled.holds(final, [d])) for d in dominoes} - solved = bool(all(a.holds(final) for a in task.goal)) - print(f"# steps : {len(traj.actions)}") - print(f"# toppled : {toppled}") - print(f"# solved : {solved}") - - if args.dump_plan: - # The verdict rides along in the header so a plan file found later - # still says what it did in sim. replay_plan skips '#' lines. - _dump_plan(args.dump_plan, plan, [ - f"scene : {CFG.domino_real_scene}", - f"sketch : {args.sketch}", - f"seed : {args.seed}", - f"steps : {len(traj.actions)}", - f"toppled : {toppled}", - f"solved : {solved}", - "replay : python scripts/domino_debug/replay_plan.py " - f"--plan {args.dump_plan} --scene {CFG.domino_real_scene}", - ]) - print(f"# plan : {args.dump_plan}") - - video = monitor.get_video()[::max(1, args.frame_stride)] - # save_video writes to CFG.video_dir/ (default videos/) and only - # creates video_dir itself, so make the nested subdir first. An absolute - # --out still works (os.path.join ignores video_dir for an absolute path). - out = args.out or os.path.join( - "probe_real_scene", - f"{os.path.splitext(os.path.basename(CFG.domino_real_scene))[0]}" - f"__{args.sketch.replace(' ', '_').replace(':', '-')}.mp4") - resolved = os.path.join(CFG.video_dir, out) - os.makedirs(os.path.dirname(resolved), exist_ok=True) - utils.save_video(out, video) - print( - f"# saved : {resolved} ({len(video)} frames @ {CFG.video_fps} fps)") - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/render_domino_initial_states.py b/scripts/domino_debug/render_domino_initial_states.py deleted file mode 100644 index 05fa39ee9d..0000000000 --- a/scripts/domino_debug/render_domino_initial_states.py +++ /dev/null @@ -1,83 +0,0 @@ -"""Render initial states of the domino test tasks for debugging. - -Reproduces the exact test tasks from a run (same env, seed, -test_env_seed_offset and domino flags) and saves a PNG of each test -task's initial state so failed tasks can be visualized. - -Usage: - PYTHONPATH=. python scripts/domino_debug/render_domino_initial_states.py -""" -import os - -import numpy as np -from PIL import Image - -from predicators import utils -from predicators.envs import create_new_env - -# Domino flags copied verbatim from the run namespace (info.log) so the -# generated test tasks match the run exactly. -_DOMINO_FLAGS = { - "env": "pybullet_domino", - "num_train_tasks": 1, - "num_test_tasks": 5, - "test_env_seed_offset": 10000, - "pybullet_camera_width": 1340, - "pybullet_camera_height": 720, - "domino_test_num_dominos": [3], - "domino_test_num_targets": [1, 2], - "domino_test_num_pivots": [0], - "domino_test_num_pos_x": 4, - "domino_test_num_pos_y": 3, - "domino_train_num_dominos": [2], - "domino_train_num_targets": [1], - "domino_train_num_pivots": [0], - "domino_train_num_pos_x": 3, - "domino_train_num_pos_y": 2, - "domino_use_continuous_place": True, - "domino_use_domino_blocks_as_target": True, - "domino_restricted_push": True, - "domino_only_straight_sequence_in_training": True, - "domino_use_skill_factories": True, - "domino_prune_actions": False, - "domino_has_glued_dominos": False, - "domino_some_dominoes_are_connected": False, - "domino_include_connected_predicate": False, - "domino_initialize_at_finished_state": False, - "domino_debug_layout": False, - "domino_domino_on_stairs": False, -} - -# Which 1-indexed test tasks failed in each seed (for labeling). -_FAILED = {0: {2, 3}, 2: {1, 2, 3, 5}} - -_OUT_DIR = ("logs/agent_sim_learning/" - "domino-agent_oracle_hybrid_sim_oracle_samplers/initial_states") - - -def main() -> None: - """Render and save initial states of the domino test tasks.""" - os.makedirs(_OUT_DIR, exist_ok=True) - for seed in (0, 2): - utils.reset_config({**_DOMINO_FLAGS, "seed": seed}) - # do_cache=False: a cached env keeps its seed-0 test tasks, so each - # seed must build a fresh env to regenerate its own test tasks. - env = create_new_env("pybullet_domino", do_cache=False) - tasks = env.get_test_tasks() - for idx in range(len(tasks)): - env.reset("test", idx) - rgb = np.asarray(env.render()[0], dtype=np.uint8) - task_num = idx + 1 # 1-indexed to match the run logs - status = "FAILED" if task_num in _FAILED.get(seed, set()) \ - else "solved" - fname = f"seed{seed}_task{task_num}_{status}.png" - path = os.path.join(_OUT_DIR, fname) - Image.fromarray(rgb).save( # type: ignore[no-untyped-call] - path) - goal = sorted(str(a) for a in tasks[idx].goal) - print(f"seed{seed} task{task_num} [{status}] -> {path}") - print(f" goal: {goal}") - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/render_unsolved_domino_states.py b/scripts/domino_debug/render_unsolved_domino_states.py deleted file mode 100644 index 5194d65a2b..0000000000 --- a/scripts/domino_debug/render_unsolved_domino_states.py +++ /dev/null @@ -1,173 +0,0 @@ -"""Render init-state PNGs for the unsolved domino tasks (oracle-samplers runs). - -Uses the geometry-affecting flags from the experiment command line so the -regenerated test scenes match the runs exactly (verified: seed1 = [4,4,5,4,4] -dominoes, and the seed1.t2 grasp-infeasibility matches the run). Run ONE seed -per process (task-gen RNG is shared across seeds in one interpreter). - -Usage: PYTHONPATH=. python \ - scripts/domino_debug/render_unsolved_domino_states.py -""" -import os -import sys -from typing import Any, Dict, List, Optional, Sequence, Tuple - -import numpy as np -from numpy.typing import NDArray -from PIL import Image, ImageDraw, ImageFont - -from predicators import utils -from predicators.envs import create_new_env -from predicators.structs import State - - -def _project(xyz: Sequence[float], view_matrix: Sequence[float], - proj_matrix: Sequence[float], width: int, - height: int) -> Optional[Tuple[float, float]]: - """World (x,y,z) -> (u,v) pixel using pybullet's column-major matrices.""" - V = np.array(view_matrix).reshape((4, 4), order="F") - P = np.array(proj_matrix).reshape((4, 4), order="F") - clip = P @ (V @ np.array([xyz[0], xyz[1], xyz[2], 1.0])) - if clip[3] == 0: - return None - ndc = clip[:3] / clip[3] - return ((ndc[0] * 0.5 + 0.5) * width, - (1.0 - (ndc[1] * 0.5 + 0.5)) * height) - - -def _font(size: int) -> Any: - """Load a TrueType font at the given size, falling back to a default.""" - for path in ("/System/Library/Fonts/Supplemental/Arial Bold.ttf", - "/System/Library/Fonts/Helvetica.ttc"): - try: - return ImageFont.truetype( # type: ignore[no-untyped-call] - path, size) - except Exception: # pylint: disable=broad-except - pass - try: - return ImageFont.load_default(size=size) - except TypeError: - return ImageFont.load_default() - - -def _caption(rgb: NDArray[np.uint8], lines: List[str]) -> NDArray[np.uint8]: - """Draw a header banner (top-left) with the given text lines.""" - img = Image.fromarray(rgb) # type: ignore[no-untyped-call] - draw = ImageDraw.Draw(img, "RGBA") - font = _font(22) - pad, lh = 8, 26 - w = max( - draw.textlength( # type: ignore[no-untyped-call] - t, font=font) for t in lines) - draw.rectangle([0, 0, w + 2 * pad, lh * len(lines) + pad], - fill=(0, 0, 0, 170)) - for i, t in enumerate(lines): - draw.text((pad, pad + i * lh), t, fill=(255, 255, 255), font=font) - return np.asarray(img) - - -def _annotate(rgb: NDArray[np.uint8], init_state: State, - cam: Any) -> NDArray[np.uint8]: - """Label each domino with its index at its initial-state position.""" - img = Image.fromarray(rgb) # type: ignore[no-untyped-call] - draw = ImageDraw.Draw(img) - font = _font(26) - for o in sorted([o for o in init_state if o.type.name == "domino"], - key=lambda o: o.name): - x, y, z = (init_state.get(o, "x"), init_state.get(o, "y"), - init_state.get(o, "z")) - uv = _project((x, y, z + 0.13), *cam) - if uv is None: - continue - u, v = uv - idx = o.name.split("_")[-1] - col = (int(init_state.get(o, "r") * 255), - int(init_state.get(o, "g") * 255), - int(init_state.get(o, "b") * 255)) - r = 15 - draw.ellipse([u - r, v - r, u + r, v + r], - fill=(0, 0, 0), - outline=col, - width=3) - tb = draw.textbbox((0, 0), idx, font=font) - draw.text((u - (tb[2] - tb[0]) / 2, v - (tb[3] - tb[1]) / 2 - tb[1]), - idx, - fill=(255, 255, 255), - font=font) - return np.asarray(img) - - -# 1-indexed tasks unsolved in EITHER arm, with (arms, failure-mode) labels. -UNSOLVED: Dict[int, Dict[int, Tuple[str, str]]] = { - 0: { - 1: ("both", "push-dropped"), - 2: ("both", "place-MP+InFront"), - 3: ("no_demo", "pick+place-MP") - }, - 1: { - 1: ("demo", "exec-retreat-collision"), - 3: ("both", "pick+place-MP") - }, - 2: { - 1: ("no_demo", "pick+place-MP"), - 2: ("demo", "toppled-cascade"), - 4: ("both", "pick+place-MP"), - 5: ("both", "place-MP+toppled") - }, - 3: { - 5: ("demo", "holding+InFront+place-MP") - }, -} -FLAGS: Dict[str, Any] = { - "env": "pybullet_domino", - "num_train_tasks": 1, - "num_test_tasks": 5, - "pybullet_ik_validate": False, - "pybullet_camera_width": 900, - "pybullet_camera_height": 900, - "domino_initialize_at_finished_state": False, - "domino_use_domino_blocks_as_target": True, - "domino_use_continuous_place": True, - "domino_restricted_push": True, - "domino_has_glued_dominos": False, - "pybullet_birrt_extend_num_interp": 20, - "pybullet_birrt_path_subsample_ratio": 2, -} -OUT = "logs/agent_sim_learning/unsolved_init_states" - - -def main() -> None: - """Render annotated init-state PNGs for one seed's unsolved tasks.""" - seed = int(sys.argv[1]) - os.makedirs(OUT, exist_ok=True) - utils.reset_config(dict(FLAGS, seed=seed)) - env = create_new_env("pybullet_domino", do_cache=False) - tasks = env.get_test_tasks() - counts = [ - len([o for o in t.init if o.type.name == "domino"]) for t in tasks - ] - print(f"seed{seed} domino counts per task = {counts}") - # pylint: disable=protected-access - cam = env._get_camera_matrices() # type: ignore[attr-defined] - for t1, (arms, mode) in sorted(UNSOLVED.get(seed, {}).items()): - idx = t1 - 1 - env.reset("test", idx) - rgb = np.asarray(env.render()[0], dtype=np.uint8) - rgb = _annotate(rgb, tasks[idx].init, cam) - goal_ids = ",".join( - sorted( - str(a).rsplit("_", maxsplit=1)[-1].rstrip(":domino)") - for a in tasks[idx].goal)) - rgb = _caption(rgb, [ - f"seed {seed} task {t1} ({arms})", - f"goal: Toppled({goal_ids}) fail: {mode}" - ]) - fname = f"seed{seed}_task{t1}_{arms}_{mode}.png" - Image.fromarray( # type: ignore[no-untyped-call] - rgb).save(os.path.join(OUT, fname)) - goal = sorted(str(a) for a in tasks[idx].goal) - print(f" saved {fname} | {counts[idx]} dominoes | goal={goal}") - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/replay_domino_sketches.py b/scripts/domino_debug/replay_domino_sketches.py deleted file mode 100644 index f1c83e81da..0000000000 --- a/scripts/domino_debug/replay_domino_sketches.py +++ /dev/null @@ -1,226 +0,0 @@ -"""Faithfully reproduce domino refinement failures by replaying the recorded -LLM sketches through the real bilevel refinement -- no LLM required. - -The agent's plan sketches were logged verbatim in each run's ``info.log`` -(``Sketch (attempt N):`` blocks). This script extracts them, regenerates the -deterministic test task, and runs the *exact same* ``refine_sketch`` the -pipeline uses (oracle option model + oracle samplers + subgoal checks, same -per-(sketch,refine) RNG seeding). The pass/fail outcome and the "stuck at step -K" reason therefore reproduce the run's solve-time failures deterministically. - -Run ONE seed per process (task-gen RNG is shared; see -reproduce_domino_failures). - -Usage: - PYTHONPATH=. python scripts/domino_debug/replay_domino_sketches.py \ - [--all] - --all replays every task; default replays only tasks the run - did not solve. -""" - -import logging -import re -import sys -from glob import glob -from typing import Dict, List, Optional, Tuple - -from predicators import utils -from predicators.agent_sdk import bilevel_sketch -from predicators.approaches import create_approach -from predicators.approaches.agent_sim_learning_approach import \ - AgentSimLearningApproach -from predicators.envs import get_or_create_env -from predicators.ground_truth_models import get_gt_options - -logging.disable(logging.CRITICAL) - -# Refine retries per sketch, matching the refine loop the audited runs -# ran with (their agent_bilevel_max_refine_retries setting); preserved -# here so old recordings keep replaying faithfully. -_REFINE_RETRIES = 5 - -ANSI = re.compile(r"\x1b\[[0-9;]*m") -STEP = re.compile( - r"^\s*\d+:\s*([A-Za-z]\w*)\((.*?)\)(?:\s*->\s*\{(.*)\})?\s*$") -SKETCH_HDR = re.compile(r"Sketch \(attempt (\d+)\)") -TASK_RES = re.compile( - r"\[main\.py\] Task (\d+) / \d+: (.*)|Task (\d+) / \d+: (SOLVED)") - -_FLAGS = { - "env": "pybullet_domino", - "approach": "agent_sim_learning", - "num_train_tasks": 1, - "num_test_tasks": 5, - "skill_phase_use_motion_planning": True, - "pybullet_ik_validate": False, - "demonstrator": "oracle_process_planning", - "bilevel_plan_without_sim": True, - "explorer": "agent_model_based", - "agent_sim_learn_oracle_sim_program": True, - "agent_sim_learn_oracle_sim_params": True, - "agent_sim_learn_parameterized_samplers": True, - "agent_sim_learn_oracle_samplers": True, - "execution_monitor": "subgoal_annotations", - "agent_bilevel_max_execution_replans": 2, - "horizon": 400, - "excluded_objects_in_state_str": "loc,rot,angle,direction", - "excluded_predicates": "InitialBlock,MovableBlock,Tilting,Upright", - "domino_initialize_at_finished_state": False, - "domino_use_domino_blocks_as_target": True, - "domino_use_continuous_place": True, - "domino_restricted_push": True, - "process_planning_heuristic_weight": 2.0, - "domino_has_glued_dominos": False, - "pybullet_birrt_extend_num_interp": 20, - "pybullet_birrt_path_subsample_ratio": 2, - "agent_sdk_use_local_sandbox": True, - "option_model_terminate_on_repeat": False, - "agent_planner_use_simulator": True, -} - - -def find_info_log(seed: int, arm: str) -> str: - """Return the newest matching run's info.log path for seed/arm.""" - exp = f"domino-agent_oracle_hybrid_sim_oracle_samplers_{arm}" - pat = f"logs/agent_sim_learning/{exp}/seed{seed}/run_*/info.log" - hits = sorted(glob(pat)) - if not hits: - raise SystemExit(f"no info.log at {pat}") - return hits[-1] - - -def extract_sketches(info_log: str) -> Dict[int, dict]: - """Return {task_idx (0-based): {"outcome": str, "sketches": ...}}. - - Each step is (option_name, [obj_names], raw_subgoal_str). - """ - tasks: Dict[int, dict] = {} - pending: List[List[Tuple[str, List[str], str]]] = [] - cur: Optional[List[Tuple[str, List[str], str]]] = None - with open(info_log, encoding="utf-8") as f: - for raw in f: - line = ANSI.sub("", raw.rstrip("\n")) - if SKETCH_HDR.search(line): - cur = [] - pending.append(cur) - continue - m = STEP.match(line) - if m and cur is not None: - opt, args, sg = m.group(1), m.group(2), m.group(3) or "" - objs = [ - a.split(":")[0].strip() for a in args.split(",") - if a.strip() - ] - cur.append((opt, objs, sg)) - continue - cur = None # any non-step line ends the current sketch block - tm = TASK_RES.search(line) - if tm: - ti = int(tm.group(1) or tm.group(3)) - 1 - outcome = (tm.group(2) or tm.group(4) or "").strip() - tasks[ti] = {"outcome": outcome, "sketches": pending} - pending = [] - return tasks - - -def typed_text(steps: List[Tuple[str, List[str], str]], - name_to_type: Dict[str, str]) -> str: - """Rebuild typed sketch text the option-plan parser expects.""" - lines = [] - for opt, objs, sg in steps: - typed = ", ".join(f"{o}:{name_to_type.get(o, 'object')}" for o in objs) - line = f"{opt}({typed})" - if sg: - line += f" -> {{{sg}}}" - lines.append(line) - return "\n".join(lines) - - -class _DeepestFail: - """Track the deepest (highest-index) step failure seen so far.""" - - def __init__(self) -> None: - self.idx: int = -1 - self.reason: str = "" - - def record(self, idx: int, _prefix: list, reason: str) -> None: - """on_step_fail callback: keep the deepest failure.""" - if idx > self.idx: - self.idx, self.reason = idx, reason - - -def main() -> None: - """Replay recorded sketches through the real refinement.""" - seed = int(sys.argv[1]) - arm = sys.argv[2] if len(sys.argv) > 2 else "no_demo" - replay_all = "--all" in sys.argv - - info_log = find_info_log(seed, arm) - tasks = extract_sketches(info_log) - - utils.reset_config(dict(_FLAGS, seed=seed)) - - env = get_or_create_env("pybullet_domino") - options = get_gt_options(env.get_name()) - preds, _ = utils.parse_config_excluded_predicates(env) - train_tasks = [t.task for t in env.get_train_tasks()] - approach = create_approach("agent_sim_learning", preds, options, env.types, - env.action_space, train_tasks) - assert isinstance(approach, AgentSimLearningApproach) - # pylint: disable=protected-access - approach._maybe_install_oracle_samplers() - # pylint: enable=protected-access - test_tasks = env.get_test_tasks() - name_to_type = {o.name: o.type.name for o in test_tasks[0].task.init} - - print(f"# seed{seed} {arm}: replaying recorded sketches through real " - f"refinement (oracle option-model + oracle samplers, no LLM)") - for ti in sorted(tasks): - rec = tasks[ti] - solved = rec["outcome"].upper().startswith("SOLVED") - if solved and not replay_all: - continue - task = test_tasks[ti].task - print(f"\n== task{ti} (run Task{ti+1}) | run outcome: " - f"{rec['outcome'][:60]}") - if not rec["sketches"]: - print(" (no sketches recorded)") - continue - for si, steps in enumerate(rec["sketches"]): - sketch = bilevel_sketch.parse_sketch_from_text( - typed_text(steps, name_to_type), - task, - predicates=preds, - options=set(options), - types=env.types) - if not sketch: - print(f" sketch{si}: unparseable") - continue - any_success = False - deepest_idx, deepest_reason = -1, "" - for r in range(_REFINE_RETRIES): - fail = _DeepestFail() - # attempt reproduces the audited runs' per-(sketch,refine) - # RNG seeding (rng = CFG.seed + attempt inside - # _refine_sketch), so recorded failures replay exactly. - # pylint: disable-next=protected-access - _, success = approach._refine_sketch( - task, - sketch, - timeout=600.0, - attempt=si * _REFINE_RETRIES + r, - on_step_fail=fail.record) - if success: - any_success = True - break - if fail.idx > deepest_idx: - deepest_idx, deepest_reason = fail.idx, fail.reason - verdict = "REFINED-OK" if any_success else \ - f"FAILED (stuck step {deepest_idx}: {deepest_reason[:60]})" - head = " -> ".join(f"{o}({','.join(a)})" for o, a, _ in steps) - print(f" sketch{si} [{len(steps)} steps]: {verdict}") - print(f" {head}") - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/replay_ikval_sweep.py b/scripts/domino_debug/replay_ikval_sweep.py deleted file mode 100644 index babae9c7c1..0000000000 --- a/scripts/domino_debug/replay_ikval_sweep.py +++ /dev/null @@ -1,196 +0,0 @@ -"""Replay each task's recorded sketches with ik_validate=True and compare to -the recorded (ik_validate=False) run outcome — to separate pure IK-artifact -failures from genuine geometric ones, and to check for regressions on solved -tasks. - -Mirrors the pipeline solve loop: for each task, try recorded sketches in -order, up to N refine attempts each, stop at first success; enforce a -per-task wall budget. Run ONE seed+arm per process. Usage: PYTHONPATH=. -python scripts/domino_debug/replay_ikval_sweep.py -[budget_s] -""" -import logging -import re -import sys -import time -from glob import glob -from typing import Any, Callable, Dict, List, Tuple - -logging.disable(logging.CRITICAL) - -ANSI = re.compile(r"\x1b\[[0-9;]*m") -STEP = re.compile( - r"^\s*\d+:\s*([A-Za-z]\w*)\((.*?)\)(?:\s*->\s*\{(.*)\})?\s*$") -SKH = re.compile(r"Sketch \(attempt (\d+)\)") -TRES = re.compile( - r"\[main\.py\] Task (\d+) / \d+: (.*)|Task (\d+) / \d+: (SOLVED)") - - -def extract(info_log: str) -> Dict[int, dict]: - """Parse recorded sketches and outcomes per task from an info.log.""" - tasks: Dict[int, dict] = {} - pending: List[List[Tuple[str, List[str], str]]] = [] - cur: List[Tuple[str, List[str], str]] | None = None - for raw in open(info_log, encoding="utf-8"): - line = ANSI.sub("", raw.rstrip("\n")) - if SKH.search(line): - cur = [] - pending.append(cur) - continue - m = STEP.match(line) - if m and cur is not None: - cur.append((m.group(1), [ - a.split(":")[0].strip() for a in m.group(2).split(",") - if a.strip() - ], m.group(3) or "")) - continue - cur = None - tm = TRES.search(line) - if tm: - ti = int(tm.group(1) or tm.group(3)) - 1 - tasks[ti] = { - "outcome": (tm.group(2) or tm.group(4) or "").strip(), - "sketches": pending - } - pending = [] - return tasks - - -def main() -> None: - """Replay recorded sketches with ik_validate and report flips.""" - seed = int(sys.argv[1]) - arm = sys.argv[2] - budget = float(sys.argv[3]) if len(sys.argv) > 3 else 300.0 - ikv = (sys.argv[4].lower() == "true") if len(sys.argv) > 4 else True - exp = f"domino-agent_oracle_hybrid_sim_oracle_samplers_{arm}" - info_log = sorted( - glob(f"logs/agent_sim_learning/{exp}/seed{seed}/run_*/info.log"))[-1] - rec = extract(info_log) - FLAGS = { - "env": "pybullet_domino", - "approach": "agent_sim_learning", - "seed": seed, - "num_train_tasks": 1, - "num_test_tasks": 5, - "skill_phase_use_motion_planning": True, - "pybullet_ik_validate": ikv, # <-- the change under test - "demonstrator": "oracle_process_planning", - "bilevel_plan_without_sim": True, - "explorer": "agent_model_based", - "agent_sim_learn_oracle_sim_program": True, - "agent_sim_learn_oracle_sim_params": True, - "agent_sim_learn_parameterized_samplers": True, - "agent_sim_learn_oracle_samplers": True, - "execution_monitor": "subgoal_annotations", - "agent_bilevel_max_execution_replans": 2, - "horizon": 400, - "excluded_objects_in_state_str": "loc,rot,angle,direction", - "excluded_predicates": "InitialBlock,MovableBlock,Tilting,Upright", - "domino_initialize_at_finished_state": False, - "domino_use_domino_blocks_as_target": True, - "domino_use_continuous_place": True, - "domino_restricted_push": True, - "domino_has_glued_dominos": False, - "pybullet_birrt_extend_num_interp": 20, - "pybullet_birrt_path_subsample_ratio": 2, - "agent_sdk_use_local_sandbox": True, - "option_model_terminate_on_repeat": False, - "agent_planner_use_simulator": True - } - # pylint: disable=import-outside-toplevel - # Imports are deferred until after reset_config so module-level CFG - # reads in these modules observe the FLAGS set above. - from predicators import utils - utils.reset_config(FLAGS) - from predicators.agent_sdk import bilevel_sketch - from predicators.approaches import create_approach - from predicators.envs import get_or_create_env - from predicators.ground_truth_models import get_gt_options - env = get_or_create_env("pybullet_domino") - options = get_gt_options(env.get_name()) - preds, _ = utils.parse_config_excluded_predicates(env) - # Cast to Any: this script probes approach-specific protected members - # (sampler installers, option model) absent from the BaseApproach API. - ap: Any = create_approach("agent_sim_learning", preds, options, env.types, - env.action_space, - [t.task for t in env.get_train_tasks()]) - ap._maybe_install_oracle_samplers() # pylint: disable=protected-access - n2t = {o.name: o.type.name for o in env.get_test_tasks()[0].task.init} - - def typed(steps: List[Tuple[str, List[str], str]]) -> str: - """Render parsed sketch steps as typed operator lines.""" - lines = [] - for op, objs, sg in steps: - args = ", ".join(o + ":" + n2t.get(o, "object") for o in objs) - line = op + "(" + args + ")" - if sg: - line += " -> {" + sg + "}" - lines.append(line) - return "\n".join(lines) - - header = (f"# seed{seed} {arm} | ik_validate={ikv} | " - f"NEW task-gen | budget={budget}s") - print(header) - for ti in sorted(rec): - task = env.get_test_tasks()[ti].task - recout = "SOLVED" if rec[ti]["outcome"].upper().startswith( - "SOLVED") else "FAILED" - t0 = time.perf_counter() - solved_by = None - deepest: Tuple[int, str] = (-1, "") - for si, steps in enumerate(rec[ti]["sketches"]): - if time.perf_counter() - t0 > budget: - break - sk = bilevel_sketch.parse_sketch_from_text(typed(steps), - task, - predicates=preds, - options=set(options), - types=env.types) - if not sk: - continue - for r in range(2): - if time.perf_counter() - t0 > budget: - break - fail: Dict[str, object] = {"idx": -1, "reason": ""} - - def make_rc( - f: Dict[str, - object]) -> Callable[[int, object, str], None]: - """Build an on_step_fail recording the deepest fail.""" - - def rc(i: int, _p: object, reason: str) -> None: - if i > f["idx"]: # type: ignore[operator] - f["idx"], f["reason"] = i, reason - - return rc - - # attempt reproduces this script's historical RNG streams - # (rng = CFG.seed + attempt inside _refine_sketch). - # pylint: disable-next=protected-access - _, ok = ap._refine_sketch(task, - sk, - timeout=budget, - attempt=si * 5 + r, - on_step_fail=make_rc(fail)) - if ok: - solved_by = (si, r) - break - if fail["idx"] > deepest[0]: # type: ignore[operator] - deepest = (fail["idx"], fail["reason"]) # type: ignore - if solved_by: - break - dt = time.perf_counter() - t0 - verdict = "SOLVED" if solved_by else "FAILED" - flip = "" if verdict == recout else ( - " *** REGRESSION" if recout == "SOLVED" else " *** FIXED") - if solved_by: - extra = f"by sketch{solved_by[0]}" - else: - extra = f"deepest step{deepest[0]}: {deepest[1][:32]}" - line = (f" task{ti+1}: recorded(F)={recout:6s} -> " - f"new={verdict:6s} [{dt:5.0f}s] {extra}{flip}") - print(line) - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/replay_plan.py b/scripts/domino_debug/replay_plan.py deleted file mode 100644 index 70079128bb..0000000000 --- a/scripts/domino_debug/replay_plan.py +++ /dev/null @@ -1,218 +0,0 @@ -"""Replay an EXACT solved option plan on the real-scene domino env -(deterministic, no LLM). Grounds the plan's Pick/Place/Push/Wait options with -their exact parameters and rolls them through the env in TEST mode. With -``real_robot_execute=True`` a RealRobotExecutor is attached to the env and each -option's joint trajectory is shipped to the Franka as that option ends. The -scene is NOT re-perceived between options here: this tool replays a fixed plan, -and correcting the twin mid-replay would let the option policies see states the -recorded plan was never chosen against. The default dry-run stays pure sim -(optionally rendered to MP4). - -Plan-file format (one option per line; ``-> {...}`` subgoals optional/ignored): - Pick(robot:robot, domino_1:domino)[0.06] -> {Holding(robot, domino_1)} - Place(robot:robot)[0.70, 1.16, 0.55, 1.75] - Push(robot:robot, domino_0:domino)[0.03, 0.05] - Wait(robot:robot)[] - -Three rungs, in order. Take them all; each adds exactly one new source of -failure (from the predicators repo root, robot-ml; PYTHONHASHSEED=0): - - # 1. pure sim: no executor, no robot object at all - PYTHONPATH=. python scripts/domino_debug/replay_plan.py --plan plan.txt - - # 2. dry arm: the whole RealRobot minus the arm. Attachment, per-option - # chunking and the gripper split all run; nothing moves. Needs - # babyrobot importable, needs no hardware powered on. - PYTHONPATH=.:/path/to/BabyRobotPredicator \ - python scripts/domino_debug/replay_plan.py --plan plan.txt \ - --execute --dry - - # 3. MOVES THE ARM - PYTHONPATH=.:/path/to/BabyRobotPredicator \ - python scripts/domino_debug/replay_plan.py --plan plan.txt --execute -""" -import argparse -import logging -import os -import re -from typing import List, Tuple - -import numpy as np - -from predicators import utils -from predicators.approaches import create_approach -from predicators.cogman import CogMan, run_episode_and_get_observations -from predicators.envs import get_or_create_env -from predicators.envs.pybullet_domino_real import PyBulletDominoRealEnv -from predicators.execution_monitoring import create_execution_monitor -from predicators.ground_truth_models import get_gt_options -from predicators.perception import create_perceiver -from predicators.pybullet_helpers.real_robot_executor import attach_real_robot -from predicators.settings import CFG -from scripts.cluster_utils import SingleSeedRunConfig, generate_run_configs - -# pylint: disable=protected-access -_LINE = re.compile(r"^\s*(\w+)\s*\(([^)]*)\)\s*\[([^\]]*)\]") - - -def _parse_plan(text: str) -> List[Tuple[str, List[str], List[float]]]: - """[(option_name, [obj_names], [param_floats]), ...] from the plan text.""" - steps: List[Tuple[str, List[str], List[float]]] = [] - for raw in text.splitlines(): - line = raw.split("->", 1)[0].strip() - if not line or line.startswith("#"): - continue - m = _LINE.match(line) - if not m: - continue - name, args, params = m.group(1), m.group(2), m.group(3) - objs = [ - a.split(":", 1)[0].strip() for a in args.split(",") if a.strip() - ] - floats = [float(v) for v in params.split(",") if v.strip()] - steps.append((name, objs, floats)) - return steps - - -def main() -> None: - """Ground the plan's options and roll them through the real-scene env.""" - ap = argparse.ArgumentParser( - description=__doc__, - formatter_class=argparse.RawDescriptionHelpFormatter) - ap.add_argument("--plan", required=True, help="plan text file") - ap.add_argument("--config", default="predicatorv3/exp_domino_real.yaml") - ap.add_argument( - "--scene", - default=None, - help="override CFG.domino_real_scene (must match the plan)") - ap.add_argument("--execute", - action="store_true", - help="EXECUTE ON THE REAL FRANKA (needs the babyrobot " - "submodule installed). Default: dry-run (pure sim, no " - "motion).") - ap.add_argument("--dry", - action="store_true", - help="with --execute: build the whole RealRobot but with " - "NO arm attached. The executor still attaches, every " - "option still chunks and ships, the gripper split still " - "runs -- and nothing moves. This is the rung between pure " - "sim and real motion; take it before every new plan.") - ap.add_argument("--observe", - action="store_true", - help="look at the scene between options and correct the " - "twin from what was seen (bring-up Stage 5). Opens the " - "cameras. The replay stops being a pure replay -- that is " - "the point of the rung, not a side effect.") - ap.add_argument("--out", default=None, help="optional MP4 of the rollout") - args = ap.parse_args() - logging.basicConfig(level=logging.INFO, format="%(message)s") - - rc = list(generate_run_configs(args.config))[0] - assert isinstance(rc, SingleSeedRunConfig) - flags = dict(rc.flags) - flags.update({"env": rc.env, "approach": rc.approach, "seed": rc.seed}) - flags.pop("log", None) - if args.scene: - flags["domino_real_scene"] = args.scene - flags["real_robot_execute"] = bool(args.execute) - # Build the arm-less RealRobot: everything downstream of the executor runs - # for real, so this exercises chunking, the gripper split and the drift - # guard without a Franka in the room (and without one powered on). - flags["real_robot_dry"] = bool(args.dry) - # By default this tool replays an EXACT plan and never looks: re-syncing - # the twin mid-replay lets the option policies see states the recorded plan - # was never chosen against, and not looking keeps the tool usable with the - # cameras down. --observe opts into exactly that mid-replay correction, - # which is the closed-loop rung. Perception follows, because RealRobot - # opens its session at CONSTRUCTION -- left at the "zed" default a run that - # never looks would still hold both cameras open. - flags["real_robot_perception"] = "zed" if args.observe else "none" - flags["real_robot_observe_at_option_boundary"] = bool(args.observe) - # ...and no human reset either: that rebuilds the episode's task from a - # live look, which would rename and re-place the very objects the recorded - # plan refers to. The task must stay the captured scene the plan was - # written against. - flags["real_robot_human_reset"] = False - # ...which is exactly the case the stale-task guard exists for, so opt in - # explicitly: these poses are the ones the plan was written against. - flags["real_robot_allow_captured_scene_task"] = True - utils.reset_config(flags) - - env = get_or_create_env(CFG.env) - assert isinstance(env, PyBulletDominoRealEnv), \ - f"replay_plan drives the real-scene env; got {CFG.env}" - # Attaches the arm under --execute and is a no-op otherwise, so the env - # stays the same object either way. - attach_real_robot(env) - opts = {o.name: o for o in get_gt_options(env.get_name())} - env_task = env.get_test_tasks()[0] - task = env_task.task - by_name = {o.name: o for o in task.init} - - with open(args.plan, encoding="utf-8") as f: - steps = _parse_plan(f.read()) - assert steps, "no plan steps parsed" - - plan = [] - for name, obj_names, params in steps: - option = opts[name] - objs = [by_name[n] for n in obj_names] - plan.append(option.ground(objs, np.array(params, dtype=np.float32))) - print("# grounded plan:") - for g in plan: - print(" ", g.simple_str()) - # Say plainly whether metal is about to move: this banner is the last - # thing a human reads before deciding where their hand is. - if not CFG.real_robot_execute: - print("# SIM ONLY -- no executor attached, no arm, nothing moves") - elif CFG.real_robot_dry: - print("# DRY ARM -- RealRobot built without an arm; chunks ship, " - "nothing moves") - else: - print("# *** THE REAL FRANKA WILL MOVE *** (in-process RealRobot)") - - policy = utils.option_plan_to_policy( - plan, abstract_function=lambda s: utils.abstract(s, env.predicates)) - monitor = utils.VideoMonitor(env.render) if args.out else None - - # Roll out through CogMan's episode loop -- the same one main.py uses -- - # driven by an override policy, exactly as the online-learning path does - # (main.py sets the override, then resets). With an override in place - # CogMan never asks the approach to solve, so the plan being replayed is - # the plan that executes. - # - # The approach is therefore never consulted for control, and only has to - # *construct*. "random_options" is a plain BaseApproach that needs nothing - # but the option set. Not "oracle": that one builds ground-truth NSRTs in - # its constructor, and this env has none -- it is planned over processes, - # so get_gt_nsrts raises NotImplementedError for pybullet_domino_real and - # the replay dies before it renders a frame. - cogman = CogMan( - create_approach("random_options", env.predicates, - get_gt_options(env.get_name()), env.types, - env.action_space, [task]), - create_perceiver(CFG.perceiver), create_execution_monitor("trivial")) - cogman.set_override_policy(policy) - cogman.set_termination_function(lambda s: False) - cogman.reset(env_task) - (_, actions), _, _ = run_episode_and_get_observations( - cogman, - env, - "test", - 0, - max_num_steps=CFG.horizon, - terminate_on_goal_reached=False, - exceptions_to_break_on={utils.OptionExecutionFailure}, - monitor=monitor) - print(f"# steps={len(actions)} goal_reached={env.goal_reached()}") - - if args.out and monitor is not None: - os.makedirs(os.path.join(CFG.video_dir, - os.path.dirname(args.out) or "."), - exist_ok=True) - utils.save_video(args.out, monitor.get_video()) - print(f"# saved {os.path.join(CFG.video_dir, args.out)}") - - -if __name__ == "__main__": - main() diff --git a/scripts/domino_debug/reproduce_domino_failures.py b/scripts/domino_debug/reproduce_domino_failures.py deleted file mode 100644 index 563f3bfb40..0000000000 --- a/scripts/domino_debug/reproduce_domino_failures.py +++ /dev/null @@ -1,146 +0,0 @@ -"""Deterministic, LLM-free reproduction of the domino oracle-samplers failures. - -Reproduces the geometric / parsing root causes behind the unsolved tasks in -``domino-agent_oracle_hybrid_sim_oracle_samplers_{demo,no_demo}`` (seeds 0-4), -*without* invoking the LLM sketcher. The test-task scenes are deterministic -given the seed, so the BiRRT motion-planning infeasibilities and the option-plan -parser bug reproduce exactly. - -IMPORTANT: run ONE seed per process. Generating tasks for several seeds inside -one interpreter advances the shared RNG and changes the scenes (e.g. seed1 would -regenerate as [4,5,5,5,4] dominoes instead of the real [4,4,5,4,4]). The bash -wrapper at the bottom of the module docstring loops correctly. - -Usage: - # motion-planning reproduction for a single seed (fresh process each): - # for s in 0 1 2 3 4; do PYTHONPATH=. python \ - # scripts/domino_debug/reproduce_domino_failures.py mp $s; done - # option-plan parser (Push) bug: - # PYTHONPATH=. python \ - # scripts/domino_debug/reproduce_domino_failures.py push 0 -""" - -import logging -import sys -from typing import List, Optional, Tuple - -import numpy as np - -from predicators import utils -from predicators.envs import get_or_create_env -from predicators.envs.base_env import BaseEnv -from predicators.ground_truth_models import get_gt_options -from predicators.structs import ParameterizedOption, State, _Option - -logging.disable(logging.CRITICAL) - -# Geometry-affecting flags copied verbatim from the experiment command line. -_ARGS = { - "env": "pybullet_domino", - "approach": "oracle", - "num_train_tasks": 1, - "num_test_tasks": 5, - "pybullet_ik_validate": False, - "skill_phase_use_motion_planning": True, - "domino_initialize_at_finished_state": False, - "domino_use_domino_blocks_as_target": True, - "domino_use_continuous_place": True, - "domino_restricted_push": True, - "domino_has_glued_dominos": False, - "pybullet_birrt_extend_num_interp": 20, - "pybullet_birrt_path_subsample_ratio": 2, -} -_GRASP_Z_OFFSET = 0.0825 # value used by the oracle Pick sampler in the runs. -_POS_GAP = 0.098 # domino chain spacing (env.py: domino_width * 1.4). -_MAX_STEPS = 80 - - -def _setup(seed: int) -> Tuple[BaseEnv, List[ParameterizedOption]]: - """Reset the config and create the env + ground-truth options.""" - args = dict(_ARGS, seed=seed) - utils.reset_config(args) - env = get_or_create_env("pybullet_domino") - options = list(get_gt_options(env.get_name())) - return env, options - - -def _run_option(env: BaseEnv, opt: _Option, - state: State) -> Tuple[Optional[bool], str]: - """Drive a grounded option to termination; return (ok, failure_msg).""" - if not opt.initiable(state): - return None, "not-initiable" - s = state - for _ in range(_MAX_STEPS): - try: - a = opt.policy(s) - except utils.OptionExecutionFailure as e: - return False, str(e) - s = env.step(a) - if opt.terminal(s): - return True, "ok" - return True, "ran-max-steps" - - -def reproduce_mp(seed: int) -> None: - """Report grasp-infeasible dominoes and probe one Place into the gap.""" - env, options = _setup(seed) - Pick = next(o for o in options if o.name == "Pick") - tasks = env.get_test_tasks() - for ti in range(len(tasks)): - env.reset("test", ti) - st = env._current_state # pylint: disable=protected-access - dominoes = sorted([o for o in st if o.type.name == "domino"], - key=lambda o: o.name) - infeasible = [] - for d in dominoes: - env.reset("test", ti) - s = env._current_state # pylint: disable=protected-access - rb = next(o for o in s if o.type.name == "robot") - dd = next(o for o in s if o.name == d.name) - opt = Pick.ground([rb, dd], - np.array([_GRASP_Z_OFFSET], dtype=np.float32)) - ok, _ = _run_option(env, opt, s) - if ok is False: - infeasible.append(d.name) - print( - f"seed{seed} task{ti} (run Task{ti+1}): {len(dominoes)} dominoes " - f"| grasp-INFEASIBLE: {infeasible if infeasible else 'none'}") - - -def reproduce_push_bug(seed: int) -> None: - """Show the parser drops a Push line that names a target domino.""" - env, options = _setup(seed) - push = next(o for o in options if o.name == "Push") - print(f"Push option signature: types={[t.name for t in push.types]}") - state = env.get_test_tasks()[0].init - objects = list(state) - cases = { - "LLM-style 'Push(robot, domino_0)'": - "Pick(robot:robot, domino_1:domino)\n" - "Push(robot:robot, domino_0:domino)\nWait(robot:robot)", - "legal 'Push(robot)'": - "Pick(robot:robot, domino_1:domino)\n" - "Push(robot:robot)\nWait(robot:robot)", - } - for label, txt in cases.items(): - plan = utils.parse_model_output_into_option_plan( - txt, objects, env.types, options, parse_continuous_params=False) - names = [op.name for op, _, _ in plan] - flag = "PUSH DROPPED!" if "Push" not in names else "ok" - print(f" {label:42s} -> {names} ({flag})") - - -def _main() -> None: - """Dispatch to the requested reproduction mode.""" - mode = sys.argv[1] if len(sys.argv) > 1 else "mp" - seed = int(sys.argv[2]) if len(sys.argv) > 2 else 0 - if mode == "mp": - reproduce_mp(seed) - elif mode == "push": - reproduce_push_bug(seed) - else: - raise SystemExit(f"unknown mode {mode!r} (expected 'mp' or 'push')") - - -if __name__ == "__main__": - _main() diff --git a/scripts/dump_continual_arm_prompts.py b/scripts/dump_continual_arm_prompts.py index 4136d7ce5c..f20ca07414 100644 --- a/scripts/dump_continual_arm_prompts.py +++ b/scripts/dump_continual_arm_prompts.py @@ -9,7 +9,7 @@ is sent to a model. python -m scripts.dump_continual_arm_prompts \ - --config predicatorv3/continual_eight_agent_noisy_sweep.yaml \ + --config empiric/benchmark.yaml \ --domain balloons --out docs/prompt-review/2026-09-18-balloons """ import argparse diff --git a/scripts/engaging/launch.py b/scripts/engaging/launch.py index a56d31ea11..64b4107d0c 100644 --- a/scripts/engaging/launch.py +++ b/scripts/engaging/launch.py @@ -5,27 +5,33 @@ job, with one array task per seed, so all experiments run concurrently on compute nodes rather than in the current terminal/login node. -Usage example: +Usage example (continual configs name a round, which suffixes the run +folders so a new launch never resumes an earlier one): - python scripts/engaging/launch.py -c predicatorv3/exp_domino.yaml + python scripts/engaging/launch.py -c empiric/benchmark.yaml --round r2 mit_normal is often saturated. To run on the much larger (but evictable) preemptable partition instead: - python scripts/engaging/launch.py -c predicatorv3/exp_domino.yaml \ + python scripts/engaging/launch.py -c empiric/benchmark.yaml --round r2 \ --partition mit_preemptable +To launch a subset of the config's envs, approaches or seeds: + + python scripts/engaging/launch.py -c empiric/benchmark.yaml \ + --round fan_fix_r1 --envs fan --approaches mb_opus --seeds 2-4 + Agent runs draw on a Claude account's usage limit. To spread a launch's runs over several accounts (token files under ~/.claude-tokens, see claude_accounts.py): - python scripts/engaging/launch.py -c predicatorv3/exp_domino.yaml \ + python scripts/engaging/launch.py -c empiric/benchmark.yaml --round r2 \ --partition mit_preemptable --accounts a,b """ import argparse import sys from pathlib import Path -from typing import Optional +from typing import List, Optional, Tuple # Add project root to sys.path so `scripts` is importable without PYTHONPATH=. # parents[0] = scripts/engaging, parents[1] = scripts, parents[2] = repo root @@ -33,7 +39,7 @@ # pylint: disable=wrong-import-position from scripts.cluster_utils import BatchSeedRunConfig, config_to_cmd_flags, \ - config_to_logfile, generate_run_configs + config_to_logfile, generate_run_configs, parse_seed_range from scripts.engaging.claude_accounts import resolve_accounts from scripts.engaging.submit_engaging_job import submit_engaging_job @@ -70,21 +76,73 @@ def _main() -> None: "a token file ~/.claude-tokens/ or the reserved name 'login' " "(the CLI's stored login). Defaults to $PREDICATORS_CLAUDE_ACCOUNTS, " "else 'login'.") + parser.add_argument( + "--round", + type=str, + default=None, + help="Name of this launch's round, appended to every experiment id " + "(-_); overrides the config's ROUND. " + "Continual configs require one.") + parser.add_argument( + "--envs", + type=str, + default=None, + help="Comma-separated env keys of the config to launch, e.g. " + "fan,boil. Defaults to every env the config does not SKIP.") + parser.add_argument( + "--approaches", + type=str, + default=None, + help="Comma-separated approach keys of the config to launch, e.g. " + "mb_opus,mf_opus. Defaults to every approach the config does not " + "SKIP.") + parser.add_argument( + "--seeds", + type=str, + default=None, + help="Seeds to launch, N or N-M (e.g. 2-4); overrides the config's " + "START_SEED and NUM_SEEDS.") args = parser.parse_args() - _launch_experiments(args.config, args.partition, args.requeue, - args.accounts) + _launch_experiments( + args.config, + args.partition, + args.requeue, + args.accounts, + round_name=args.round, + envs=_keys(args.envs), + approaches=_keys(args.approaches), + seeds=parse_seed_range(args.seeds) if args.seeds is not None else None) + + +def _keys(text: Optional[str]) -> Optional[List[str]]: + """Split a comma-separated --envs or --approaches value.""" + if text is None: + return None + return [key.strip() for key in text.split(",") if key.strip()] def _launch_experiments(config_file: str, partition: Optional[str] = None, requeue: Optional[bool] = None, - accounts: Optional[str] = None) -> None: - # Validate the account list once, before anything is submitted. + accounts: Optional[str] = None, + round_name: Optional[str] = None, + envs: Optional[List[str]] = None, + approaches: Optional[List[str]] = None, + seeds: Optional[Tuple[int, int]] = None) -> None: + # Validate the account list and resolve every run once, before + # anything is submitted. account_names = resolve_accounts(accounts) + run_configs = list( + generate_run_configs(config_file, + batch_seeds=True, + round_name=round_name, + envs=envs, + approaches=approaches, + seeds=seeds, + require_round=True)) # Loop over run configs. The experiment's index staggers the account # round-robin across sibling experiments (claude_accounts.py). - for index, cfg in enumerate( - generate_run_configs(config_file, batch_seeds=True)): + for index, cfg in enumerate(run_configs): assert isinstance(cfg, BatchSeedRunConfig) cmd_flags = config_to_cmd_flags(cfg) log_dir = "logs" diff --git a/scripts/engaging/relaunch_on_timeout.py b/scripts/engaging/relaunch_on_timeout.py deleted file mode 100644 index 0113b2e600..0000000000 --- a/scripts/engaging/relaunch_on_timeout.py +++ /dev/null @@ -1,156 +0,0 @@ -"""Watch Slurm jobs and relaunch each one's experiment once on TIMEOUT. - -Preemption self-heals via ``sbatch --requeue``, but a job that hits its -wall-clock limit is simply killed. For auto_resume experiments the fix is -to resubmit the identical command so the run continues from its latest -checkpoint. This watcher does that, once per watched job, and exits when -every watched job has reached a terminal state. - -Each watched job maps to ONE experiment (approach x env arm) inside its -launch config, and only that experiment is resubmitted. Relaunching the -whole config would also resubmit sibling arms that may still be running -(jobs in one generation drift apart through preemption requeues), and a -duplicate arm races the live one on the same auto_resume checkpoints. - -Usage (detach with nohup; poll every 10 min): - - nohup python scripts/engaging/relaunch_on_timeout.py \ - -p mit_preemptable \ - 21338737:predicatorv3/exp_bridge_v2.yaml:bridge_v2-agent_pi_al \ - 21338740:predicatorv3/exp_bridge_v2.yaml:bridge_v2-agent_pi_al_pol \ - >> logs/auto_relaunch.log 2>&1 & - -Pass the launch's --accounts list too, so a relaunched experiment keeps -the Claude accounts its seeds started on (the assignment is a function -of the config, the account list and the experiment's index). -""" - -import argparse -import subprocess -import sys -import time -from dataclasses import dataclass -from pathlib import Path -from typing import List, Optional - -# Add project root to sys.path so `scripts` is importable without PYTHONPATH=. -sys.path.insert(0, str(Path(__file__).resolve().parents[2])) - -# pylint: disable=wrong-import-position -from scripts.cluster_utils import BatchSeedRunConfig, config_to_cmd_flags, \ - config_to_logfile, generate_run_configs -from scripts.engaging.claude_accounts import resolve_accounts -from scripts.engaging.submit_engaging_job import submit_engaging_job - - -@dataclass -class _WatchedJob: - job_id: str - config_file: str - experiment_id: str - resolved: bool = False - - -def _parse_spec(spec: str) -> _WatchedJob: - parts = spec.split(":") - if len(parts) != 3: - raise ValueError(f"Expected JOBID:CONFIG:EXPERIMENT_ID, got {spec!r}") - return _WatchedJob(job_id=parts[0], - config_file=parts[1], - experiment_id=parts[2]) - - -def _in_queue(job_id: str) -> bool: - out = subprocess.run(["squeue", "-h", "-j", job_id], - capture_output=True, - text=True, - check=False) - return bool(out.stdout.strip()) - - -def _terminal_state(job_id: str) -> Optional[str]: - """The job's sacct state, or None if sacct has nothing yet.""" - out = subprocess.run(["sacct", "-j", job_id, "-X", "-n", "-o", "State%30"], - capture_output=True, - text=True, - check=False) - lines = [ln.strip() for ln in out.stdout.splitlines() if ln.strip()] - return lines[0] if lines else None - - -def _relaunch_experiment(config_file: str, - experiment_id: str, - partition: str, - accounts: Optional[str] = None) -> None: - """Resubmit only ``experiment_id`` from ``config_file``.""" - account_names = resolve_accounts(accounts) - for index, cfg in enumerate( - generate_run_configs(config_file, batch_seeds=True)): - assert isinstance(cfg, BatchSeedRunConfig) - if cfg.experiment_id != experiment_id: - continue - cmd_flags = config_to_cmd_flags(cfg) - log_prefix = config_to_logfile(cfg, suffix="") - submit_engaging_job("main.py", cfg.experiment_id, "logs", log_prefix, - cmd_flags, cfg.start_seed, cfg.num_seeds, - cfg.use_gpu, cfg.use_mujoco, partition, True, - account_names, index) - return - print( - f"WARNING: experiment {experiment_id} not found in {config_file}; " - "nothing relaunched", - flush=True) - - -def _main() -> None: - parser = argparse.ArgumentParser() - parser.add_argument("specs", - nargs="+", - help="JOBID:CONFIG:EXPERIMENT_ID triples") - parser.add_argument("-p", "--partition", default="mit_preemptable") - parser.add_argument("--poll-seconds", type=int, default=600) - parser.add_argument("--accounts", - type=str, - default=None, - help="The launch's Claude account list (see " - "claude_accounts.py); defaults to " - "$PREDICATORS_CLAUDE_ACCOUNTS, else 'login'.") - args = parser.parse_args() - # Fail on a bad account list now, not at the first TIMEOUT. - resolve_accounts(args.accounts) - jobs: List[_WatchedJob] = [_parse_spec(s) for s in args.specs] - stamp = time.strftime("%m-%d %H:%M") - print( - f"{stamp} watching {len(jobs)} jobs: " - f"{', '.join(j.job_id for j in jobs)}", - flush=True) - while not all(j.resolved for j in jobs): - time.sleep(args.poll_seconds) - for job in jobs: - if job.resolved or _in_queue(job.job_id): - continue - state = _terminal_state(job.job_id) - if state is None: - continue - stamp = time.strftime("%m-%d %H:%M") - if state.startswith("TIMEOUT"): - print( - f"{stamp} job {job.job_id} TIMEOUT -> relaunching " - f"{job.experiment_id} from {job.config_file}", - flush=True) - _relaunch_experiment(job.config_file, job.experiment_id, - args.partition, args.accounts) - else: - print( - f"{stamp} job {job.job_id} ended {state}; " - "no relaunch", - flush=True) - job.resolved = True - print( - f"{time.strftime('%m-%d %H:%M')} all watched jobs resolved; " - "exiting", - flush=True) - - -if __name__ == "__main__": - _main() diff --git a/scripts/local/generate_random_action_gifs.py b/scripts/local/generate_random_action_gifs.py index 383e2fe63d..05f0db1fe4 100644 --- a/scripts/local/generate_random_action_gifs.py +++ b/scripts/local/generate_random_action_gifs.py @@ -10,7 +10,7 @@ Options: --skip-run Skip running the experiments (just convert existing MP4s) --config, -c Config file to use - (default: mara2/random_actions_pybullet.yaml) + (default: ExoPredicator/random_actions_pybullet.yaml) --video-dir Directory where MP4s are written (default: videos) --output-dir Directory for output GIFs (default: docs/envs/assets/random_action_gifs) @@ -114,7 +114,7 @@ def main() -> None: parser.add_argument( "-c", "--config", - default="mara2/random_actions_pybullet.yaml", + default="ExoPredicator/random_actions_pybullet.yaml", help="Config YAML file (relative to scripts/configs/).", ) parser.add_argument( diff --git a/scripts/local/render_init_state_gifs.py b/scripts/local/render_init_state_gifs.py index ffb230549c..0d71d54148 100644 --- a/scripts/local/render_init_state_gifs.py +++ b/scripts/local/render_init_state_gifs.py @@ -6,9 +6,11 @@ experiments use: - the five benchmark domains take their entry in - scripts/configs/predicatorv3/envs/continual.yaml, -- the other recent domains take their entry in envs/all.yaml, -- the older domains take their entry in random_actions_pybullet.yaml. + scripts/configs/empiric/envs.yaml, +- the older domains take their entry in + scripts/configs/ExoPredicator/random_actions_pybullet.yaml, +- the rest (busyboard, crane, icerink, launcher, magnets) render with their + defaults; no experiment changed their task distributions. Observation noise flags are irrelevant here (states are rendered, not observed). Run on a compute node, one env per call: @@ -27,32 +29,30 @@ from predicators import utils from predicators.envs import create_new_env -_CONFIG_DIR = "scripts/configs/predicatorv3" +_CONFIG_DIR = "scripts/configs" -# env name -> (menu file, menu key) +# env name -> (menu file, menu key) for the benchmark domains. _MENUS: Dict[str, Tuple[str, str]] = { - "pybullet_balloons": ("envs/continual.yaml", "balloons"), - "pybullet_bridge": ("envs/continual.yaml", "bridge"), - "pybullet_boil": ("envs/continual.yaml", "boil"), - "pybullet_fan": ("envs/continual.yaml", "fan"), - "pybullet_domino": ("envs/continual.yaml", "domino_high_friction_turn"), - "pybullet_busyboard": ("envs/all.yaml", "busyboard"), - "pybullet_crane": ("envs/all.yaml", "crane"), - "pybullet_icerink": ("envs/all.yaml", "icerink"), - "pybullet_launcher": ("envs/all.yaml", "launcher"), - "pybullet_magnets": ("envs/all.yaml", "magnets"), + "pybullet_balloons": ("empiric/envs.yaml", "balloons"), + "pybullet_bridge": ("empiric/envs.yaml", "bridge"), + "pybullet_boil": ("empiric/envs.yaml", "boil"), + "pybullet_fan": ("empiric/envs.yaml", "fan"), + "pybullet_domino": ("empiric/envs.yaml", "domino_high_friction_turn"), } -_LEGACY_MENU = "random_actions_pybullet.yaml" +_LEGACY_MENU = "ExoPredicator/random_actions_pybullet.yaml" def _menu_flags(env_name: str) -> Dict[str, Any]: - """Return the FLAGS of the menu entry that defines this env's tasks.""" + """Return the FLAGS of the menu entry that defines this env's tasks, or + none for an env that no menu lists.""" menu_file, key = _MENUS.get(env_name, (_LEGACY_MENU, "")) with open(os.path.join(_CONFIG_DIR, menu_file), encoding="utf-8") as f: config = yaml.safe_load(f) envs = config["ENVS"] if not key: - key = next(k for k, v in envs.items() if v["NAME"] == env_name) + key = next((k for k, v in envs.items() if v["NAME"] == env_name), "") + if not key: + return {} flags = dict(envs[key].get("FLAGS", {})) return {k: v for k, v in flags.items() if not k.startswith("continual_")} diff --git a/scripts/plotting/monitor_benchmark_arms.py b/scripts/plotting/monitor_benchmark_arms.py index e0a0ab0f8e..b6dbce8411 100644 --- a/scripts/plotting/monitor_benchmark_arms.py +++ b/scripts/plotting/monitor_benchmark_arms.py @@ -273,7 +273,8 @@ def report_text(original: str, rows: List[Row], plot: Dict[str, Any], "See the [illustrated task description]" "(../amps/fan-exposed-transfer.md) and " "[launch configuration]" - "(../../scripts/configs/predicatorv3/" + "(https://github.com/BasisResearch/predicators/blob/" + "iclr-empiric-submission/scripts/configs/predicatorv3/" "continual_fan_transfer_pilot_r1.yaml).\n" "This column is excluded from the paper figure and its " "data selection.\n\n") diff --git a/tests/agent_sdk/test_prompt_goldens.py b/tests/agent_sdk/test_prompt_goldens.py index 1823ce87b9..a04b765e4a 100644 --- a/tests/agent_sdk/test_prompt_goldens.py +++ b/tests/agent_sdk/test_prompt_goldens.py @@ -549,7 +549,8 @@ def test_golden_continual_system_ablation(arm): if arm == "no_uncertainty": flags["continual_obs_noise_declared"] = False # As the experiments run them: every arm but No uncertainty carries - # the joint belief (continual_common.yaml, approaches/continual.yaml). + # the joint belief (scripts/configs/empiric/common.yaml and + # approaches.yaml). flags["belief_joint_draws"] = 0 if arm == "no_uncertainty" else 16 utils.reset_config(flags) tools = ["run_python"] + list(CONTINUAL_TOOL_NAMES) diff --git a/tests/approaches/test_agent_continual_direct_scene_files.py b/tests/approaches/test_agent_continual_direct_scene_files.py index a806cd3951..ed0ca473d2 100644 --- a/tests/approaches/test_agent_continual_direct_scene_files.py +++ b/tests/approaches/test_agent_continual_direct_scene_files.py @@ -20,11 +20,12 @@ # pylint: disable=protected-access -CONFIG = "predicatorv3/continual_direct_scene_files_benchmark_r1.yaml" +CONFIG = "empiric/benchmark.yaml" +ARM = "mf_scene_package_opus" def _arm_flags(env_name: str) -> Dict[str, Any]: - cfg = next(c for c in generate_run_configs(CONFIG, False) + cfg = next(c for c in generate_run_configs(CONFIG, False, approaches=[ARM]) if c.env == env_name) # The launcher pins machine-specific output paths; tests keep their own. flags = { @@ -50,10 +51,10 @@ def _make(tmp_path: Any, **overrides: Any) -> Any: def test_config_is_the_direct_agent_with_the_scene_files() -> None: - """Five settings, three seeds, the MF arm with the package flag and none of + """Five settings, five seeds, the MF arm with the package flag and none of the model arm's flags.""" - runs = list(generate_run_configs(CONFIG, False)) - assert len(runs) == 15 + runs = list(generate_run_configs(CONFIG, False, approaches=[ARM])) + assert len(runs) == 25 for run in runs: assert run.approach == "agent_continual_model_free" assert run.flags["continual_provide_scene_package"] is True diff --git a/tests/approaches/test_agent_continual_real_to_sim_approach.py b/tests/approaches/test_agent_continual_real_to_sim_approach.py index 541db18424..122ca9946e 100644 --- a/tests/approaches/test_agent_continual_real_to_sim_approach.py +++ b/tests/approaches/test_agent_continual_real_to_sim_approach.py @@ -20,7 +20,27 @@ # pylint: disable=protected-access -CONFIG = "predicatorv3/continual_real_to_sim_benchmark_r1.yaml" +# The agentic real-to-sim arm (no domain twin: the generic PyBulletEnv, a +# domain-agnostic SceneBase, the scene manifest and the asset files; the +# harness fits nothing and runs no uncertainty machinery) and the EMPIRIC +# from-assets arm it derives from. Neither is a benchmark arm; each is the +# benchmark's EMPIRIC run with these flags. +REAL_TO_SIM_FLAGS = { + "agent_sim_learn_declared_params_only": True, + "continual_uncertainty_decisions": False, + "agent_sim_learn_param_uncertainty": False, + "agent_plan_validation_rule_param_margin": False, + "agent_plan_validation_physics_margin": False, + "agent_explorer_info_seeking": False, + "agent_explorer_info_seeking_adaptive": False, + "agent_explorer_info_seeking_noise_aware": False, + "code_sim_learning_interval_belief": False, + "code_sim_learning_carry_posterior": False, +} +FROM_ASSETS_FLAGS = { + "agent_sim_learn_declared_params_only": False, + "continual_uncertainty_decisions": True, +} # The agent's simulator for the Boil training scene, written the way the # prompt describes: a SceneBase subclass whose initialize_pybullet loads # every body of the manifest under its observed name. @@ -115,16 +135,23 @@ def connect(*args: Any, **kwargs: Any) -> int: p.disconnect(client) -def _arm_flags() -> Dict[str, Any]: - cfg = next(c for c in generate_run_configs(CONFIG, False) - if c.env == "pybullet_boil") - # The launcher pins machine-specific output paths; tests keep their own. - flags = { +def _benchmark_flags(env_name: str) -> Dict[str, Any]: + """The benchmark EMPIRIC run's flags on ``env_name``, without the machine- + specific output paths (tests keep their own).""" + cfg = next(c for c in generate_run_configs( + "empiric/benchmark.yaml", False, approaches=["mb_opus"]) + if c.env == env_name) + return { k: v for k, v in cfg.flags.items() if k not in ("log", "continual_runs_dir") } - flags.update(approach=cfg.approach, - env=cfg.env, + + +def _arm_flags() -> Dict[str, Any]: + flags = _benchmark_flags("pybullet_boil") + flags.update(REAL_TO_SIM_FLAGS, + approach="agent_continual_real_to_sim", + env="pybullet_boil", continual_render=False, continual_make_video=False) return flags @@ -134,17 +161,10 @@ def _make(tmp_path: Any, from_assets: bool = False) -> Any: flags = _arm_flags() name = "agent_continual_real_to_sim" if from_assets: - cfg = next(c for c in generate_run_configs( - "predicatorv3/continual_from_assets_pilot_r1.yaml", False) - if c.env == "pybullet_bridge") - # The launcher pins machine-specific output paths; tests keep their own. - flags = { - k: v - for k, v in cfg.flags.items() - if k not in ("log", "continual_runs_dir") - } name = "agent_continual_from_assets" - flags.update(approach=name, + flags = _benchmark_flags("pybullet_bridge") + flags.update(FROM_ASSETS_FLAGS, + approach=name, env="pybullet_boil", continual_render=False, continual_make_video=False) @@ -158,17 +178,6 @@ def _make(tmp_path: Any, from_assets: bool = False) -> Any: return env, approach -def test_config_is_the_benchmark_arm() -> None: - """Five settings, three seeds, no fitting and no uncertainty flags.""" - runs = list(generate_run_configs(CONFIG, False)) - assert len(runs) == 15 - for run in runs: - assert run.approach == "agent_continual_real_to_sim" - assert run.flags["agent_sim_learn_declared_params_only"] is True - assert run.flags["continual_uncertainty_decisions"] is False - assert run.flags["continual_require_model_on_test"] is True - - def test_arm_refuses_uncertainty_machinery(tmp_path: Any) -> None: """A launcher that leaves an uncertainty switch on fails at construction.""" @@ -346,30 +355,3 @@ def query(message: str, *_args: Any, **_kwargs: Any) -> Any: fit = approach._last_fit_result assert fit is not None and fit.names == ["heat_rate"] assert fit.samples[0, 0] == pytest.approx(0.01) - - -def test_from_assets_pilot_retains_empiric_capabilities() -> None: - """Two seeds per domain, fitting and EMPIRIC's joint belief, no - preflight.""" - runs = list( - generate_run_configs( - "predicatorv3/continual_from_assets_pilot_r1.yaml", False)) - assert len(runs) == 10 - assert {r.env - for r in runs} == { - "pybullet_fan", "pybullet_bridge", "pybullet_domino", - "pybullet_balloons", "pybullet_boil" - } - for run in runs: - assert run.approach == "agent_continual_from_assets" - assert run.flags["agent_sim_learn_declared_params_only"] is False - assert run.flags["continual_uncertainty_decisions"] is True - assert run.flags["code_sim_learning_interval_belief"] is True - # The joint belief, with the prior at its declared centre. - assert run.flags["belief_joint_draws"] == 16 - assert run.flags["code_sim_learning_carry_posterior"] is False - assert run.flags["continual_skill_preflight"] is False - if run.env == "pybullet_fan": - assert run.flags["fan_ramp_transfer"] is True - assert run.flags["fan_ramp_rise"] == 0.003 - assert run.flags["fan_ramp_landing_extension"] == 0.10 diff --git a/tests/approaches/test_agent_continual_scene_package.py b/tests/approaches/test_agent_continual_scene_package.py index 05454418d4..803071c706 100644 --- a/tests/approaches/test_agent_continual_scene_package.py +++ b/tests/approaches/test_agent_continual_scene_package.py @@ -20,18 +20,27 @@ # pylint: disable=protected-access -CONFIG = "predicatorv3/continual_empiric_scene_package_benchmark_r1.yaml" +# EMPIRIC with what an agentic real-to-sim agent receives: the engine +# wrapper, the scene manifest and the URDF and mesh files, plus the twin's +# own core module where the domain declares one. Not a benchmark arm; it +# is the benchmark's EMPIRIC arm with these two flags. +SCENE_PACKAGE_FLAGS = { + "agent_sim_provide_base_sim_source": True, + "continual_provide_scene_package": True, +} def _arm_flags(env_name: str) -> Dict[str, Any]: - cfg = next(c for c in generate_run_configs(CONFIG, False) + cfg = next(c for c in generate_run_configs( + "empiric/benchmark.yaml", False, approaches=["mb_opus"]) if c.env == env_name) # The launcher pins machine-specific output paths; tests keep their own. flags = { k: v for k, v in cfg.flags.items() if k not in ("log", "continual_runs_dir") } - flags.update(approach=cfg.approach, + flags.update(SCENE_PACKAGE_FLAGS, + approach=cfg.approach, env=cfg.env, continual_render=False, continual_make_video=False) @@ -55,18 +64,6 @@ def _refs(sandbox: Path) -> List[str]: for q in (sandbox / "reference").rglob("*") if q.is_file()) -def test_config_is_empiric_with_the_scene_package() -> None: - """Five settings, three seeds, the MB arm with both reference flags.""" - runs = list(generate_run_configs(CONFIG, False)) - assert len(runs) == 15 - for run in runs: - assert run.approach == "agent_continual" - assert run.flags["continual_provide_scene_package"] is True - assert run.flags["agent_sim_provide_base_sim_source"] is True - assert run.flags["continual_require_model_on_test"] is True - assert "agent_sim_learn_declared_params_only" not in run.flags - - def test_fan_lists_the_twin_core_and_the_package(tmp_path: Any) -> None: """A domain with a declared core module gets it beside the package.""" _, approach = _make(tmp_path, "pybullet_fan") diff --git a/tests/approaches/test_continual_comparison_approach.py b/tests/approaches/test_continual_comparison_approach.py index e67665d07a..3db04992cb 100644 --- a/tests/approaches/test_continual_comparison_approach.py +++ b/tests/approaches/test_continual_comparison_approach.py @@ -1,4 +1,5 @@ """Continual comparison contracts exercised through real play tools.""" +import dataclasses import os import re import shlex @@ -20,15 +21,31 @@ from predicators.run.level_players import create_level_player from predicators.settings import CFG from predicators.structs import Dataset -from scripts.cluster_utils import config_to_cmd_flags, generate_run_configs +from scripts.cluster_utils import SingleSeedRunConfig, config_to_cmd_flags, \ + generate_run_configs from tests.approaches.test_agent_continual_approach import _call, _config, \ _result -# The benchmark sweep: eight arms on the five benchmark settings, three -# seeds each (Sept 18, 2026). -CONFIG = "predicatorv3/continual_eight_agent_noisy_sweep.yaml" -ARM_COUNT = 8 -SEEDS = {0, 1, 2} +# The benchmark: seven arms (six approach classes; the scene-package arm +# shares the model-free class) on the five settings, five seeds each. +CONFIG = "empiric/benchmark.yaml" +ARM_COUNT = 7 +CLASS_COUNT = 6 +SEEDS = set(range(5)) +# Comparison arms outside the benchmark. Their menu entries set only the +# model, like the standalone arm's, so they reuse its run config. +UNBENCHMARKED = ("agent_continual_scene_only", "agent_continual_zero_shot") + + +def _arm_config(domain: str, approach: str) -> SingleSeedRunConfig: + """The benchmark run config of ``approach`` on ``domain``.""" + if approach in UNBENCHMARKED: + base = _arm_config(domain, "agent_continual_program_world_model") + return dataclasses.replace(base, approach=approach) + cfg = next(c for c in generate_run_configs(CONFIG, False) + if c.env == f"pybullet_{domain}" and c.approach == approach) + assert isinstance(cfg, SingleSeedRunConfig) + return cfg @pytest.fixture(autouse=True) @@ -58,10 +75,10 @@ def connect(*args: Any, **kwargs: Any) -> int: def test_comparison_config_matches_existing_domains(monkeypatch: Any) -> None: - """Eight arms share each domain's settings and run paired seeds.""" + """The seven arms share each domain's settings and run paired seeds.""" new = list(generate_run_configs(CONFIG, False)) approaches = {c.approach for c in new} - assert len(approaches) == ARM_COUNT + assert len(approaches) == CLASS_COUNT assert len(new) == ARM_COUNT * 5 * len(SEEDS) seeds: Dict[Tuple[str, str], Set[int]] = {} for cfg in new: @@ -123,8 +140,7 @@ def test_no_uncertainty_rejects_smoothing(flag: str) -> None: def test_ablation_play_tools(tmp_path: Any, monkeypatch: Any, arm: str, domain: str) -> None: """No-uncertainty uses raw observations; tools enforce arm restrictions.""" - cfg = next(c for c in generate_run_configs(CONFIG, False) - if c.env == f"pybullet_{domain}" and c.approach.endswith(arm)) + cfg = _arm_config(domain, f"agent_continual_{arm}") _config( tmp_path, **{ **{k: v @@ -286,8 +302,7 @@ def _sandbox_listing(ctx: Any) -> str: def _configure_domain_comparison(tmp_path: Any, domain: str, approach: str) -> str: """Use the benchmark's real domain settings with local test outputs.""" - cfg = next(c for c in generate_run_configs(CONFIG, False) - if c.env == f"pybullet_{domain}" and c.approach == approach) + cfg = _arm_config(domain, approach) _config( tmp_path, **{ **{k: v diff --git a/tests/approaches/test_oracle_process_planning_boil.py b/tests/approaches/test_oracle_process_planning_boil.py index 03d808874b..f407e64db2 100644 --- a/tests/approaches/test_oracle_process_planning_boil.py +++ b/tests/approaches/test_oracle_process_planning_boil.py @@ -1,7 +1,7 @@ """End-to-end test: oracle_process_planning solves a boil task. -Mirrors the config from ``predicatorv3/oracle.yaml`` + -``predicatorv3/envs/all.yaml`` + ``predicatorv3/common.yaml`` so that a +Mirrors the retired phased config (``predicatorv3/oracle.yaml`` + +``envs/all.yaml`` + ``common.yaml`` at tag iclr-empiric-submission) so that a regression in either the approach (process planning + bilevel refinement) or the boil env's skill execution would surface here. @@ -30,7 +30,7 @@ def _oracle_boil_config() -> dict: - """Flags from predicatorv3/{common,envs/all,oracle}.yaml flattened. + """Flags from the retired predicatorv3/{common,envs/all,oracle}.yaml. Kept minimal: 1 train task and 1 test task, no online learning cycles (oracle approach is not learning-based), no LLM (oracle diff --git a/tests/approaches/test_oracle_process_planning_bridge.py b/tests/approaches/test_oracle_process_planning_bridge.py index bd8480aace..1737a21914 100644 --- a/tests/approaches/test_oracle_process_planning_bridge.py +++ b/tests/approaches/test_oracle_process_planning_bridge.py @@ -1,10 +1,10 @@ """End-to-end test: oracle_process_planning solves a bridge task. -Mirrors the config from ``predicatorv3/oracle.yaml`` + -``predicatorv3/envs/all.yaml`` (bridge entry) + ``predicatorv3/common.yaml`` -so that a regression in the approach (process planning + bilevel -refinement), the bridge env's glue/weld machinery, or the skill -factories would surface here. +Mirrors the retired phased config (``predicatorv3/oracle.yaml`` + +``envs/all.yaml`` (bridge entry) + ``common.yaml`` at tag +iclr-empiric-submission) so that a regression in the approach (process +planning + bilevel refinement), the bridge env's glue/weld machinery, or +the skill factories would surface here. Runs the smallest viable config (1 train task, 1 test task; 5 blocks, 2 glue joints) and asserts: @@ -34,7 +34,7 @@ def _oracle_bridge_config(n_spans: int = 3, pool: int = 0) -> dict: - """Flags from predicatorv3/{common,envs/all,oracle}.yaml flattened.""" + """Flags from the retired predicatorv3/{common,envs/all,oracle}.yaml.""" return { # --- env: bridge from envs/all.yaml --- "env": diff --git a/tests/code_sim_learning/test_bridge_transfer_oracle.py b/tests/code_sim_learning/test_bridge_transfer_oracle.py index 60d94b9230..7d6fa88f36 100644 --- a/tests/code_sim_learning/test_bridge_transfer_oracle.py +++ b/tests/code_sim_learning/test_bridge_transfer_oracle.py @@ -18,57 +18,17 @@ from tests.code_sim_learning.test_continual_oracle import _load -def test_oracle_repair_pilot_is_six_new_preflight_off_runs() -> None: - """The pilot cannot launch other domains or resume historical seeds.""" - runs = list( - generate_run_configs( - "predicatorv3/continual_oracle_validation_r2.yaml", False)) - assert len(runs) == 6 - assert {r.env for r in runs} == {"pybullet_bridge", "pybullet_domino"} - seeds = set() - for run in runs: - assert isinstance(run, SingleSeedRunConfig) - seeds.add(run.seed) - assert run.approach == "agent_continual_oracle_dynamics" - assert run.flags["continual_skill_preflight"] is False - assert "benchmark_r2" in run.experiment_id - assert seeds == {0, 1, 2} - - -def test_empiric_r2_is_ten_new_shadow_runs() -> None: - """The prospective round has two fresh seeds, no mandatory gate.""" - runs = list( - generate_run_configs( - "predicatorv3/continual_empiric_benchmark_r2.yaml", False)) - assert len(runs) == 10 - assert len({r.env for r in runs}) == 5 - for run in runs: - assert isinstance(run, SingleSeedRunConfig) - assert run.seed in (3, 4) - assert run.approach == "agent_continual" - assert run.flags["continual_skill_preflight"] is False - assert run.flags["continual_validation_audit"] is True - assert run.flags["continual_validation_audit_seconds"] == 600. - assert run.experiment_id.endswith("-mb_opus_benchmark_r2") - - def test_transfer_comparisons_preserve_arm_contracts(monkeypatch: Any) -> None: - """The current eight-arm benchmark uses the four-span task consistently. - - The old pilot launchers were removed when these settings became the - benchmark defaults; test the maintained menu-based launcher instead. - """ - transfer = [ - c for c in generate_run_configs( - "predicatorv3/continual_eight_agent_noisy_sweep.yaml", False) - if c.env == "pybullet_bridge" - ] - assert len(transfer) == 24 - assert len({cfg.approach for cfg in transfer}) == 8 + """Every arm of the benchmark plays the same four-span Bridge task, and the + launch command parses back to it.""" + transfer = list( + generate_run_configs("empiric/benchmark.yaml", False, envs=["bridge"])) + assert len(transfer) == 7 * 5 + assert len({cfg.approach for cfg in transfer}) == 6 for cfg in transfer: assert isinstance(cfg, SingleSeedRunConfig) assert cfg.env == "pybullet_bridge" - assert cfg.seed in (0, 1, 2) + assert cfg.seed in range(5) assert cfg.flags["bridge_train_span_blocks"] == 3 assert cfg.flags["bridge_test_span_blocks"] == 4 assert cfg.flags["continual_steps_per_level"] == 10000 diff --git a/tests/code_sim_learning/test_continual_oracle.py b/tests/code_sim_learning/test_continual_oracle.py index a8f58f4164..4137d9db37 100644 --- a/tests/code_sim_learning/test_continual_oracle.py +++ b/tests/code_sim_learning/test_continual_oracle.py @@ -353,10 +353,12 @@ def test_scene_only_physical_calibration(domain: str) -> None: AgentContinualSceneOnlyApproach # pylint: disable=import-outside-toplevel from scripts.cluster_utils import \ generate_run_configs # pylint: disable=import-outside-toplevel + + # The benchmark's env and common flags; the scene-only class needs no + # arm flags of its own. config = next(c for c in generate_run_configs( - "predicatorv3/continual_eight_agent_noisy_sweep.yaml", False) - if c.env == f"pybullet_{domain}" - and c.approach == "agent_continual_scene_only") + "empiric/benchmark.yaml", False, approaches=["mb_opus"]) + if c.env == f"pybullet_{domain}") utils.reset_config({ **{k: v for k, v in config.flags.items() if k != "log"}, "env": config.env, diff --git a/tests/envs/test_pybullet_fan_transfer.py b/tests/envs/test_pybullet_fan_transfer.py index 17826b2723..235cb58600 100644 --- a/tests/envs/test_pybullet_fan_transfer.py +++ b/tests/envs/test_pybullet_fan_transfer.py @@ -17,13 +17,14 @@ @pytest.mark.parametrize("seed", [0, 1]) @pytest.mark.parametrize("arm", [0, 1]) -@pytest.mark.parametrize("variant", ["transfer", "inertial", "ramp"]) -def test_launch_config_constructs_both_levels(seed, arm, variant): - """Resolve the actual launcher, including list overrides, before reset.""" +def test_launch_config_constructs_both_levels(seed, arm): + """Resolve the benchmark's Fan setting, including list flags, before + reset.""" configs = list( - generate_run_configs( - f"predicatorv3/continual_fan_{variant}_pilot_r1.yaml", - batch_seeds=True)) + generate_run_configs("empiric/benchmark.yaml", + batch_seeds=True, + envs=["fan"], + approaches=["mb_opus", "mf_opus"])) assert len(configs) == 2 config = configs[arm] flags = { diff --git a/tests/test_benchmark_plots.py b/tests/test_benchmark_plots.py index af53d8f498..fab8f18200 100644 --- a/tests/test_benchmark_plots.py +++ b/tests/test_benchmark_plots.py @@ -54,22 +54,18 @@ def test_current_fan_is_only_ramp() -> None: assert plot.DISPLAY_TITLE["Fan (ramp transfer)"] == "Fan" -def test_default_fan_config_preserves_archived_variants() -> None: - """The menu promotes the reviewed layout without altering old pilots.""" +def test_benchmark_fan_is_the_ramp_transfer() -> None: + """The benchmark's Fan setting is the reviewed ramp layout the paper + figures plot.""" from scripts.cluster_utils import \ parse_configs # pylint: disable=import-outside-toplevel - config = next(parse_configs("predicatorv3/envs/continual.yaml")) + config = next(parse_configs("empiric/envs.yaml")) fan = config["ENVS"]["fan"]["FLAGS"] assert fan["fan_ramp_transfer"] assert fan["fan_inertial_transfer"] assert fan["fan_exposed_transfer"] assert fan["fan_ramp_rise"] == 0.003 assert fan["fan_ramp_landing_extension"] == 0.10 - for variant in ("transfer", "inertial", "ramp"): - pilot = next( - parse_configs( - f"predicatorv3/continual_fan_{variant}_pilot_r1.yaml")) - assert pilot["ENVS"][f"fan_{variant}"]["EXTENDS"] == "fan_maze" def test_oracle_r2_layout_and_scope() -> None: diff --git a/tests/test_cluster_utils_configs.py b/tests/test_cluster_utils_configs.py index 7e26c28b92..3b93cc59d6 100644 --- a/tests/test_cluster_utils_configs.py +++ b/tests/test_cluster_utils_configs.py @@ -1,29 +1,132 @@ -"""The experiment config loader: includes, parked menu entries and EXTENDS.""" +"""The experiment config loader and the EMPIRIC benchmark config: includes, +parked entries, rounds and launch subsets.""" import os -from typing import Any, Dict +from typing import Any, Dict, List import pytest import yaml -from scripts.cluster_utils import _resolve_extends, generate_run_configs +from scripts.cluster_utils import SingleSeedRunConfig, generate_run_configs, \ + parse_seed_range +BENCHMARK = "empiric/benchmark.yaml" -def test_oracle_defaults_preserve_repaired_pilot_policy() -> None: - """Both Oracle presets retain the validated nonblocking policy.""" - with open("scripts/configs/predicatorv3/approaches/continual.yaml", - encoding="utf-8") as stream: - approaches = yaml.safe_load(stream)["APPROACHES"] - for name in ("oracle_dynamics_opus", "oracle_dynamics_sonnet"): - flags = approaches[name]["FLAGS"] - assert flags["continual_skill_preflight"] is False - assert flags["continual_validation_audit"] is False - runs = list( - generate_run_configs( - "predicatorv3/continual_oracle_validation_r2.yaml", False)) - assert len(runs) == 6 + +def _benchmark_runs(**kwargs: Any) -> List[SingleSeedRunConfig]: + runs = [] + for config in generate_run_configs(BENCHMARK, False, **kwargs): + assert isinstance(config, SingleSeedRunConfig) + runs.append(config) + return runs + + +def test_benchmark_is_seven_arms_on_five_settings() -> None: + """The default config runs the seven arms on the five settings, five seeds + each, with each arm's capability contract.""" + runs = _benchmark_runs(round_name="r9") + assert len(runs) == 7 * 5 * 5 + assert {r.seed for r in runs} == set(range(5)) + assert len({(r.experiment_id, r.seed) for r in runs}) == len(runs) + assert {r.env + for r in runs} == { + "pybullet_balloons", "pybullet_bridge", "pybullet_boil", + "pybullet_fan", "pybullet_domino" + } + assert {r.experiment_id.split("-", 1)[1] + for r in runs} == { + f"{arm}_opus_r9" + for arm in ("mb", "mf", "mf_scene_package", "standalone", + "oracle_dynamics", "no_fitting", "no_uncertainty") + } + # The scene-package arm shares the model-free class, not its flags. + assert len({r.approach for r in runs}) == 6 for run in runs: - assert run.flags["continual_skill_preflight"] is False - assert run.flags["continual_validation_audit"] is False + flags = run.flags + assert flags["agent_sdk_model_name"] == "claude-opus-5" + assert flags["experiment_protocol"] == "continual" + assert flags["partially_observable"] + assert not flags.get("continual_skill_preflight", False) + assert not flags.get("continual_validation_audit", False) + assert flags["continual_wall_clock_hours"] == 48.0 + assert "auto_resume" in run.args + # The principled joint belief is the default; only No uncertainty + # turns it off. + assert not flags["code_sim_learning_carry_posterior"] + assert flags["belief_joint_draws"] == ( + 0 if run.approach == "agent_continual_no_uncertainty" else 16) + if run.approach == "agent_continual_model_free": + assert not flags["agent_planner_use_simulator"] + assert not flags["continual_uncertainty_decisions"] + assert bool(flags.get("continual_provide_scene_package")) == ( + "mf_scene_package" in run.experiment_id) + elif run.approach == "agent_continual_no_fitting": + assert flags["agent_sim_learn_declared_params_only"] + assert flags["continual_uncertainty_decisions"] + assert flags["continual_require_model_on_test"] + elif run.approach == "agent_continual_no_uncertainty": + for name in ("continual_obs_noise_declared", + "continual_uncertainty_decisions", + "agent_sim_learn_param_uncertainty", + "code_sim_learning_interval_belief", + "code_sim_learning_rollout_noise_filter", + "continual_belief_frame"): + assert not flags[name] + assert flags["continual_require_model_on_test"] + elif run.approach == "agent_continual_oracle_dynamics": + # The repaired Oracle policy: no automatic execution gate or + # shadow audit. + assert flags["continual_skill_preflight"] is False + assert flags["continual_validation_audit"] is False + if run.env == "pybullet_bridge": + assert flags["bridge_train_span_blocks"] == 3 + assert flags["bridge_test_span_blocks"] == 4 + assert flags["continual_steps_per_level"] == 10000 + else: + assert flags["continual_steps_per_level"] == 5000 + if run.env == "pybullet_balloons": + assert flags["num_train_tasks"] == 2 + assert flags["balloons_goal_dwell_steps"] == 25 + if run.env == "pybullet_fan": + assert flags["fan_ramp_transfer"] + assert flags["fan_ramp_rise"] == 0.003 + if run.env == "pybullet_boil": + assert flags["boil_num_jugs_test"] == [2] + + +def test_benchmark_subsets_and_rounds() -> None: + """--envs, --approaches and --seeds pick a subset; --round names every + experiment id.""" + runs = _benchmark_runs(round_name="fan_fix_r1", + envs=["fan"], + approaches=["mb_opus", "mf_opus"], + seeds=parse_seed_range("2-4")) + assert sorted({r.experiment_id + for r in runs + }) == ["fan-mb_opus_fan_fix_r1", "fan-mf_opus_fan_fix_r1"] + assert sorted(r.seed for r in runs) == [2, 2, 3, 3, 4, 4] + with pytest.raises(ValueError, match="unknown envs"): + _benchmark_runs(round_name="r1", envs=["fan_maze"]) + with pytest.raises(ValueError, match="unknown approaches"): + _benchmark_runs(round_name="r1", approaches=["mb_sonnet"]) + with pytest.raises(ValueError, match="letters, digits"): + _benchmark_runs(round_name="r 1") + + +def test_continual_launch_requires_a_round() -> None: + """A launch that could resume an earlier one's run folders fails until it + names a round.""" + with pytest.raises(ValueError, match="name a round"): + _benchmark_runs(require_round=True) + assert _benchmark_runs(round_name="r2", require_round=True) + + +def test_parse_seed_range() -> None: + """Seed ranges are N or N-M, inclusive.""" + assert parse_seed_range("3") == (3, 1) + assert parse_seed_range("0-4") == (0, 5) + for bad in ("4-2", "a", "1-", "-3"): + with pytest.raises(ValueError): + parse_seed_range(bad) def _write(configs_dir: str, name: str, content: Dict[str, Any]) -> None: @@ -31,10 +134,11 @@ def _write(configs_dir: str, name: str, content: Dict[str, Any]) -> None: yaml.safe_dump(content, f) -def test_extends_gives_a_menu_entry_a_new_id(monkeypatch: Any, - tmp_path: Any) -> None: - """A launcher un-parks a menu env and derives round-specific arms from - parked menu arms with EXTENDS; ids, names and merged flags follow.""" +def test_includes_skip_round_and_precedence(monkeypatch: Any, + tmp_path: Any) -> None: + """A launcher includes menus and a common file; parked entries stay out + unless selected, the ROUND key names the ids, and env flags override arm + flags, which override the common ones.""" configs_dir = tmp_path / "configs" os.makedirs(configs_dir / "menu") _write( @@ -52,7 +156,6 @@ def test_extends_gives_a_menu_entry_a_new_id(monkeypatch: Any, "ENVS": { "balloons": { "NAME": "pybullet_balloons", - "SKIP": True, "FLAGS": { "over": "env" } @@ -68,7 +171,6 @@ def test_extends_gives_a_menu_entry_a_new_id(monkeypatch: Any, "APPROACHES": { "mb": { "NAME": "agent_continual", - "SKIP": True, "FLAGS": { "gate": True, "over": "arm" @@ -79,21 +181,12 @@ def test_extends_gives_a_menu_entry_a_new_id(monkeypatch: Any, _write( str(configs_dir), "launch.yaml", { "includes": ["common.yaml", "menu/envs.yaml", "menu/arms.yaml"], - "ENVS": { - "balloons": { - "SKIP": False - } - }, + "ROUND": "r2", "APPROACHES": { - "mb_r2": { - "EXTENDS": "mb", + "mb": { "FLAGS": { "round": 2 } - }, - "mb_parked": { - "EXTENDS": "mb", - "SKIP": True } } }) @@ -107,34 +200,12 @@ def test_extends_gives_a_menu_entry_a_new_id(monkeypatch: Any, assert run.flags["gate"] is True assert run.flags["round"] == 2 assert run.flags["shared"] == 1 - # Env flags override arm flags, which override the common ones. assert run.flags["over"] == "env" - - -def test_extends_rejects_unknown_and_chained_bases() -> None: - """EXTENDS names a menu entry of the same section, one level deep.""" - with pytest.raises(ValueError, match="unknown"): - _resolve_extends({"a": {"EXTENDS": "missing"}}) - with pytest.raises(ValueError, match="itself EXTENDS"): - _resolve_extends({ - "base": { - "NAME": "x" - }, - "mid": { - "EXTENDS": "base" - }, - "top": { - "EXTENDS": "mid" - } - }) - resolved = _resolve_extends({ - "base": { - "NAME": "x", - "SKIP": True - }, - "top": { - "EXTENDS": "base" - } - }) - assert resolved["base"]["SKIP"] is True - assert resolved["top"] == {"NAME": "x", "SKIP": False} + # A command-line round overrides the config's; naming a parked entry + # launches it. + runs = list( + generate_run_configs("launch.yaml", + batch_seeds=True, + round_name="r3", + envs=["bridge"])) + assert [r.experiment_id for r in runs] == ["bridge-mb_r3"] diff --git a/tests/test_docker_option_plan.py b/tests/test_docker_option_plan.py index 9b12d164b2..3c8b177f9d 100644 --- a/tests/test_docker_option_plan.py +++ b/tests/test_docker_option_plan.py @@ -28,7 +28,7 @@ import predicators.utils as pred_utils from predicators.settings import CFG -# Config matching predicatorv3/predicator_v3.yaml (mf_agent approach) +# Flags of the retired phased mf_agent config. _CFG_OVERRIDES = { "env": "pybullet_domino", "approach": "agent_model_free", diff --git a/tests/test_five_seed_benchmark.py b/tests/test_five_seed_benchmark.py deleted file mode 100644 index 66a5abb951..0000000000 --- a/tests/test_five_seed_benchmark.py +++ /dev/null @@ -1,54 +0,0 @@ -"""The additional paper seeds preserve the intended capability contracts.""" -from scripts.cluster_utils import SingleSeedRunConfig, generate_run_configs - - -def test_five_seed_launch_contracts() -> None: - """Exactly six arms, five domains, and two fresh seeds with matched - costs.""" - runs = [] - for config in generate_run_configs( - "predicatorv3/continual_benchmark_five_seeds.yaml", False): - assert isinstance(config, SingleSeedRunConfig) - runs.append(config) - assert len(runs) == 60 - assert {r.seed for r in runs} == {3, 4} - assert len({(r.experiment_id, r.seed) for r in runs}) == 60 - assert len({r.env for r in runs}) == 5 - # Direct + scene shares the direct-agent class, not its capability flags. - assert len({r.approach for r in runs}) == 5 - assert len({r.experiment_id.split("-", 1)[1] for r in runs}) == 6 - for run in runs: - flags = run.flags - assert flags["agent_sdk_model_name"] == "claude-opus-5" - assert flags["partially_observable"] - assert not flags["continual_skill_preflight"] - assert not flags["continual_validation_audit"] - assert flags["continual_wall_clock_hours"] == 48.0 - assert "auto_resume" in run.args - if run.approach == "agent_continual_model_free": - assert not flags["agent_planner_use_simulator"] - assert not flags["continual_uncertainty_decisions"] - assert bool(flags.get("continual_provide_scene_package")) == ( - "mf_scene_package" in run.experiment_id) - elif run.approach == "agent_continual_no_fitting": - assert flags["agent_sim_learn_declared_params_only"] - assert flags["continual_uncertainty_decisions"] - assert flags["continual_require_model_on_test"] - elif run.approach == "agent_continual_no_uncertainty": - for name in ("continual_obs_noise_declared", - "continual_uncertainty_decisions", - "agent_sim_learn_param_uncertainty", - "code_sim_learning_interval_belief", - "code_sim_learning_carry_posterior", - "code_sim_learning_rollout_noise_filter", - "continual_belief_frame"): - assert not flags[name] - assert flags["continual_require_model_on_test"] - if run.env == "pybullet_bridge": - assert flags["bridge_test_span_blocks"] == 4 - assert flags["continual_steps_per_level"] == 10000 - else: - assert flags["continual_steps_per_level"] == 5000 - if run.env == "pybullet_balloons": - assert flags["num_train_tasks"] == 2 - assert flags["balloons_goal_dwell_steps"] == 25