Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion docs/amps/empiric-from-assets.md
Original file line number Diff line number Diff line change
Expand Up @@ -45,7 +45,10 @@ Three settings keep this arm's belief from claiming knowledge the data do not gi
The model status lists each widened range.
Added after seed 0 of the round `fixes_r2` declared spinning friction on 0 to 0.05 and rolling friction on 0 to 0.002, while the world has 0.5 and 0.006: its 16 joint draws all predicted a cascade, and the world's chain stalled at its second bridge.

The supplied-base arms keep the anchored prior: their starting values are the base's calibrated defaults.
On a test level the arm acts only on a fitted model: `skills_invoke` and `skills_execute_plan` refuse until `sim.fit()` has run on the current content of `simulator.py` (`_fit_readiness`), so the belief that chooses the plan comes from the recordings.
Seed 0 of `fixes_r2` found the level-1 fit slow and uninformative, tuned its values by hand, and played the test level unfitted.

The supplied-base arms keep the anchored prior and leave fitting to the agent: their starting values are the base's calibrated defaults.
The real-to-sim comparison fits nothing, samples nothing and keeps the declared ranges.

## Pilot
Expand Down
9 changes: 7 additions & 2 deletions predicators/agent_sdk/play_prompts.py
Original file line number Diff line number Diff line change
Expand Up @@ -121,7 +121,8 @@ def build_play_system_prompt(tool_names: Sequence[str],
frozen_model_supplied: bool = False,
oracle_dynamics: bool = False,
scene_built: bool = False,
scene_package: bool = False) -> str:
scene_package: bool = False,
fit_gate: bool = False) -> str:
"""The system prompt of the run's conversation.

The tool surface selects the variant: an arm with ``run_python``
Expand Down Expand Up @@ -152,6 +153,8 @@ def build_play_system_prompt(tool_names: Sequence[str],
engine, the manifest and the assets) describes that reference
listing instead; for the model-free arm it adds that listing as
plain files to use however the agent likes, with no simulator.
``fit_gate`` (EMPIRIC from assets) says the test-level gate also
waits for a fit of the current ``simulator.py``.
"""
names = set(tool_names)
model = "run_python" in names
Expand Down Expand Up @@ -252,7 +255,9 @@ def build_play_system_prompt(tool_names: Sequence[str],
if oracle_dynamics:
sections.append(render("play_system", "oracle_discrepancies"))
if CFG.continual_require_model_on_test:
sections.append(render("play_system", "model_gate"))
sections.append(
render("play_system",
"model_gate_fit" if fit_gate else "model_gate"))
if CFG.continual_skill_preflight:
sections.append(render("play_system", "skill_preflight"))
if CFG.agent_model_repair and not frozen:
Expand Down
9 changes: 9 additions & 0 deletions predicators/agent_sdk/prompts/play_system.md
Original file line number Diff line number Diff line change
Expand Up @@ -184,6 +184,15 @@ The refusal says which condition is unmet.
Train levels are not gated: collect evidence there first.
Loading a model does not itself enable automatic rehearsal; use `sim` to check plans before execution.

<!-- section: model_gate_fit -->
### Test levels require a fitted model

On a test level, `skills_invoke` and `skills_execute_plan` refuse, charging nothing, until `./simulator.py` loads, declares `RESIDUAL_FEATURES` and, when it declares parameters, has been fitted with `sim.fit()` in its current form.
Every edit of the file needs a new fit before the next skill on a test level, so finish editing first.
The refusal says which condition is unmet.
Train levels are not gated: collect evidence there first.
Loading a model does not itself enable automatic rehearsal; use `sim` to check plans before execution.

<!-- section: skill_preflight -->
### Every skill request is rehearsed first

Expand Down
29 changes: 23 additions & 6 deletions predicators/approaches/agent_continual_approach.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,8 @@
from predicators.approaches.continual_play_mixin import ContinualPlayMixin
from predicators.approaches.scene_package_mixin import ScenePackageMixin
from predicators.code_sim_learning.rollout_env import dispose_env
from predicators.code_sim_learning.utils import LearnedSimulator, apply_rules
from predicators.code_sim_learning.utils import LearnedSimulator, \
apply_rules, read_residual_env
from predicators.envs import create_new_env
from predicators.observation_noise import ObservationNoise
from predicators.option_model import _OptionModelBase, _OracleOptionModel
Expand Down Expand Up @@ -517,9 +518,11 @@ def _model_readiness(
file exists and execs, and it declares ``RESIDUAL_FEATURES``
(the deploy asserts it; Boil and Fan Sonnet runs on Sept 16,
2026 wrote models without it and ran on base physics). Fitting
is the agent's call: a round deploys an unfitted model at its
carried or declared values. Loads are cached by content digest
so a sweep of ``skills_invoke`` calls execs the file once.
is the agent's call here: a round deploys an unfitted model at
its carried or declared values. An arm whose parameters start
from guesses also asks for a fit (:meth:`_fit_readiness`). Loads
are cached by content digest so a sweep of ``skills_invoke``
calls execs the file once.
"""
head = "Test level: this arm acts through its model. "
if not os.path.isfile(simulator_file):
Expand All @@ -533,13 +536,17 @@ def _model_readiness(
cache = {}
setattr(self, "_readiness_cache_store", cache)
if cache.get("digest") != digest:
rules, specs, features, _ns = \
rules, specs, features, ns = \
self._load_simulator_from_module_file(
simulator_file, trajectories)
subclass = (read_residual_env(ns)
if isinstance(ns, dict) else None)
cache.clear()
cache.update(digest=digest,
loadable=rules is not None and specs is not None,
features=features is not None)
features=features is not None,
params=bool(specs)
or bool(getattr(subclass, "AGENT_PARAM_SPECS", [])))
if not cache["loadable"]:
return (head + "`./simulator.py` does not load (run "
"`sim.reset(current=True)` in run_python to see the "
Expand All @@ -550,6 +557,16 @@ def _model_readiness(
"the subclass first: the observed features the fit "
"scores your model on, such as the poses forces move "
"and the readings your mechanisms change.")
if cache["params"]:
return self._fit_readiness(head, digest)
return None

def _fit_readiness( # pylint: disable=useless-return
self, head: str, digest: str) -> Optional[str]:
"""Why a loadable model with parameters must not act yet, given the
digest of its file, or None; the supplied-base arms leave fitting to
the agent."""
del head, digest
return None

def _base_physics_probe_model(self) -> _OracleOptionModel:
Expand Down
14 changes: 13 additions & 1 deletion predicators/approaches/agent_continual_from_assets_approach.py
Original file line number Diff line number Diff line change
Expand Up @@ -135,7 +135,7 @@ def _base_sim_reference_paths(self) -> List[str]:
# -- The prompt ---------------------------------------------------------

def _play_prompt_options(self) -> Dict[str, Any]:
return {"scene_built": True}
return {"scene_built": True, "fit_gate": self._belief_over_guesses}

def _play_model_contract(self, **options: Any) -> str:
return super()._play_model_contract(
Expand All @@ -160,6 +160,18 @@ def _fit_status_text(self) -> str:
return (f"{status}; declared ranges widened to the engine's "
f"plausible range: {', '.join(widened)}")

def _fit_readiness(self, head: str, digest: str) -> Optional[str]:
"""On a test level, a model with parameters acts only once its current
file is fitted: the starting values are the agent's guesses, and the
belief that chooses the plan should come from the recordings."""
if not self._belief_over_guesses or \
self._probe_fit_state().get("digest") == digest:
return None
return (head + "`./simulator.py` has not been fitted in its current "
"form. Call `sim.fit()` in run_python and read its report "
"before invoking a skill; every edit of the file needs a new "
"fit.")

# -- Materials the model declares ----------------------------------------

def _widened_ranges(self) -> List[str]:
Expand Down
21 changes: 19 additions & 2 deletions tests/approaches/test_agent_continual_real_to_sim_approach.py
Original file line number Diff line number Diff line change
Expand Up @@ -244,6 +244,9 @@ def test_agent_builds_the_scene_and_rehearses_in_it(tmp_path: Any,
assert "EMPIRIC from assets" in prompt
assert "harness fits nothing" not in prompt
assert "visible physics from the first round" not in prompt
# The test-level gate also waits for a fit in EMPIRIC from assets.
assert ("Test levels require a fitted model" in prompt) == from_assets
assert ("Test levels require a loaded model" in prompt) != from_assets
assert "visible base physics" not in prompt
assert "physical parameter menu" not in prompt.lower()
seen: List[str] = []
Expand Down Expand Up @@ -368,8 +371,15 @@ def query(message: str, *_args: Any, **_kwargs: Any) -> Any:
# The world behind sim is the agent's class, not the twin.
assert type(approach._base_env).__name__ == "BoilScene"
assert approach._tool_context.env is approach._base_env
assert approach._model_readiness(str(sandbox / "simulator.py"),
[]) is None
# EMPIRIC from assets acts on a test level only once the current
# file is fitted; the fit below opens the gate. Real-to-sim fits
# nothing and needs only a loadable model.
readiness = approach._model_readiness(str(sandbox / "simulator.py"),
[])
if from_assets:
assert readiness is not None and "sim.fit()" in readiness
else:
assert readiness is None
# Parameter changes, independent fit worlds and reset isolation use
# the scene class, not a supplied domain simulator.
worlds = [approach._get_rollout_fit_env()() for _ in range(2)]
Expand Down Expand Up @@ -406,6 +416,13 @@ def roll(world: Any) -> float:
assert not fitted.startswith("ERROR"), fitted
assert approach._probe_fit_state().get(
"fit_result") is not None, fitted
assert approach._model_readiness(str(sandbox / "simulator.py"),
[]) is None
# An edit needs a new fit.
with open(sandbox / "simulator.py", "a", encoding="utf-8") as f:
f.write("\n# edited\n")
assert "sim.fit()" in approach._model_readiness(
str(sandbox / "simulator.py"), [])
assert "Give-up recorded" in _call(approach, "give_up", note="done")
return _result()

Expand Down
Loading