From 0eaa0e519ea4dbc4f921a7788e103b9fa7cf83a2 Mon Sep 17 00:00:00 2001 From: Yichao Liang Date: Sat, 10 Oct 2026 09:35:14 -0400 Subject: [PATCH 1/3] Defer to guessed starting values no more than the prior does The grid sweep treats a candidate within 5% of the model-bias SSE as data-equivalent and keeps the one nearest the anchor, and the anchor ablation pins moved parameters back to their anchors. Both defer to the anchor beyond the fit's prior, which a calibrated baseline earns. Under code_sim_learning_prior_spans_bounds the anchors are the agent's guesses: seed 0 of the Domino round fixes_r3 kept a guessed spinning friction of 0.005 although its best candidate, 0.67 with the world at 0.5, fit 2.8 nats better. The deployed value then sat outside the belief's own 68% interval of 0.39 to 0.71, the report contradicted itself, and the agent replayed its plan at the guess. With guessed starting values the flat band is the likelihood floor alone, the scale the parameter belief scores with, and the anchor ablation does not run. Supplied-base arms keep both. --- predicators/code_sim_learning/config.py | 20 +++++ predicators/code_sim_learning/grid_seed.py | 2 +- .../code_sim_learning/physical_sysid.py | 8 +- predicators/settings.py | 7 ++ .../code_sim_learning/test_physical_sysid.py | 75 +++++++++++++++++++ 5 files changed, 109 insertions(+), 3 deletions(-) diff --git a/predicators/code_sim_learning/config.py b/predicators/code_sim_learning/config.py index bfd5c2f551..afea697ee3 100644 --- a/predicators/code_sim_learning/config.py +++ b/predicators/code_sim_learning/config.py @@ -107,6 +107,25 @@ class SysIdConfig: # scales (predicators/observation_noise.py); None when observations # are exact or the channel is undeclared. observation_noise: Optional[ObservationNoise] = None + # The fit's starting values are the agent's guesses + # (code_sim_learning_prior_spans_bounds), not calibrated baselines. + anchors_are_guesses: bool = False + + @property + def flat_band_frac(self) -> float: + """The relative part of the flat band (``grid_flat_frac``), 0 when the + anchors are guesses. + + The band lets a candidate within ``grid_flat_frac`` of the + model-bias SSE count as data-equivalent, so the anchor-nearest + one wins: a deference to the anchor beyond the prior, which a + calibrated baseline earns and a guess does not. With guesses, + data-equivalence is the likelihood floor alone + (:func:`~predicators.code_sim_learning.grid_seed.flat_tolerance`), + the scale the parameter belief scores with, so the fitted point + stays inside the belief's high-density region. + """ + return 0.0 if self.anchors_are_guesses else self.grid_flat_frac @classmethod def from_cfg(cls) -> SysIdConfig: @@ -162,4 +181,5 @@ def from_cfg(cls) -> SysIdConfig: track_frame_yaw=CFG.code_sim_learning_track_frame_yaw, track_frame_xy=tuple(CFG.code_sim_learning_track_frame_xy), observation_noise=_declared_observation_noise(), + anchors_are_guesses=CFG.code_sim_learning_prior_spans_bounds, ) diff --git a/predicators/code_sim_learning/grid_seed.py b/predicators/code_sim_learning/grid_seed.py index d19debe6b9..648387af89 100644 --- a/predicators/code_sim_learning/grid_seed.py +++ b/predicators/code_sim_learning/grid_seed.py @@ -260,7 +260,7 @@ def _grid_seed_physical_specs( num_points = config.grid_seed_points num_passes = max(1, config.grid_sweep_passes) refine_evals = config.grid_refine_evals - flat_frac = config.grid_flat_frac + flat_frac = config.flat_band_frac physical_names = [s.name for s in physical_specs] all_specs = list(physical_specs) + list(rule_specs) names = [s.name for s in all_specs] diff --git a/predicators/code_sim_learning/physical_sysid.py b/predicators/code_sim_learning/physical_sysid.py index 1cca69cedd..72242955c6 100644 --- a/predicators/code_sim_learning/physical_sysid.py +++ b/predicators/code_sim_learning/physical_sysid.py @@ -340,7 +340,11 @@ def fit_params_rollout( sensitivity=sensitivity, lm_notes=lm_notes) n_lm = num_rollouts_run() - n_start - n_grid - if (config.anchor_ablation and config.grid_flat_frac > 0 and trajectories): + # The ablation pins moved parameters back to their anchors, which + # only calibrated baselines warrant; a guess has no claim beyond the + # prior the fit already folds in. + if (config.anchor_ablation and not config.anchors_are_guesses + and config.grid_flat_frac > 0 and trajectories): result = _anchor_backward_elimination( base_env, trajectories, @@ -538,7 +542,7 @@ def refit_pinned(surviving: List[ParamSpec], # The grid flat set's tolerance (interval-belief terms included, # see grid_seed.flat_tolerance), so "data-equivalent" means the # same thing here as in the sweep. - tol = flat_tolerance(sse_curr, noise_floor, config.grid_flat_frac, + tol = flat_tolerance(sse_curr, noise_floor, config.flat_band_frac, noise_sse, sigma_tol) cost_curr = prior_cost(point) best: Optional[Tuple[float, ParamSpec, Dict[str, float]]] = None diff --git a/predicators/settings.py b/predicators/settings.py index e7abae389e..f9bc7b1723 100644 --- a/predicators/settings.py +++ b/predicators/settings.py @@ -2410,6 +2410,13 @@ class GlobalSettings: # wide as its range, or the belief claims a knowledge it lacks # (2026-10-04: a spinning friction started at 0.05 came out # 0.057-0.12 with the world at 0.5). + # Guesses earn no deference beyond that prior either: the grid's flat + # band drops its relative term (SysIdConfig.flat_band_frac), so + # data-equivalence is the likelihood floor the belief scores with, + # and the anchor ablation does not run. With the band, a spinning + # friction whose best candidate (0.67, world 0.5) beat the guessed + # 0.005 by 2.8 nats stayed at 0.005, outside the belief's own 68% + # interval of 0.39 to 0.71 (2026-10-09). code_sim_learning_prior_spans_bounds = False # Goodness-of-fit trimming threshold for rollout sysID, as a # multiple of the fit's noise_sigma (0 disables): a segment whose diff --git a/tests/code_sim_learning/test_physical_sysid.py b/tests/code_sim_learning/test_physical_sysid.py index 0ab172511a..335d7bcb45 100644 --- a/tests/code_sim_learning/test_physical_sysid.py +++ b/tests/code_sim_learning/test_physical_sysid.py @@ -1434,6 +1434,81 @@ def _ablation_fixtures(): return spec_a, spec_b, anchors, prior_sigma, config, traj +def _spin_sse(params): + """Spinning friction above 0.3 fits 2% better: inside the relative flat + band of the model-bias SSE, beyond a zero likelihood floor.""" + return 10.0 - (0.2 if params["spinning_friction"] > 0.3 else 0.0) + + +def test_grid_sweep_defers_to_calibrated_anchors_only(monkeypatch): + """A calibrated anchor keeps its value when a better candidate lies inside + the relative flat band; a guessed one does not. + + With guessed starting values (code_sim_learning_prior_spans_bounds) + the flat band is the likelihood floor alone, the scale the parameter + belief scores with. Seed 0 of the Domino round fixes_r3 kept a + guessed spinning friction of 0.005 although its best candidate, 0.67 + with the world at 0.5, was 2.8 nats better, outside the belief's own + 68% interval. + """ + specs = [ParamSpec("spinning_friction", 0.005, lo=0.0, hi=1.0)] + anchors = {"spinning_friction": 0.005} + seeds, _info = _run_sweep(monkeypatch, + _spin_sse, + specs, + anchors, + code_sim_learning_prior_spans_bounds=False) + assert seeds["spinning_friction"] == 0.005 + seeds, _info = _run_sweep(monkeypatch, + _spin_sse, + specs, + anchors, + code_sim_learning_prior_spans_bounds=True) + assert seeds["spinning_friction"] > 0.3 + + +def test_anchor_ablation_runs_for_calibrated_anchors_only(monkeypatch): + """The anchor ablation pins moved parameters back to calibrated baselines; + with guessed starting values it does not run, since a guess has no claim + beyond the prior the fit folds in.""" + from predicators.settings import CFG + specs = [ParamSpec("mu", 0.5, lo=0.1, hi=1.0)] + calls = [] + + def fake_lm(_env, + _trajs, + physical_specs, + _features, + _rules=(), + rule_specs=(), + _latent=None, + **_kwargs): + all_specs = list(physical_specs) + list(rule_specs) + return np.array([s.init_value for s in all_specs], dtype=float), None + + def fake_grid(_env, _trajs, physical_specs, *_args, **_kwargs): + return list(physical_specs), {} + + def fake_ablation(*args, **_kwargs): + calls.append(True) + return args[11] # the fit result, unchanged + + monkeypatch.setattr(physical_sysid, "fit_map_lm_rollout", fake_lm) + monkeypatch.setattr(physical_sysid, "_grid_seed_physical_specs", fake_grid) + monkeypatch.setattr(physical_sysid, "compute_rollout_sse", + lambda *_args, **_kwargs: 1.0) + monkeypatch.setattr(physical_sysid, "_anchor_backward_elimination", + fake_ablation) + trajectory = _trajectory([0.0, 0.1, 0.2]) + for guesses, ran in ((False, True), (True, False)): + monkeypatch.setattr(CFG, "code_sim_learning_prior_spans_bounds", + guesses) + calls.clear() + physical_sysid.fit_params_rollout(None, [trajectory], specs, + _RESIDUAL_FEATURES) + assert bool(calls) is ran + + def test_anchor_ablation_reverts_compensatory_param(): """A co-adapted MAP on the compensation ridge is resolved to the anchor- consistent basin: the tight-prior param reverts to its anchor and the wide- From 3a612f725ec43f7635aef94863c65a205b99ff86 Mon Sep 17 00:00:00 2001 From: Yichao Liang Date: Sat, 10 Oct 2026 09:51:18 -0400 Subject: [PATCH 2/3] Start the fit-cache test from the default config test_run_rollout_sysid_fit_cache_and_report_isolation checks the trust selection, which runs only without the joint belief, but read whatever config the previous test left. After test_continual_joint_belief, which sets belief_joint_draws, the fit took the joint-belief path and applied its point estimate (gain 1.9993 instead of the trusted 1.5), so the test failed whenever the two landed in that order. --- tests/code_sim_learning/test_orchestrator.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tests/code_sim_learning/test_orchestrator.py b/tests/code_sim_learning/test_orchestrator.py index 905702fa8b..4deaf6eb19 100644 --- a/tests/code_sim_learning/test_orchestrator.py +++ b/tests/code_sim_learning/test_orchestrator.py @@ -71,6 +71,9 @@ def _trajectory(num_steps=10, gain=2.0): def test_run_rollout_sysid_fit_cache_and_report_isolation(): """The fit core is memoized per (artifact, data) key; adjusters and trust selection stay caller-local on deep-copied reports.""" + # The trust selection runs only without the joint belief, and an + # earlier test may leave belief_joint_draws set. + utils.reset_config({}) env = _GainEnv() spec = ParamSpec("gain", 1.0, lo=0.1, hi=10.0, scale="log") traj = _trajectory() From eaeb8f400c21b5ea55c219499620dae922422c43 Mon Sep 17 00:00:00 2001 From: Yichao Liang Date: Sat, 10 Oct 2026 09:51:22 -0400 Subject: [PATCH 3/3] Report the belief's own most likely value and densest interval The fit report printed each parameter as "most likely " next to the central 68% interval of its belief factor. The point estimate need not be the factor's mode, and a central interval excludes a mode at a bound, so the report could contradict itself: seed 0 of the Domino round fixes_r3 read "most likely 0.05; 68% posterior interval [0.2694, 0.291]" and stopped trusting the belief. Each line now gives the factor's mode and its 68% highest-density interval, which contains the mode, and names the fit's point estimate when it lies outside that interval. The note under the report says the planning model runs at the point estimate. --- predicators/agent_sdk/tools/synthesis.py | 8 +-- .../code_sim_learning/parameter_belief.py | 70 +++++++++++++++++-- .../test_parameter_belief.py | 32 +++++++++ 3 files changed, 102 insertions(+), 8 deletions(-) diff --git a/predicators/agent_sdk/tools/synthesis.py b/predicators/agent_sdk/tools/synthesis.py index ea3ebf4410..657ed999cf 100644 --- a/predicators/agent_sdk/tools/synthesis.py +++ b/predicators/agent_sdk/tools/synthesis.py @@ -734,10 +734,10 @@ def _evaluate_rollout_fit(rules: list, "not a parameter value.") else: lines.append( - "Applied to the planning base env: the most likely value " - "of every parameter. sim.run, sim.refine, experiment " - "scores and execution monitoring draw the parameters " - "from this belief.") + "Applied to the planning base env: the fit's point " + "estimate of every parameter. sim.run, sim.refine, " + "experiment scores and execution monitoring draw the " + "parameters from this belief.") return "\n".join(lines) interval_belief = CFG.code_sim_learning_interval_belief if interval_belief: diff --git a/predicators/code_sim_learning/parameter_belief.py b/predicators/code_sim_learning/parameter_belief.py index f9e76bb0e5..5c7b344cf5 100644 --- a/predicators/code_sim_learning/parameter_belief.py +++ b/predicators/code_sim_learning/parameter_belief.py @@ -179,6 +179,27 @@ def quantile(self, q: float) -> float: return float(self.grid[0]) return float(self._invert(np.array([q]))[0]) + def mode(self) -> float: + """The value of highest density.""" + return float(self.grid[int(np.argmax(self.density))]) + + def highest_density_interval(self, coverage: float) -> Tuple[float, float]: + """The span of the densest pieces that together hold at least + ``coverage`` of the mass. + + For a unimodal density this is the shortest interval with that + mass, and it always contains the mode; a central interval + excludes a mode that sits at a bound. + """ + if self.is_point: + return float(self.grid[0]), float(self.grid[0]) + a, b, fa, fb, mass = self._pieces() + order = np.argsort(-(fa + fb), kind="stable") + cum = np.cumsum(mass[order]) + count = int(np.searchsorted(cum, coverage * cum[-1])) + 1 + chosen = order[:min(count, order.size)] + return float(np.min(a[chosen])), float(np.max(b[chosen])) + @dataclass(frozen=True) class DiscretePosterior: @@ -222,6 +243,10 @@ def quantile(self, q: float) -> float: idx = int(np.searchsorted(cum, min(max(q, 0.0), 1.0) - 1e-12)) return float(self.values[min(idx, self.values.size - 1)]) + def mode(self) -> float: + """The most probable value.""" + return float(self.values[int(np.argmax(self.probs))]) + @dataclass class ParameterBelief: @@ -297,8 +322,41 @@ def interval(self, return float(np.exp(ends[0])), float(np.exp(ends[1])) return float(ends[0]), float(ends[1]) + def most_likely(self, name: str) -> float: + """The mode of ``name``'s factor (external units); the fit's point + estimate for a held parameter.""" + if name in self.discrete: + return self.discrete[name].mode() + line = self.lines.get(name) + if line is None: + return float(self.map_estimate[name]) + mode = line.mode() + if self.scales[self.names.index(name)] == "log": + return float(np.exp(mode)) + return float(mode) + + def highest_density_interval(self, + name: str, + coverage: float = 0.68 + ) -> Tuple[float, float]: + """The highest-density ``coverage`` interval of ``name``'s factor + (external units), which contains its mode.""" + line = self.lines.get(name) + if line is None or name in self.discrete: + return self.interval(name, coverage) + lo, hi = line.highest_density_interval(coverage) + if self.scales[self.names.index(name)] == "log": + return float(np.exp(lo)), float(np.exp(hi)) + return float(lo), float(hi) + def describe(self) -> List[str]: - """One readable line per parameter, plus the noise-level note.""" + """One readable line per parameter, plus the noise-level note. + + Each line gives the factor's own most likely value and its 68% + highest-density interval; the fit's point estimate, which the + planning model runs at, is named when it lies outside that + interval. + """ lines = [] if self.noise_scale > 1.0: lines.append( @@ -317,9 +375,13 @@ def describe(self) -> List[str]: f"{dist.probs[i]:.2f}" for i in top) lines.append(f" {name}: discrete, posterior {shares}") continue - lo, hi = self.interval(name) - lines.append(f" {name}: most likely {value:.4g}; 68% " - f"posterior interval [{lo:.4g}, {hi:.4g}]") + lo, hi = self.highest_density_interval(name) + line = (f" {name}: most likely {self.most_likely(name):.4g}; 68% " + f"posterior interval [{lo:.4g}, {hi:.4g}]") + if not lo <= value <= hi: + line += (f"; the fit's point estimate {value:.4g} lies " + "outside it") + lines.append(line) return lines def to_dict(self) -> Dict[str, Any]: diff --git a/tests/code_sim_learning/test_parameter_belief.py b/tests/code_sim_learning/test_parameter_belief.py index 7f34d071c5..df6c9b9701 100644 --- a/tests/code_sim_learning/test_parameter_belief.py +++ b/tests/code_sim_learning/test_parameter_belief.py @@ -58,6 +58,38 @@ def test_line_posterior_moments_and_sampling(): assert np.all(point.sample(np.random.default_rng(0), 3) == 0.4) +def test_highest_density_interval_holds_a_mode_at_a_bound(): + """A density that falls away from a bound has its mode at the bound: the + central interval excludes it, the highest-density interval starts there and + holds at least the same mass.""" + line = LinePosterior.from_neg_log([0.0, 1.0, 2.0], [0.0, 1.0, 2.0]) + assert line.mode() == 0.0 + assert line.quantile(0.16) > 0.0 + lo, hi = line.highest_density_interval(0.68) + assert lo == 0.0 + inside = (line.grid >= lo) & (line.grid <= hi) + mass = np.trapz(line.density[inside], line.grid[inside]) + assert 0.68 <= mass <= 0.72 + # A narrower interval with the same mass does not exist. + assert hi - lo <= line.quantile(0.84) - line.quantile(0.16) + + +def test_report_names_a_point_estimate_outside_the_belief(): + """The report gives the belief's own most likely value and 68% interval, + and names the fit's point estimate when it lies outside that interval.""" + specs = [ParamSpec("a", 0.0, lo=-1.0, hi=1.0)] + residuals = _gaussian_residuals(["a"], [0.3], [[20.0]]) + off = _build(specs, {"a": 0.0}, residuals) + lo, hi = off.highest_density_interval("a") + assert lo <= off.most_likely("a") <= hi + assert off.most_likely("a") == pytest.approx(0.3, abs=0.02) + assert not lo <= 0.0 <= hi + text = "\n".join(off.describe()) + assert "the fit's point estimate 0 lies outside it" in text + on = _build(specs, {"a": 0.3}, residuals) + assert "point estimate" not in "\n".join(on.describe()) + + def test_gaussian_lines_recover_conditional_variances(): """Each line has the mean-field variance ``1 / Lambda_jj``.""" specs = [