Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
23 changes: 23 additions & 0 deletions AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -80,6 +80,29 @@ A2/A3 probability adapters and five-fold Adult integration; full Covertype runs
and remaining adapters/searches/D5 are open. Sprint 049 records all five full
Covertype folds timing out at the 90-second fit cap. Next profile the current
CPU path on that same input before expanding adapters; A3 validation is incomplete.
Sprint 050 identifies histogram aggregation and repeated candidate row hashing
in a bounded full-input profile. Next hoist invariant candidate row hashing with
exact identity/candidate checks, then rerun the bounded workload. Sprint 051
completes that change: fold zero passes in 87.4 seconds with exact fresh replay;
the other folds were pending at that revision. Sprint 052 reuses selected
histogram statistics with exact conformance checks;
all five full Covertype folds pass within the unchanged cap and replay exactly.
Sprint 053 connects A5 independent quantiles to all five frozen Bike origins
with exact fresh replay and independently recomputed pinball scores. Sprint 054
adds A7 explicit count/exposure binding on all five frozen frequency folds with
exact replay. Sprint 055 adds A8 claim severity on all five frozen grouped folds.
Sprint 056 adds direct A9 annualized Tweedie on all five aggregate folds, with
verified exposure weights and period conversion. Sprint 057 binds matched paid
events to all five frozen A9 input folds. Sprint 058 trains and replays the public
composition on all five packets with independent component selection. Sprint 059
adds A10 fixed-scale survival on all five frozen folds. Sprint 060 adds A12
structured Formula with age separated from tree features on all five folds.
Sprint 061 adds the current A4 query-aware adapter with synthetic direct/fresh
parity; MSLR source/binding remains open. [Sprint 062](v1-sprints/062-cpu-exit-and-gpu-entry.md)
reviews CPU exit and GPU entry. Next: installed D5 probes for distinct run-ID RNG
streams and stale preparation rejection, then source/workflow and gate reconciliation.
Real searches, joint A9 selection and formal E5 remain open. F3 has not started;
the review defines its first vertical path without changing phase ordering.
These internal trials
do not establish E5/E7.
Independent validation stopping is implemented in Sprint 039. A6 CPU workflows and shared
Expand Down
161 changes: 161 additions & 0 deletions benchmarks/v1/composition_smoke.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,161 @@
"""Five hash-bound composition fits and inference replays; no quality gate."""

import argparse
import importlib.metadata
import json
import os
import platform
import subprocess
import sys
from pathlib import Path

import numpy as np

from benchmarks.v1.paid_event_data import digest
from benchmarks.v1.process_runner import execute


def run(manifest, directory):
manifest, directory = Path(manifest).resolve(), Path(directory).resolve()
bound = json.loads(manifest.read_text())
if len(bound["cells"]) != 5 or {c["fold"] for c in bound["cells"]} != set(range(5)):
raise ValueError("all five bound folds required")
directory.mkdir(parents=True, exist_ok=True)
if any(directory.iterdir()):
raise ValueError("fresh output directory required")
repo = Path(__file__).resolve().parents[2]
script = Path(__file__).with_name("composition_worker.py")
sources = [
*sorted((repo / "src/openboost").glob("*.py")),
Path(__file__),
script,
Path(__file__).with_name("process_runner.py"),
Path(__file__).with_name("paid_event_data.py"),
]
report = dict(
scope="Component-selected composition validation only; no joint selection or quality gate",
revision=subprocess.check_output(["git", "rev-parse", "HEAD"], text=True).strip(),
dirty=bool(subprocess.check_output(["git", "status", "--porcelain"])),
argv=[
sys.executable,
"-m",
"benchmarks.v1.composition_smoke",
str(manifest),
str(directory),
],
binding_sha256=digest(manifest),
python=platform.python_version(),
os=platform.platform(),
packages={n: importlib.metadata.version(n) for n in ["numpy", "openboost"]},
cpu_count=os.cpu_count(),
threads=1,
memory_cap=None,
gpu=None,
sources={str(p.relative_to(repo)): digest(p) for p in sources},
cells=[],
)
for cell in bound["cells"]:
packet = manifest.parent / cell["path"]
if digest(packet) != cell["sha256"]:
raise ValueError("composition packet hash differs")
job = dict(
seed=cell["fold"],
patience=3,
config=dict(rounds=4, learning_rate=0.1, bins=32, max_depth=2, reg_lambda=1),
)
job_path = directory / f"job-{cell['fold']}.json"
job_path.write_text(json.dumps(job, indent=2) + "\n")
output = directory / str(cell["fold"])
record = execute(
[sys.executable, str(script), "fit", str(job_path), str(packet)],
output,
timeout_s=90,
threads=1,
)
record.update(fold=cell["fold"], job=job, input_sha256=digest(packet))
if record["status"] == "pass":
try:
with np.load(packet) as a:
np.savez(
output / "features.npz",
x=a["x_validation"],
row_ids=a["row_ids_validation"],
exposure=a["exposure_validation"],
)
command = [
sys.executable,
str(script),
"predict",
str(output / "model.bin"),
str(output / "features.npz"),
str(output / "replay.npz"),
]
record["replay_command"] = command
subprocess.run(
command,
check=True,
capture_output=True,
timeout=30,
env=dict(
os.environ,
OMP_NUM_THREADS="1",
OPENBLAS_NUM_THREADS="1",
MKL_NUM_THREADS="1",
),
)
with (
np.load(output / "predictions.npz") as a,
np.load(output / "replay.npz") as b,
np.load(output / "features.npz") as f,
):
assert (
set(a.files)
== set(b.files)
== {
"row_ids",
"paid_count_rate",
"paid_count_mean",
"severity_mean",
"annualized_mean",
"period_mean",
}
)
for k in a.files:
np.testing.assert_array_equal(a[k], b[k])
np.testing.assert_array_equal(a["row_ids"], f["row_ids"])
for k in set(a.files) - {"row_ids"}:
assert (
a[k].shape == a["row_ids"].shape
and np.isfinite(a[k]).all()
and (a[k] > 0).all()
)
np.testing.assert_allclose(
a["annualized_mean"], a["paid_count_rate"] * a["severity_mean"], rtol=1e-12
)
np.testing.assert_allclose(
a["period_mean"], a["annualized_mean"] * f["exposure"], rtol=1e-12
)
np.testing.assert_allclose(
a["paid_count_mean"], a["paid_count_rate"] * f["exposure"], rtol=1e-12
)
record.update(
fresh_process_exact=True,
products_and_units=True,
replay_sha256=digest(output / "replay.npz"),
training=json.loads((output / "training.json").read_text()),
)
except Exception as error:
record.update(status="error", reason=str(error))
report["cells"].append(record)
(directory / "summary.json").write_text(json.dumps(report, indent=2) + "\n")
print(f"Fold {cell['fold']}: {record['status']}", flush=True)
return report


if __name__ == "__main__":
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("manifest", type=Path)
parser.add_argument("directory", type=Path)
args = parser.parse_args()
if any(c["status"] != "pass" for c in run(args.manifest, args.directory)["cells"]):
raise SystemExit(1)
105 changes: 105 additions & 0 deletions benchmarks/v1/composition_worker.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,105 @@
"""Current CPU matched-payment composition fit and inference-only replay."""

import argparse
import json
from dataclasses import asdict
from pathlib import Path

import numpy as np

from openboost import NumericData, RunContext
from openboost.composition import FrequencySeverity, paid_loss_problems


def fit(job, arrays):
from openboost.recipes import gamma, poisson

if (
set(job) != {"seed", "config", "patience"}
or type(job["seed"]) is not int
or job["seed"] < 0
):
raise ValueError("explicit composition seed/config/patience required")
cfg = job["config"]
if set(cfg) != {"rounds", "learning_rate", "bins", "max_depth", "reg_lambda"}:
raise ValueError("unsupported composition configuration")
if type(cfg["rounds"]) is not int or cfg["rounds"] <= 0:
raise ValueError("positive round budget required")
roles = {"x", "row_ids", "paid_count", "paid_total", "exposure"}
if set(arrays) != {f"{r}_{p}" for r in roles for p in ("train", "validation")}:
raise ValueError("matched training/validation roles only")
problems = []
names = None
for part in ("train", "validation"):
x, ids = arrays["x_" + part], arrays["row_ids_" + part]
if x.ndim != 2 or not len(x) or not np.isfinite(x).all():
raise ValueError("nonempty finite encoded features required")
names = tuple(f"x{i}" for i in range(x.shape[1])) if names is None else names
data = NumericData(x, ids, names)
problems.append(
paid_loss_problems(
data,
arrays["paid_count_" + part],
arrays["paid_total_" + part],
arrays["exposure_" + part],
)
)
if np.intersect1d(arrays["row_ids_train"], arrays["row_ids_validation"]).size:
raise ValueError("policy partitions overlap")
models, components = [], {}
selection = "final" if job["patience"] is None else "best_component_validation"
for i, (name, recipe) in enumerate((("frequency", poisson), ("severity", gamma))):
result = recipe(
problems[0][i],
problems[1][i],
context=RunContext(name, job["seed"]),
patience=job["patience"],
**cfg,
)
model = result.state.model if job["patience"] is None else result.state.best_model
models.append(model)
components[name] = dict(
stop={**asdict(result.stop), "reason": result.stop.reason},
selected_identity=model.identity,
best_score=result.state.best_score,
accepted_commits=result.state.version,
)
model = FrequencySeverity(*models)
data = problems[1][0].data
outputs = model.predict(data, data, arrays["exposure_validation"])
return model, outputs, dict(selection=selection, joint_selection=False, components=components)


def replay(model_path, packet):
if set(packet) != {"x", "row_ids", "exposure"}:
raise ValueError("inference features, policy IDs and exposure only")
model = FrequencySeverity.load(model_path)
data = NumericData(packet["x"], packet["row_ids"], model.frequency.feature_names)
return model.predict(data, data, packet["exposure"])


def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("mode", choices=("fit", "predict"))
parser.add_argument("spec", type=Path)
parser.add_argument("packet", type=Path)
parser.add_argument("output", type=Path, nargs="?")
args = parser.parse_args()
with np.load(args.packet, allow_pickle=False) as a:
arrays = dict(a)
if args.mode == "predict":
if args.output is None:
raise ValueError("prediction output path required")
outputs = replay(args.spec, arrays)
np.savez(args.output, row_ids=arrays["row_ids"], **outputs)
else:
if args.output is not None:
raise ValueError("fit writes to its dedicated working directory")
model, outputs, training = fit(json.loads(args.spec.read_text()), arrays)
model.save("model.bin")
np.savez("predictions.npz", row_ids=arrays["row_ids_validation"], **outputs)
Path("training.json").write_text(json.dumps(training, indent=2, allow_nan=False) + "\n")


if __name__ == "__main__":
main()
19 changes: 19 additions & 0 deletions benchmarks/v1/evidence/aggregate-056/A9/0/execution.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
{
"artifacts": {
"model.bin": "05cb37e2b36082e4258cea1ee81d26d048d677dcc265df9f06887cf59bec40a2",
"predictions.npz": "106aa1fc8ff5625ddc993ae7dedf00a86eef64ba429f5bda7b93616b9d21c6b5",
"training.json": "73c8cf78fd6b2886c56ae11ccf1cd4c5913541482d811562d36423dd8c252c8e",
"worker.log": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
},
"command": [
"/Users/jiaruixu/work_space/openboost/.venv/bin/python3",
"/Users/jiaruixu/work_space/openboost/benchmarks/v1/openboost_worker.py",
"/private/tmp/openboost-aggregate-056/A9/0/job.json"
],
"exit_code": 0,
"reason": "",
"status": "pass",
"threads": 1,
"timeout_s": 90,
"wall_s": 19.495609958001296
}
Loading
Loading