Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
157 commits
Select commit Hold shift + click to select a range
961b6f9
Retrospect v1 progress and plan authoring and execution validation
jxucoder Sep 7, 2026
e15704d
Incorporate landscape feedback into foundation authoring probes
jxucoder Sep 7, 2026
df23796
Plan programmable stopping and installed isolation sprint
jxucoder Sep 7, 2026
8c9f3de
Break remaining v1 work into bounded sprint cards
jxucoder Sep 7, 2026
2871f0f
Allow external stopping policies through structural completion records
jxucoder Sep 7, 2026
e2b6c4c
Verify installed custom stopping and independent run isolation
jxucoder Sep 7, 2026
e3af6c9
Record resource preflight evidence and CPU profiling checkpoint
jxucoder Sep 7, 2026
5243020
Verify numerical worker memory ceiling and peak accounting
jxucoder Sep 7, 2026
fad659b
Freeze bounded practical CPU profiling harness and cases
jxucoder Sep 7, 2026
f5ca13b
Correct Modal generator launch without changing diagnostic policy
jxucoder Sep 7, 2026
c96f7af
Preserve retry declaration in future diagnostic freezes
jxucoder Sep 7, 2026
dff0590
Recover only the interrupted diagnostic profile after preemption
jxucoder Sep 7, 2026
b9cc57a
Record practical CPU replay bottleneck and incomplete sweep evidence
jxucoder Sep 7, 2026
e99e89c
Evaluate new proposal terms with verified reusable encodings
jxucoder Sep 7, 2026
aa4adee
Verify incremental counts and installed extension isolation
jxucoder Sep 7, 2026
da0fede
Preregister same-container wheel comparison for CPU architecture conf…
jxucoder Sep 7, 2026
dd2d696
Close incremental runtime sprint with exact paired CPU evidence
jxucoder Sep 7, 2026
2f5aca2
Add explicit scalar summary retention across CPU recipes
jxucoder Sep 7, 2026
ae8c7af
Preregister bounded full-summary CPU retention comparison
jxucoder Sep 7, 2026
bd82e68
Verify installed summary retention and bounded logical history
jxucoder Sep 7, 2026
e9058cc
Close retention sprint with exact paired predictions and memory evidence
jxucoder Sep 7, 2026
63700db
Bind integrity checks to evaluator-owned frozen execution manifests
jxucoder Sep 7, 2026
1a7bfd5
Use explicit summary retention for current evaluation workers
jxucoder Sep 7, 2026
acac36a
Record frozen-judge evidence and remaining evaluation readiness gaps
jxucoder Sep 7, 2026
26a6797
Add explicit Linux worker privileges and bounded access probes
jxucoder Sep 7, 2026
8eace67
Use summary diagnostics in frequency severity evaluation
jxucoder Sep 7, 2026
4a4cc2d
Record real worker permission and resource probe evidence
jxucoder Sep 7, 2026
176d556
Integrate explicit worker permissions into current selection smoke
jxucoder Sep 7, 2026
a0bc473
Add bounded Linux protected selection preflight
jxucoder Sep 7, 2026
39aa31d
Record Linux selection validation and receipt portability counterexample
jxucoder Sep 7, 2026
68d7b0d
Allow bounded score roundoff in independently verified receipts
jxucoder Sep 7, 2026
c9201ae
Compile frozen OpenBoost A6 resource preflight matrix
jxucoder Sep 7, 2026
4a2960a
Add frozen real A6 resource probes with protected workers
jxucoder Sep 7, 2026
9387d9f
Record real data dispatch approval boundary and local verification
jxucoder Sep 7, 2026
dfe678c
Record real A6 resource passes and practical runtime reflection
jxucoder Sep 7, 2026
6c608b2
Add bounded profiling mode for the real A6 workload
jxucoder Sep 7, 2026
dede27e
Preserve profile failures and isolate periodic stack dumping
jxucoder Sep 7, 2026
17ab9de
Record A6 scoring profile and bounded optimization plan
jxucoder Sep 7, 2026
b642bd5
Cache bounded immutable vector statistic layouts
jxucoder Sep 7, 2026
a3b78a3
Record installed vector layout reuse profile
jxucoder Sep 7, 2026
21f76b1
Use validated scratch leaves for vector candidate scoring
jxucoder Sep 7, 2026
028f309
Record scratch scoring profile and paired-fit checkpoint
jxucoder Sep 7, 2026
73755b1
Prepare exact paired A6 real-fit cost check
jxucoder Sep 7, 2026
2ccc8ec
Record exact paired A6 fit with observed CPU time reduction
jxucoder Sep 7, 2026
cd69c80
Freeze complete A6 CPU method matrix and dispatch blockers
jxucoder Sep 7, 2026
63b4859
Translate explicit comparator bin budgets and verify native parameters
jxucoder Sep 7, 2026
9e34c37
Record installed comparator bin-budget evidence and refreshed plan
jxucoder Sep 7, 2026
02f3d41
Prepare protected real A6 comparator probes and fresh replay
jxucoder Sep 7, 2026
bb08bfc
Record real A6 comparator resource and exact replay evidence
jxucoder Sep 7, 2026
2521619
Prepare frozen deeper 1000-round comparator preflight
jxucoder Sep 7, 2026
a6d5cfd
Record deeper comparator resource and exact replay results
jxucoder Sep 7, 2026
2e8c81c
Reprioritize independent authoring and bounded programmable CUDA
jxucoder Sep 7, 2026
cfca092
Prepare D1 D2 author materials and audit device ownership seam
jxucoder Sep 7, 2026
b8e30bd
Record clean D1 D2 author view and wheel provenance
jxucoder Sep 7, 2026
77aa105
Implement explicit context-owned CUDA storage and real-device checks
jxucoder Sep 7, 2026
8617b38
Record real T4 storage ownership evidence and remaining GPU budget
jxucoder Sep 7, 2026
2e97f3a
Plan programmable CUDA construction and next evidence checkpoint
jxucoder Sep 7, 2026
6ae51dd
Freeze independent routed histogram fixtures before CUDA kernels
jxucoder Sep 7, 2026
4435e1e
Implement owned CUDA fields and routed histogram operations
jxucoder Sep 7, 2026
ad2f4e6
Freeze final bounded T4 aggregation run and exact case judging
jxucoder Sep 7, 2026
4251a36
Record passing T4 aggregation evidence and retrospective checkpoint
jxucoder Sep 7, 2026
0ba39a3
Freeze public CUDA split contracts and exhaustive scalar D2 cases
jxucoder Sep 7, 2026
0bcb52f
Add public CUDA split scores, feasibility masks, routing and leaves
jxucoder Sep 7, 2026
161c02e
Freeze 88-case CUDA split validation package pending new allowance
jxucoder Sep 7, 2026
043861a
Record approved single T4 run for the frozen split package
jxucoder Sep 7, 2026
a29749b
Record Modal source-transfer approval block before dispatch
jxucoder Sep 7, 2026
9ce790e
Record explicit approval for the frozen Modal source upload
jxucoder Sep 7, 2026
e1a9c20
Record 88 passing T4 split tests and foundation retrospective
jxucoder Sep 7, 2026
3561738
Freeze resident scalar ownership design and two-round oracle
jxucoder Sep 7, 2026
e616451
Compose experimental resident squared geometry and scalar trees
jxucoder Sep 7, 2026
807232e
Add experimental resident scalar transactions and recipe
jxucoder Sep 7, 2026
3a8ca34
Freeze 202-case resident CUDA validation package pending approval
jxucoder Sep 7, 2026
c65a969
Record approval for frozen resident CUDA run and private upload
jxucoder Sep 7, 2026
5d3fd28
Record automatic review block before resident CUDA dispatch
jxucoder Sep 7, 2026
c415755
Record explicit approval for private resident CUDA package and run
jxucoder Sep 7, 2026
f5922fb
Record resident CUDA failures and score-symmetry retrospective
jxucoder Sep 7, 2026
13fdd83
Freeze CUDA score-symmetry diagnostics and correction plan
jxucoder Sep 7, 2026
d3ee1c3
Round CUDA scalar score products independently
jxucoder Sep 7, 2026
aab6882
Freeze 212-case CUDA score-correction validation package
jxucoder Sep 7, 2026
af026ef
Record approval for frozen CUDA score-correction run
jxucoder Sep 7, 2026
48a1386
Verify bounded resident CUDA training with 212 passing T4 checks
jxucoder Sep 7, 2026
05b7c66
Plan Normal device composition and freeze independent update fixtures
jxucoder Sep 7, 2026
f75f501
Freeze Normal float32 support and numerical evaluation policy
jxucoder Sep 7, 2026
ab6a37b
Add resident Normal geometry and reusable direction operations
jxucoder Sep 7, 2026
782c9c4
Generalize resident transactions to mapped multi-term updates
jxucoder Sep 7, 2026
1498aad
Compose resident Normal joint and ordered boosting recipes
jxucoder Sep 7, 2026
ac7b1e1
Add external device cohort learner and Normal replay checks
jxucoder Sep 7, 2026
a5de692
Freeze bounded Normal and installed D2 CUDA run six
jxucoder Sep 7, 2026
4143d18
Record approval for frozen Normal CUDA run six
jxucoder Sep 7, 2026
ad9792f
Archive Normal CUDA run six with two acceptance failures
jxucoder Sep 7, 2026
abd5b1a
Record Normal CUDA retrospective and acceptance follow-up
jxucoder Sep 7, 2026
ba46b3a
Add independent Normal acceptance difference oracle
jxucoder Sep 7, 2026
e0a043b
Capture original Normal acceptance failures without changing decisions
jxucoder Sep 7, 2026
1e0acfb
Freeze run seven for original Normal acceptance diagnostics
jxucoder Sep 7, 2026
80740f2
Record approval for frozen Normal acceptance run seven
jxucoder Sep 7, 2026
6f204de
Archive Normal run seven with measured acceptance failure traces
jxucoder Sep 7, 2026
e5abf3c
Record acceptance retrospective and propose stable comparison boundary
jxucoder Sep 7, 2026
8d0e92f
test: bound Normal loss changes with independent interval arithmetic
jxucoder Sep 7, 2026
a5967b5
test: preregister Normal comparison cohorts and historical case mapping
jxucoder Sep 7, 2026
a636066
docs: archive bounded Normal comparison evidence and reflect on 092-A
jxucoder Sep 7, 2026
51b8a96
feat: expose bounded Normal loss changes on CPU
jxucoder Sep 7, 2026
cf8d189
feat: add resident objective loss comparison with explicit device bounds
jxucoder Sep 8, 2026
4e881ce
docs: specify separate loss-comparison consumers and snapshot ownership
jxucoder Sep 8, 2026
5b11828
Use objective comparisons for CPU Normal consumers
jxucoder Sep 8, 2026
b016b24
Own best validation anchors in objective-mode device runs
jxucoder Sep 8, 2026
2c5927c
Use separate objective comparisons throughout resident Normal recipes
jxucoder Sep 8, 2026
c881259
Bind all historical requirements to revised comparison cohorts
jxucoder Sep 8, 2026
dbeff8a
Freeze separate historical and revised CUDA comparison cohorts
jxucoder Sep 8, 2026
8a70e0b
Document bounded run-8 request and retrospective boundary
jxucoder Sep 8, 2026
af8ad2a
Review foundation progress and prioritize author and workload evidence
jxucoder Sep 8, 2026
0a85320
Add standalone D1 and D2 development verifiers
jxucoder Sep 8, 2026
44d516f
Archive installed D1 and D2 verifier evidence
jxucoder Sep 8, 2026
40dd684
Refresh author packets and add rejection and isolation probes
jxucoder Sep 8, 2026
862c404
Verify the package typing marker in author wheels
jxucoder Sep 8, 2026
424ebdc
Archive current author packet and failed native isolation evidence
jxucoder Sep 8, 2026
cda1947
Prepare an isolated Linux author-worker smoke
jxucoder Sep 8, 2026
251edb4
Archive clean Linux worker preparation checks
jxucoder Sep 8, 2026
7985645
Record approval for the frozen Linux CPU smoke
jxucoder Sep 8, 2026
3eb37f6
Retain failed Linux worker identity smoke and retrospective
jxucoder Sep 8, 2026
518eccf
Enforce worker identity before executing the Linux smoke
jxucoder Sep 8, 2026
aa1664d
Archive passing Linux identity smoke and retrospective
jxucoder Sep 8, 2026
71ef69e
Add fail-closed author request accounting and local deadlines
jxucoder Sep 8, 2026
67ecc94
Reconcile cancelled background requests without releasing stopped work
jxucoder Sep 8, 2026
68951ef
Freeze bounded provider accounting smoke and acceptance criteria
jxucoder Sep 8, 2026
5c0f31a
Authorize the frozen provider accounting smoke
jxucoder Sep 8, 2026
421ea54
Archive real token-cap evidence and unexercised cancellation result
jxucoder Sep 8, 2026
2e446d6
Add active-response cancellation and freeze one-request smoke
jxucoder Sep 8, 2026
f7f3c60
Defer agent evaluation and prioritize foundation execution
jxucoder Sep 8, 2026
b58b168
Authorize the frozen Normal CUDA validation checkpoint
jxucoder Sep 8, 2026
dffcfd5
Restore pending GPU authorization after approval review block
jxucoder Sep 8, 2026
469ca0e
Record explicit approval for the frozen T4 validation
jxucoder Sep 8, 2026
c36f96a
Archive Normal CUDA run 8 and diagnose the no-op prefix assertion
jxucoder Sep 8, 2026
6026ebb
Correct forward no-op best-prefix expectation and verify CPU semantics
jxucoder Sep 8, 2026
7fe4937
Prepare bounded CUDA recipe revalidation packet
jxucoder Sep 8, 2026
a7173d9
Record explicit approval for frozen CUDA recipe run 9
jxucoder Sep 8, 2026
c84d566
Archive passing CUDA recipe run and close bounded comparison coverage
jxucoder Sep 8, 2026
8677bba
Freeze early squared and Normal CPU CUDA performance checkpoint
jxucoder Sep 8, 2026
b9790fe
Record upload review block and preserve pending performance packet
jxucoder Sep 8, 2026
c8f7ebc
Record explicit approval for frozen Modal performance upload
jxucoder Sep 8, 2026
840cc41
Archive early performance results and plan parallel validation
jxucoder Sep 8, 2026
680bf84
Preserve exact benchmark inputs and complete per-fit evidence
jxucoder Sep 8, 2026
8db2aef
Parallelize CUDA field validation with one cooperative block per column
jxucoder Sep 8, 2026
2101fd7
Freeze bounded original and candidate CUDA validation comparison
jxucoder Sep 8, 2026
dd84247
Approve the frozen run-11 source upload and T4 invocation
jxucoder Sep 8, 2026
346dee5
Archive passing T4 validation optimization and sprint retrospective
jxucoder Sep 8, 2026
270fe4b
Freeze binary and Poisson device objective contracts
jxucoder Sep 8, 2026
bed9605
Add resident binary and Poisson objective components
jxucoder Sep 8, 2026
5943f4f
Preserve device class schemas and add GLM round conformance cases
jxucoder Sep 8, 2026
95b5232
Freeze convex GLM loss-change mathematics and counterexamples
jxucoder Sep 8, 2026
67788ba
Add bounded resident binary and Poisson loss comparisons
jxucoder Sep 8, 2026
703d9bd
Integrate compared scalar binary and Poisson recipes
jxucoder Sep 8, 2026
3f90252
Freeze bounded GLM CUDA validation and retained evidence
jxucoder Sep 8, 2026
fe12beb
Record approval of the bounded run-12 GLM validation
jxucoder Sep 8, 2026
c965cb6
Retain passing GLM T4 validation and complete sprint 108 retrospective
jxucoder Sep 8, 2026
7b18667
Prepare foundation PR with reproducible CPU CI and strict docs
jxucoder Sep 8, 2026
8df852c
Test current Normal comparisons without altering frozen cohorts
jxucoder Sep 8, 2026
bbd69ea
Separate frozen report integrity from bounded numerical replay
jxucoder Sep 8, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
6 changes: 4 additions & 2 deletions .github/workflows/gpu-tests.yml
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
name: v1 GPU Verification Pending
name: v1 GPU Validation Requires an Approved Packet

on:
workflow_dispatch:
Expand All @@ -12,5 +12,7 @@ jobs:
steps:
- name: Explain current v1 boundary
run: |
echo "No v1 CUDA implementation exists. Restore real-device verification at F3; reference tests are not GPU validation."
echo "Experimental CUDA components and recipes have committed real-device evidence."
echo "This hosted CPU job performs no GPU validation. Use a frozen packet and its separate hardware/upload allowance."
echo "See v1-sprints/README.md for the current evidence and execution boundary."
exit 1
8 changes: 6 additions & 2 deletions .github/workflows/unit-tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -10,17 +10,21 @@ jobs:
references:
runs-on: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest, macos-latest]
python-version: ["3.10", "3.12"]
steps:
- uses: actions/checkout@v4
with:
# Historical execution SHAs are part of the committed evidence contract.
fetch-depth: 0
Comment on lines 18 to +21

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Badge Preserve execution commits in reachable history

A fresh checkout of this commit cannot recover the execution commits merely by setting fetch-depth: 0: 9da718c has only parent 47108db, and its reachable history excludes the manifest revisions fe12beb and dd84247. The non-GPU CI command nevertheless runs test_glm_freeze.py and test_validation_freeze.py, which call git show on those revisions, so these evidence checks fail once the separate development branch is unavailable. Merge the execution history as parents, retain reachable refs, or make the verifiers use committed source archives instead.

AGENTS.md reference: AGENTS.md:L151-L155

Useful? React with 👍 / 👎.

- uses: actions/setup-python@v5
with:
python-version: ${{ matrix.python-version }}
- uses: astral-sh/setup-uv@v4
- run: uv sync --locked --extra test
- run: uv run --no-sync ruff check src/openboost tests/v1 tests/conftest.py
- name: Run current v1 references, not retired production tests
run: uv run --no-sync pytest tests/ -n 0 -q
- name: Run current CPU correctness and offline evidence checks
run: uv run --no-sync pytest tests/ -m "not gpu and not benchmark" -n 0 -q
- run: uv build
330 changes: 276 additions & 54 deletions AGENTS.md

Large diffs are not rendered by default.

60 changes: 50 additions & 10 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -23,17 +23,37 @@ these operations and persist validated numeric/categorical trees. The first comp
and [Normal recipe](docs/v1/normal.md) support weights, offsets and
fixed/backtracking steps on CPU. Normal exposes ordinary/Fisher directions and
joint mean/log-scale updates. [Formula and sequential runs](docs/v1/formula-runs.md)
add structured full-metric updates and independent heterogeneous jobs. CUDA
execution is not implemented yet. [Binary classification](docs/v1/binary.md) now
add structured full-metric updates and independent heterogeneous jobs. Experimental
resident scalar CUDA training passes all 212 bounded T4 checks, including
weighted/missing parity, transactions and saved CPU inference. The shared scoring
correction resolves the previous 14 failures without changing tolerances. See the
[recorded result and scope](benchmarks/v1/evidence/cuda-score-symmetry-089/README.md).
The first [Normal K=2 and installed D2 T4 run](benchmarks/v1/evidence/cuda-normal-090/README.md)
passes 381/383 checks, including all earlier scalar cases and nineteen saved-model
CPU replays. Two ordered acceptance decisions fail the frozen reference; full
Normal conformance remains open. The [follow-up diagnostic run](benchmarks/v1/evidence/cuda-acceptance-091/README.md)
preserves both failures and identifies rounding-induced false improvement at
near-stationary loss. The subsequent objective-owned comparison correction and
[bounded revalidation](benchmarks/v1/evidence/cuda-recipe-103/README.md) establish
514 earlier passes plus fifteen new recipe passes with identical production.
All 529 revised requirements have passing evidence across two executions;
the original failed verdicts remain preserved. This is not full Normal conformance.
[Binary classification](docs/v1/binary.md) now
persists typed class order and exposes probability/label inference.
[Multiclass and vector leaves](docs/v1/multiclass.md) add joint softmax updates
and separate split/leaf statistics with arbitrary output mappings.

Independent references and comparator/data checks remain evaluation preparation.
F0.3 is still open; the user approved overlapping B03–B06 construction without
removing any v1 scope or acceptance requirements. No real quality, GPU performance
removing any v1 scope or acceptance requirements. No real quality, competitive GPU performance
or agent/adoption advantage has been established for the new foundation.

The [early same-host performance checkpoint](benchmarks/v1/evidence/early-performance-104/README.md)
measures squared-error boosting at 10,000 rows in 7.04 s on CPU and 2.94 s on a
warm T4, with comparable quality. The four other timing pairs and the separate
profile are incomplete after deadlines. This synthetic internal result establishes
neither external-library speed parity nor practical performance across all recipes.

- [Execution and reflections](v1-sprints/README.md)
- [Construction design](planning/foundation-construction-design.md)
- [v1 plan](planning/agent-boosting-foundation-plan.md)
Expand All @@ -48,14 +68,16 @@ and formula models, and train-many each need their own implementation and eviden

```bash
uv sync --extra test
uv run pytest tests/ -n 0 -q
uv run pytest tests/ -m "not gpu and not benchmark" -n 0 -q
uv run ruff check src/openboost tests/v1 tests/conftest.py
uv build
```

Python 3.10+. Current tests are CPU-only reference checks. CUDA is a future
required execution subset, not an implemented capability of this reset checkout.
GPU and publishing workflows stay unavailable until their v1 gates are met.
Python 3.10+. Current tests cover CPU implementation, independent references and
evaluation infrastructure. Experimental CUDA storage, named fields and histograms
and bounded resident squared training have real T4 evidence. The full required
device recipe, quality and cost gates remain open. Run GPU-marked tests only on
real hardware; publishing remains separate.

## Historical implementation and evidence

Expand Down Expand Up @@ -91,9 +113,9 @@ and persists two-model inference with explicit output units. Real A9 evaluation
[Log-normal AFT](docs/v1/aft.md) adds event/right-censored CPU training and
persisted scale-aware survival outputs. Real A10 evaluation remains open.

[Current CPU coverage audit](v1-sprints/035-cpu-coverage-audit.md) identifies
external author workflows and real-data integration as remaining CPU
prerequisites; Sprint 036 supplies the audited A6 recipe/scaling gap.
[Current execution and reflections](v1-sprints/README.md)
separates implemented CPU coverage from remaining authoring, practical execution,
real selection and GPU evidence.

[Multi-output squared regression](docs/v1/multioutput.md) supports independent/shared trees,
projected splits and persisted training-only target scaling. Real A6 evaluation remains open.
Expand All @@ -102,3 +124,21 @@ projected splits and persisted training-only target scaling. Real A6 evaluation
across independent jobs, verified at M=1/8/32.
[Independent stopping](docs/v1/stopping.md) adds validation patience to every CPU
recipe while keeping model acceptance and best-model selection independent.


[Experimental CUDA operations](docs/v1/execution.md) provide context-owned buffers,
named fields, once-only weighting, routed histograms, candidate scores, composable
feasibility masks, routing and scalar leaves, with
[88 passing real T4 checks](benchmarks/v1/evidence/cuda-splits-078/README.md).
Independent cohort constraints change split selection through the public device
operations. Separate experimental resident squared geometry, scalar trees and
accepted/proposal training now pass the separate
[212-case T4 matrix](benchmarks/v1/evidence/cuda-score-symmetry-089/README.md).
Shared mapped transactions, Normal geometry and joint/ordered recipes now have
[bounded passing comparison and recipe evidence](benchmarks/v1/evidence/cuda-recipe-103/README.md),
with the historical numerical failures preserved in the earlier archives.
The installed D2 learner uses the same public field/feasibility/tree operations.
Binary/Poisson objectives, numerical loss-change comparisons and the shared scalar
recipe now pass [153 GLM T4 checks plus 418 regressions](benchmarks/v1/evidence/cuda-glm-108/README.md),
with retained input bytes, class-aware inference and 32 audited final/best models.
Other required CUDA recipes and full phase acceptance remain open.
106 changes: 106 additions & 0 deletions benchmarks/v1/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -608,6 +608,26 @@ release. All trials and failures are retained. It scores no test labels and is
not the frozen real-data quality grid, a speed benchmark or fused train-many.
A6 final comparative quality reporting and the remaining real searches remain open.

The same smoke accepts `--protected` on Linux with a root evaluator. It places
protocol, validation/scale records, test features and the selection receipt under
an evaluator-owned mode-0700 directory. Workers receive read-only train/validation
packets and job files, run as UID/GID 65534 with an 8-GiB address limit and a
1800-second deadline, and retain the current worker's one-thread contract. Completed
trial directories are reclaimed by root before the next trial starts. Installed
Python/package/source paths and output ancestors must be traversable by that UID.
Unsupported hosts fail before creating output; there is no privilege fallback.

The [Linux selection integration](evidence/protected-selection-070/README.md) passes
all three focused tests and a retained 16-trial run. The saved Linux receipt now has a macOS replay regression. Score-only
recomputation permits at most eight times the smaller binary64 spacing of each
finite value. All other receipt fields and the selected winner remain exact;
changes to ordering across a near tie still reject release. The original receipt
byte pin and every artifact hash remain exact. This bounded rule does not promise
portability across arbitrary numerical libraries or metric changes.
It is a synthetic four-round grid, not the full 300/1000-round search. The earlier
standalone permission probe does not validate this call path. Network/new-session
restrictions and separate author containers remain outside this mode's guarantee.

### A6 paired quality reporting

A6 quality cells require all `rmse_k` primary metrics followed by
Expand Down Expand Up @@ -648,3 +668,89 @@ remain pending. The smoke accepts explicit A2/A3 selections, with seven classes
for Covertype. Defaults remain A1/A6/A11 to avoid silently broadening existing
runs. These four-round probes establish no classification quality/calibration,
search, CUDA or speed claim.

## Evaluator-owned execution freeze (Sprint 070)

For a pinned execution, provide a manifest independently frozen by the evaluator:

```bash
uv run --no-sync python -m benchmarks.v1.judge /path/to/producer-run \
--frozen-manifest /path/to/evaluator/frozen.json --frozen-sha256 EVALUATOR_PINNED_FILE_SHA256
```

Both options are required together. The file must be outside the producer run
directory and match the evaluator-supplied raw-byte hash. Strict JSON parsing and
manifest validation apply to both inputs. The producer manifest must match the
entire evaluator manifest, including the expected cells, protocol and provenance.
Rehashing a producer-shrunken matrix cannot satisfy the independent freeze.
The Python API accepts `frozen_manifest=` from a trusted caller; its reported
`frozen_manifest_sha256` hashes canonical JSON and can differ from the CLI raw-file pin.

Without this argument, integrity remains relative to the producer's declared
matrix and `frozen_manifest_match` is null. A true match does not validate the
experimental design, authenticate provenance, prove all R/C/A/E obligations or
establish quality. `gate_results` stays empty. The evaluator must control the pin,
invocation and reference file. An outside-directory check is not an OS permission
boundary: process/container isolation remains separate work in Sprint 070.

### Current OpenBoost trial retention

`openboost_worker` explicitly runs its A1–A12 recipe and independent A5 quantile
paths with `retention="summary"`, reporting `diagnostic_retention` in training
metadata. This is a worker policy, not a new search hyperparameter or a change to
the public recipe default. Frozen model configurations, selected-model semantics
and prediction artifacts are unchanged. It reduces stored round arrays without
qualifying the full search's resource or quality gate. Other worker families must
be audited separately before large jobs.

### Explicit Linux worker identity and address limits

`process_runner.execute(..., address_limit_bytes=8 * 1024**3, unprivileged=True)`
requires a Linux root evaluator. It launches with hard/soft RLIMIT_AS limits,
UID/GID 65534, no supplementary groups, no_new_privs and a minimal explicit
environment. The fresh output directory belongs to that worker; its ancestors
must permit traversal. Evaluator-private inputs need separate root ownership and
permissions. Unsupported setup is rejected; there is no advisory fallback.

The runner records actual launch commands, configured limits, identity/environment
policy, logs and errors. It kills remaining same-process-group descendants before
artifact inspection in this mode. This is not a general hostile-code sandbox:
new sessions and network are not restricted. Use separate containers for independent
attempts and never share same-UID outputs across them. RLIMIT_AS limits virtual
address space, not measured resident memory; enclosing-container policy is separate.
Default execution keeps the existing inherited-identity/environment behavior.

`python -m benchmarks.v1.access_preflight /tmp/fresh-output` runs a bounded Modal
CPU probe with synthetic protected fixtures; it uploads no real datasets or sealed
tasks. Actual permission/resource errors remain distinct from the earlier injected
judge statuses. Passing this probe does not qualify full-search or author-eval gates.

### Full A6 OpenBoost resource planning

`python -m benchmarks.v1.a6_preflight_plan OUTPUT.json` compiles the frozen
300/1000-round configurations into 160 OpenBoost jobs (shared/independent topology,
five folds, sixteen configurations each). It writes once and launches nothing.
The plan records source-freeze hashes, fit-only upper bounds, first resource probes
and remaining requirements. It does not represent the full comparator matrix or
a completed full-search resource check. See `v1-sprints/070-a6-resource-plan.json`.

### Explicit comparator bin budgets

The numeric baseline worker accepts optional `config.bins`, an integer in [2, 256].
XGBoost and LightGBM receive `max_bin=bins`; CatBoost receives
`border_count=bins-1`, since that parameter counts split borders rather than
intervals. This applies to the worker's finite encoded inputs; it does not align
native quantization algorithms. NGBoost explicitly rejects this setting. Omitted
bins retain native defaults. `bin_budget_smoke.py` checks installed effective
parameters, stopping records and fresh-process A6 replay at 7 and 255 bins.

`a6_resource_preflight --comparators` runs only the three frozen fold-zero
configuration-00 comparator resource probes. It verifies the updated A6 plan's
input pins, uses the protected worker policy and fresh A6 replay, and stops on
failure without retries. Profile, paired and comparator modes are mutually
exclusive. This mode does not execute or certify the 400-job search.

Add `--comparator-config 5` to that mode for the frozen deeper-tree/1000-round
configuration-05 probes. Only indices 0 and 5 are accepted; early stopping remains
active, so a 1000-round budget need not produce 1000 completed rounds. These
comparator probes do not qualify deeper OpenBoost execution or the full search.
95 changes: 95 additions & 0 deletions benchmarks/v1/a6_preflight_plan.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,95 @@
"""Compile the OpenBoost portion of A6 resource preflight; never launch jobs."""

import argparse
import hashlib
import json
from pathlib import Path

from benchmarks.v1.selection import digest


def compile_plan(design):
if design["schema"] != "openboost-search-design-v1":
raise ValueError("unknown search design")
budgets = design["budgets"]
if any(
budgets[k] != v
for k, v in dict(trial_wall_s=1800, ram_mib=8192, cpu_threads=2, trial_retries=0).items()
):
raise ValueError("resource policy differs from frozen preflight")
if design["selection"]["folds"] != 5 or design["selection"]["trials_per_method"] != 16:
raise ValueError("five folds and sixteen trials required")
configs = design["families"]["xgboost"]
fields = {"rounds", "learning_rate", "max_depth", "reg_lambda", "seed_from_fold"}
if (
len(configs) != 16
or any(set(c) != fields for c in configs)
or len({digest(c) for c in configs}) != 16
or {c["rounds"] for c in configs} != {300, 1000}
or any(c["seed_from_fold"] is not True for c in configs)
):
raise ValueError("complete supported frozen numeric-tree configuration family required")
jobs = [
dict(
id=f"openboost-{mode}:{fold}:{index:02}",
fold=fold,
application="A6",
library="openboost",
device="cpu",
threads=1,
seed=fold,
early_stopping_rounds=design["shared"]["early_stopping_rounds"],
config={**config, "mode": mode, "bins": design["shared"]["bin_budget"]},
)
for mode in ("shared", "independent")
for fold in range(5)
for index, config in enumerate(configs)
]
return dict(
schema="openboost-a6-resource-plan-v1",
search_design_sha256=digest(design),
scope="OpenBoost-only resource planning; not complete comparator coverage or launch authorization",
jobs=jobs,
job_count=len(jobs),
trials_per_fold=32,
maximum_sequential_worker_seconds=len(jobs) * budgets["trial_wall_s"],
maximum_reserved_cpu_seconds=len(jobs) * budgets["trial_wall_s"] * budgets["cpu_threads"],
policy=dict(
timeout_s=budgets["trial_wall_s"],
address_limit_bytes=8 * 1024**3,
requested_container_memory_mib=budgets["ram_mib"],
reserved_cpus=2,
worker_threads=1,
retries=0,
retention="summary",
),
first_probe_ids=["openboost-shared:0:00", "openboost-independent:0:00"],
stop_on_failure=True,
open_requirements=[
"Bind verified real train/validation packets without test material",
"Run exact full-round workers under protected resource policy",
"Complete all required comparator methods and coverage ledger",
"Qualify selected-model release and per-target real quality",
],
)


def main(output):
design_path = Path(__file__).with_name("search-design.json")
plan = compile_plan(json.loads(design_path.read_text()))
plan["input_files"] = {
str(path.relative_to(design_path.parent)): hashlib.sha256(path.read_bytes()).hexdigest()
for path in (
design_path,
design_path.parent / "datasets/preprocessing.json",
design_path.parent / "datasets/parkinsons.json",
)
}
with Path(output).open("x") as stream:
stream.write(json.dumps(plan, indent=2, allow_nan=False) + "\n")


if __name__ == "__main__":
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("output", type=Path)
main(parser.parse_args().output)
Loading
Loading