diff --git a/CLAUDE.md b/CLAUDE.md index ce1b71c3e..73a10fb72 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -106,3 +106,9 @@ with the journal pointing to them. Update this guide in the same PR whenever the workspace layout, test commands, or release flow change. If you find it contradicting the repo, trust the repo and fix this file. + +UK size experiments use `tools/build_uk_rowwise_candidate.py --dataset-households` +with the same pool inputs as the dense candidate. The flag changes exported +support, not clone K. Sizes remain candidate-only until their matched comparison +and promotion scorecard are adjudicated; see +[the size plan](docs/uk-dataset-size-plan-355.md). diff --git a/changelog.d/uk-dataset-sizes-355.added.md b/changelog.d/uk-dataset-sizes-355.added.md new file mode 100644 index 000000000..eef55e134 --- /dev/null +++ b/changelog.d/uk-dataset-sizes-355.added.md @@ -0,0 +1,6 @@ +Add exact-household-count UK rowwise candidates using contribution-informed L0 initialization, protected target carriers, fixed-size sampling and refitting. Preserve the dense pool doctrine and local gates, export compact linked entities, and record selection provenance without promoting candidates to the certified dense release or changing dataset defaults. +A size run also ships the dense joint solve it was cut from (`dense_reference_diagnostics.csv`, a `dense_reference` summary under `solve.dataset_size`) and the selection itself (`dataset_size_selection.csv`: pool row, design weight, inclusion probability, Horvitz–Thompson baseline, refit weight), and `--selection-seed` re-draws the selection on one pool and one dense reference. The exact-count draw's certainty threshold is a recorded candidate-run knob (`--selection-pi-hi`, default 1), and a refused draw carries the measured gate mass and the feasible alternatives instead of a bare error. +Compose each bound PIPR mean monthly private rent into an annual total using 12 months and the matching A17-uprated private-renter household count, with the inputs and adjudication recorded in the cross-grain receipt. +The size selection's L0 budget search stops only on a probe whose gate probabilities admit the exact-count draw at the requested certainty threshold (`feasible_draw_pi_hi`, verdicts from the new `exact_k_design_feasibility`), recording every probe in the size receipt; a size run checkpoints the dense solve and the search before the draw (`size_selection_checkpoint.{npz,json}`) and `--resume-size-checkpoint` continues at the draw on the re-derived pool after verifying the checkpoint's identity, rebuilding both results through the new `rebuild_calibration_result`. +The UK doctrine solve reports progress through the calibrator's callback seam: a loss line every hundred epochs of the dense solve, each budget probe and the refit, one line per finished probe with its drawability verdict (new `budget_probe` and `budget_search_done` events), and one when the search stops; the candidate driver writes them to stderr. +After review: the checkpoint identity carries the solve doctrine and a resume asserts the restored options against it; a stale checkpoint in `--out` refuses before the solve; `microcosm.calibrate.gates` and `microcosm.calibrate.initialization` join the attested seed-protocol sources and the pre-best-iterate oracle runs with a frozen gate module; the size receipt reports `certainty_share`, `boundary_draws` and `zero_target_rows`; the manifest distinguishes the realized stretch against the refit's reference from the stretch against the pool design. diff --git a/docs/evidence/spec-engine/us-f0-coverage.json b/docs/evidence/spec-engine/us-f0-coverage.json index a8d2d977f..45e73dcbb 100644 --- a/docs/evidence/spec-engine/us-f0-coverage.json +++ b/docs/evidence/spec-engine/us-f0-coverage.json @@ -736,8 +736,8 @@ "legacy_sinks": [], "mode": "compiler_semantic", "pointer_class": "all", - "pointer_count": 824, - "pointer_sha256": "7537385c3fd399a2dbb7dcd8ed7cf1ff2481ed336db621eafcbfd741d5792f40", + "pointer_count": 826, + "pointer_sha256": "7ff2d5d1c2fd8026d17a57244f969dc0e9625a9e47304b693ad15df329041282", "rationale": null, "relative_sink_prefix": null, "source_prefix": "/resolved/seed_protocol", @@ -772,21 +772,21 @@ "verifier": "vintages" } ], - "configuration_field_count": 42154, - "consumed_field_count": 42154, + "configuration_field_count": 42156, + "consumed_field_count": 42156, "generation0_effect_counts": { "legacy_behavior": 38476, - "no_generation0_effect": 3678 + "no_generation0_effect": 3680 }, "mode_counts": { - "compiler_semantic": 27715, + "compiler_semantic": 27717, "front_end_validation": 348, "identity_only": 103, "legacy_behavior": 13988 }, "multiple_primary_use_field_count": 0, - "pointer_inventory_sha256": "3fc6b9480ea81b9635bd0db56e180c2daf32a5cd2006a70d350586c570f96754", - "resolved_binding_field_count": 9770, + "pointer_inventory_sha256": "2c0423a08dc16bf5f142134804a1a6545f887e32ca5c991c856a2f8bcfaea0f0", + "resolved_binding_field_count": 9772, "unused_field_count": 0 }, "inventory_coverage": { @@ -1656,13 +1656,13 @@ "compiler_ir.node_slices" ], "expected": { - "map_sha256": "c94a5af8eb24866156b8cd4b77b93dc497b7df5b025204fe1022f80b1d8446ea", - "protocol_sha256": "1c53f1d9b3e185a41181fbe861f9b08d56c83880710fd1e18d7a4ceb5d6a354e" + "map_sha256": "87ba50531d9fa6683096ecb39a31655b331ab8acff3a5876c2f62b33562a0885", + "protocol_sha256": "fd3e4b06f11be4e8c13ea19fef9469ab95cbbe3e2dcfce351e860dd3e00709e4" }, "failures": [], "observed": { - "map_sha256": "c94a5af8eb24866156b8cd4b77b93dc497b7df5b025204fe1022f80b1d8446ea", - "protocol_sha256": "1c53f1d9b3e185a41181fbe861f9b08d56c83880710fd1e18d7a4ceb5d6a354e" + "map_sha256": "87ba50531d9fa6683096ecb39a31655b331ab8acff3a5876c2f62b33562a0885", + "protocol_sha256": "fd3e4b06f11be4e8c13ea19fef9469ab95cbbe3e2dcfce351e860dd3e00709e4" }, "status": "covered" }, @@ -1677,7 +1677,7 @@ "compiler_ir.seed_stream_map" ], "expected": { - "implementation_sha256": "1c53f1d9b3e185a41181fbe861f9b08d56c83880710fd1e18d7a4ceb5d6a354e", + "implementation_sha256": "fd3e4b06f11be4e8c13ea19fef9469ab95cbbe3e2dcfce351e860dd3e00709e4", "protocol": "legacy-v1", "streams": [ "build_model", @@ -1698,7 +1698,7 @@ }, "failures": [], "observed": { - "implementation_sha256": "1c53f1d9b3e185a41181fbe861f9b08d56c83880710fd1e18d7a4ceb5d6a354e", + "implementation_sha256": "fd3e4b06f11be4e8c13ea19fef9469ab95cbbe3e2dcfce351e860dd3e00709e4", "protocol": "legacy-v1", "streams": [ "build_model", @@ -2599,7 +2599,7 @@ "country": "us", "schema_id": "country_spec", "schema_version": 1, - "spec_sha256": "8b1546bc97b540afd525fe88f36a315de18b4d57bd9fab73035dacd8962e302e" + "spec_sha256": "35a02b6b19c921faba1407d441e0b9d9623c496e2cd5b711be014def281a95c6" } }, "report_schema_version": 3, @@ -2609,7 +2609,7 @@ "country": "us", "schema_id": "country_spec", "schema_version": 1, - "spec_sha256": "8b1546bc97b540afd525fe88f36a315de18b4d57bd9fab73035dacd8962e302e" + "spec_sha256": "35a02b6b19c921faba1407d441e0b9d9623c496e2cd5b711be014def281a95c6" }, "status": "pass" } diff --git a/docs/uk-dataset-size-plan-355.md b/docs/uk-dataset-size-plan-355.md new file mode 100644 index 000000000..4b2ce4b21 --- /dev/null +++ b/docs/uk-dataset-size-plan-355.md @@ -0,0 +1,187 @@ +# UK dataset sizes: implementation plan and operating boundary + +This implements the candidate-building part of [#355](https://github.com/PolicyEngine/microcosm/issues/355) +on [#870](https://github.com/PolicyEngine/microcosm/pull/870)'s branch. The PR is still open as of +2026-09-05 and itself stacks on #852. Do not base the work on the old issue's +535,080-household 2023 dataset. The authoritative inputs are the current raw-FRS +2024-25 spine, OA ladder, and pinned Chronicle facts used by the joint candidate. + +## Decisions retained + +- Pool generation keeps clone count **K=15**. Requested output households is a + different parameter, applied after cloning and materialization. +- The joint surface retains national, constituency and local-authority rows, + the declared stretch bound **10**, loss cap **10**, **grain_equal** weighting, + and **1,500 epochs** per solve in the normal driver defaults. +- Existing binding adjudications, signed deferrals, measure exclusions, + census-vintage uprating, and all local gates remain in force. Size selection + does not silently drop target rows, loosen ESS floors or change registers. +- Population-normalized engine measures are frozen from the full pool. The + refit uses their selected household contributions rather than re-running + those formulas on a smaller population. +- The Frame carries complete households, benefit units and people; every + non-dry attempt retains the existing Logbook recording envelope. + +## Implemented sequence + +1. Build the usual joint dense solve from the pinned inputs. This also supplies + the same-target reference loss for the size comparison. +2. Compute each household's maximum absolute target-contribution share from + original pool weights and the compiled sparse matrix, following + [#346's correction](https://github.com/PolicyEngine/microcosm/issues/346#issuecomment-4902880142). + Protect the largest absolute weighted carrier of each nonzero target, with + first-column tie breaking. Initialize open probabilities with + `0.1 + 0.8 * score / (score + median_positive_score)`. This bounded smooth + prior is an implementation choice for candidate evaluation, not a measured + UK release ruling. Run L0 budget search; never select the largest prior scores. +3. Use the existing exact-count Sampford sampler on learned probabilities. + Probability-one gates are certainties (`pi_hi=1`), including every protected + carrier. Refuse impossible budgets or sampling designs rather than clamp. + The certainty threshold is a candidate-run knob (`--selection-pi-hi`, + default 1). The first licensed rehearsal (2026-09-08, 55,000 of 792,690 at + 100 epochs) was refused at the draw: the budget search meets its target on + the count of not-fully-closed gates while the exact-count design can only + draw from the open-probability mass, which fell a fifth short. The refusal + and the size receipt now carry a feasibility measurement (boundary mass and + largest gate, the largest feasible count at `pi_hi=1`, the smallest feasible + threshold on a grid). Ruling (María, 2026-09-08): the smoke accepts the + measured feasible count; the 55,000 candidate runs at `pi_hi=0.95` (the US + exact-k ladder's setting) and 2,000 epochs. A threshold below one promotes + learned near-certain gates and is recorded, never a release default. + + The same 2026-09-08 ruling fixes the local private-rent binding: each 2025 + PIPR calendar-year mean monthly rent is composed into the linear annual + total `12 × mean × renter households`, using the matching authority's + A17-uprated `ons.tenure.private_rent` count. The composed total remains the + bound `rent/private_rent` target, and its inputs ride the cross-grain receipt. +4. Refit on the selected support through `microcosm.calibrate`, with no L0 + penalty. Reuse the existing normalized Horvitz–Thompson `w/q` baseline. + The stretch multiplier remains 10 **relative to that inclusion-adjusted + baseline**; this is the shared exact-k refit's contract, not a claim that + sparse weights remain within 10 times the unexpanded pool-row weights. + The manifest names the reference explicitly. Its empirical suitability for + UK release remains to be assessed alongside the size scorecard. +5. Restore the prepared carrier and export only selected linked entities. + Re-run the existing local gate battery on the compact frame. Each holdout + fold independently reruns selection on training targets; held targets do + not inform the prior or protected set. +6. Record requested/realized counts, pool positions, inclusion probabilities, + protected count, seed, learned penalty, dense/refit loss and target change, + the gate reports, and ordinary byte-pinned output metadata. + +## Running a candidate + +Use the inputs and environment from the existing +[UK dense assembly runbook](uk-dense-release-assembly-runbook-762.md). +Pass the same pinned source arguments to the existing driver and add: + +```bash +uv run python tools/build_uk_rowwise_candidate.py \ + --input-h5 "$UK_SPINE_H5" --input-sha256 "$UK_SPINE_SHA256" \ + --ladder "$UK_LADDER_NPZ" --ladder-sha256 "$UK_LADDER_SHA256" \ + --ledger-facts "$UK_LEDGER_FACTS" \ + --ledger-facts-sha256 "$UK_LEDGER_FACTS_SHA256" \ + --ledger-manifest-sha256 "$UK_LEDGER_MANIFEST_SHA256" \ + --dataset-households 50000 --seed 42 --out out/uk-k50000 +``` + +Repeat with another positive household count and a fresh output directory to +compare sizes. These are requested counts, not certified presets. Omit +`--dataset-households` to retain the existing dense path. Add `--dry-run` to +inspect input binding and parameters without solving or writing a candidate. +Never reduce `--n-clones` to request a smaller output. Full builds still need +the dense build's peak memory and add L0/refit work; the reduction is in the +exported dataset's storage and downstream loading/simulation footprint. + +### The search stops on the draw's own feasibility; the solve is checkpointed before the draw (2026-09-09) + +S2 on spine-p (55,000 at `--selection-pi-hi 0.95`, 2,000 epochs) was refused at the +exact-count draw after 4.8 hours. The mass-basis budget search had stopped inside its +±5% band at an open mass of 54,834, 166 rows *under* the request; the draw's +condition is one-sided (roughly "open mass at least the request", exactly +`(k − certainties) × max(boundary π) ≤ Σ boundary π`), and with near-binary gates the +tail below 0.95 held 291 rows of mass for 437 places. Two changes follow: + +- The size selection's search now stops only on a probe whose gate probabilities + admit the draw at the requested threshold (`calibrate(..., feasible_draw_pi_hi=…)` + on the mass basis; verdicts from `exact_k_design_feasibility`, the draw's own + inequality). An infeasible probe steers the bisection like a count miss (short + boundary mass → smaller penalty, surplus certainties → larger). Every probe and the + reason the search stopped are recorded under `selection_budget_search` in the size + receipt; if no probe is drawable within the ten-probe budget the closest run is + still returned and the draw refuses with its measurement, as before. +- A size run writes `size_selection_checkpoint.{npz,json}` into `--out` after the + dense solve and the search, before the draw (the dense weights and trajectory, the + selection's weights, gate probabilities and search receipt, the protected-carrier + mask, and the identity of the pool, the target surface and the solve settings). + `--resume-size-checkpoint DIR` re-derives the pool and the surface, verifies that + identity, rebuilds both results through `rebuild_calibration_result`, and continues + at the draw; `--selection-pi-hi` may differ from the threshold the search stopped + on and both are recorded (`selection_pi_hi`, `selection_search_pi_hi`). A draw + refusal therefore costs a re-draw, not the pool solve. `--no-size-checkpoint` + opts out. The checkpoint is candidate evidence, never a release input. + +### The solve reports progress (2026-09-09) + +A size run used to be silent between "solving ... under the doctrine..." and its manifest, five +hours later. The doctrine solve now takes a `progress` line sink (the driver writes it to stderr, +so it lands in the run log): a timestamped loss line every hundred epochs and at the last epoch +of the dense solve, of every budget probe and of the refit; one line per finished probe with its +penalty, open mass, certainties, boundary draw and mass, and its drawability verdict +(`budget_probe` events from the search); and one line when the search stops, naming why and +what it selected (`budget_search_done`). Nothing else changes: the events ride the calibrator's +existing `progress_callback` seam. + +### Review round on #877 (Vahid, 2026-09-09) + +Three should-fix items and two questions changed the machinery: + +- **A resume cannot drift from the doctrine.** The checkpoint identity now carries the solve doctrine + (`_doctrine_bounds()`: the stretch multiplier, the loss cap, the rule) beside the pins, seeds and sizes, + and the doctrine solve asserts the restored options (`max_weight_ratio`, `mass`, `target_loss_cap`) + against today's doctrine after the load, refusing by name. The writing run's code pin and build id ride + in the checkpoint's `provenance` and are reported on resume, not compared, so a receipts commit does + not invalidate a checkpoint. +- **A stale checkpoint in `--out` refuses before the solve.** The driver checks for + `size_selection_checkpoint.{npz,json}` beside the other pre-solve output guards; the writer's own + refusal stays as the last line of defence. +- **`gates.py` and `initialization.py` are attested and controlled.** Both join the seed-protocol + implementation digest (`seeds.py`), so a change to gate behaviour or the contribution prior moves the + attested identity; and the pre-best-iterate oracle now runs with a frozen copy of the gate module + (`tests/fixtures/pre_best_iterate/gates_.py`) instead of the live one, so a gate-behaviour change + surfaces as a numeric mismatch on the gated control path rather than being absorbed on both sides. +- **The draw's determinism is stated.** The size receipt carries `certainty_share` and `boundary_draws`: + on the two 55,000 candidates 99.3 % / 99.7 % of the count was decided by the threshold on learned π and + 405 / 156 rows were drawn. The manifest's stretch keys are honest: `realized_max_weight_ratio_vs_stretch_reference` + is measured against the frame the refit started from (the HT baseline on a size run) and + `realized_max_weight_ratio_vs_design` against the pool design weights themselves. +- **Zero-valued targets are counted.** `zero_target_rows` in the size receipt says how many compiled rows + carry a zero target, the case where the contribution prior uses a unit denominator. + +## Certification and publication still required + +The implementation produces **candidates**, not a new certified UK default. +A size request refuses `--release-candidate`, and its manifest records +`releasable=false` even if the diagnostic gate run passes. The dense assembler +must not interpret a compact candidate as the already reviewed dense line. +The existing registry, production pointers and pe.py default are unchanged. + +Before any size can be promoted: + +1. Run the licensed full-input build, retain all gate failures and measure + local ESS and fit. A nominal 50k size is not guaranteed to clear the floors + that motivated K=15. +2. Run #355's matched sound-comparison protocol and the referenced promotion + scorecard, including reform/distributional validation and untargeted bases. + Calibration loss alone is not a certificate. The included same-target loss + comparison is a diagnostic, not a substitute for those protocols. +3. Adjudicate any new size-specific acceptance decisions, including the + inclusion-adjusted stretch reference. A national-only product would need + its own explicit scope; this implementation does not downgrade local claims. +4. Add size-specific certified release identities, assembly contracts and + downstream bundle entries against that evidence; publication remains the + repository's deliberate human step. + +Accordingly, this increment does not close #355's default-flip requirement. +Synthetic CI checks prove code behavior and artifact structure; they do not +establish licensed-data fit, storage measurements, or release eligibility. diff --git a/experiments/355-uk-dataset-size-receipts.md b/experiments/355-uk-dataset-size-receipts.md new file mode 100644 index 000000000..424851239 --- /dev/null +++ b/experiments/355-uk-dataset-size-receipts.md @@ -0,0 +1,732 @@ +# #355 dataset-size receipts: 55,000 households by informed L0 on the #877 machinery + +Plan of record: `repos/uk-355-size-experiment-plan.md` (approved 2026-09-08, two rounds). These +receipts record what was run, on which pinned inputs, and what was measured. They are candidate +evidence only: no size run is releasable (`--release-candidate` is refused with +`--dataset-households`), no gate threshold is loosened, and nothing here promotes an artifact. +Kept separate from the #762 receipts (`experiments/762-uk-rowwise-candidate-receipts.md`). + +Rulings carried in (María, 2026-09-08): K=15 cloning stays in front of the selection (the dense +joint solve runs on the whole 792,690-row pool; the size machinery then keeps 55,000 of those +rows); `--epochs 2000` on size runs (drives the dense solve, every L0 probe and the refit); skip +holdout on the first 55k run; no dense baseline on the new inputs unless the rough comparison with +R17 shows large deviations; engine policyengine-uk 2.94.0; Chronicle feed 6fb700e. + +## Step 0 — #877 CI green before any run + +CI run 34216367053 on 56aa4e25 (the spec-engine re-pin) was red in four jobs with one cause: the +`solve.py` edit in #877 moved two more source-hashed identities. + +| lane | test | pin | fix | +|---|---|---|---| +| `fast (rest)` ×2 | `microcosm-graph/tests/test_acceptance_h_parity.py::test_h1_kernel_parity`, `test_graph_serialize.py::test_generated_parity_graphs_bind_real_kernels_and_direct_bytes` | `calibrate.adam@1` implementation hash a7a0330f… → d24fc41d… (`source_hash` over the calibrate modules) | H1 fixture regenerated on its authoring platform (arm64/darwin/py3.14; numpy 2.4.6, pandas 3.0.3, scipy 1.17.1, torch 2.12.0) with `tools/graph_parity_fixtures.py`; only `calibrate/pins.json` taken (direct bytes and graph unchanged). The generator also resets `fit.qrf/pins.json`'s platform map to the authoring platform, dropping its two linux pins; that file was reverted. | +| `engine-us (us-am)` ×2 | `test_us_multispine_pool_tool.py::test_constants_adapter_equals_live_constants_and_stays_out_of_identities` | US `spec_sha256` 9db29b4d… → 9db2d6db… | re-pinned | + +Commit 1b0583b7. Local: graph 336 passed, multispine pool tool 187 passed, spec-engine pin files 50 +passed. Pushed with C1/C2 below as d4043ec7; CI run 34223080395. + +The run on d4043ec7 then failed in the `wheels` lane only (both Python versions): the size CLI test read +the exported H5 through `pd.HDFStore`, and the wheels venv has no pytables (7,963 passed, 587 skipped, that +one failure). The file's convention is `pytest.importorskip("tables")` + `importorskip("h5py")` at the top of +every CLI test; the size test lacked them, and the two new evaluation test files write PyTables-format H5 +the same way. All three now skip without pytables (verified by blocking the import locally); the workspace +and engine lanes, which have pytables, run them in full. + +The run on fb85c8fe then failed only in the two UK lanes on one new test: the evaluation command timed +each downstream script with `/usr/bin/time -l`, a BSD flag GNU `time` rejects with exit 125, so on the +linux runners a failing stub could not surface its own exit code. Replaced by an in-process rusage shim +(the child script runs in a child interpreter that reports its own peak resident set in bytes on exit; +wall time measured by the caller; exit code the script's own), commit e8f7c6f2, pushed with the +`--selection-pi-hi` knob (a6b1e0ca) and the by-name refusal (b1d206d6); CI run 34239851890. + +## Code landed on the branch before the runs + +- **C1 (d4043ec7)** — a size run keeps the dense joint solve it was cut from: `UKRowwiseDoctrineSolve.dense_reference` (weights, initial weights, local and national diagnostics, losses, past-cap censuses; the evidence labelling moved into `_doctrine_solve_evidence` and runs for both results), written as `dense_reference_diagnostics.csv` (every target's dense estimate with a `grain` column) and summarised under `solve.dataset_size.dense_reference`; and the selection itself as `dataset_size_selection.csv` (`pool_row_index, household_id, clone_index, design_weight, inclusion_probability, certainty, ht_baseline_weight, refit_weight`). Both are listed under `outputs` with digests; dense runs are unchanged. The dense reference is byte-identical to a standalone dense run on the same inputs, seed and epochs, so the size-only delta needs no second full-pool solve. +- **C2 (d4043ec7)** — `--selection-seed` (requires `--dataset-households`, defaults to `--seed`) seeds only the informed L0 search, the exact-count draw and the refit, threaded through the holdout as well; recorded as `parameters.selection_seed` and in the size receipt's `seed`. `--seed` still governs the ladder clone assignment and the dense solve, so two selections compare on one pool and one dense reference. +- **C3/C4 (251e67d4)** — evaluation library `uk_runtime/size_evaluation.py` (`load_run`, `run_acceptance`, `fit_tables`, `weight_tables`, `area_support_tables`, `gate_table`, `paired_targets`, `dense_reference_deltas`, `frozen_vs_recomputed`, `footprint`, `summarize` with `PRE_REGISTERED_OUTCOMES_V1`) and the one command `tools/evaluate_uk_dataset_size.py` (steps `00-run-acceptance`, `10-dense-reference`, `20-vs-reference/