From 43171405d0b506fdeb679aadc0508b7e60928edd Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Wed, 9 Sep 2026 11:48:07 -0400 Subject: [PATCH 01/51] Add source-backed native SPM role enrichment with release verification --- CLAUDE.md | 12 + PROGRESS.md | 26 + README.md | 6 + ...native-spm-role-source-enrichment.added.md | 4 + docs/us-native-spm-role-source-enrichment.md | 186 ++++ .../build/us_runtime/spm_role_source.py | 395 +++++++ .../test_us_spm_role_enrichment_builder.py | 359 +++++++ .../tests/test_us_spm_role_source.py | 263 +++++ .../src/microcosm/data/contract.py | 46 +- .../src/microcosm/data/h5_enrichment.py | 409 ++++++++ .../src/microcosm/data/publish_cli.py | 33 + .../src/microcosm/data/release.py | 16 + .../src/microcosm/data/source_enrichment.py | 978 ++++++++++++++++++ .../tests/test_h5_enrichment.py | 403 ++++++++ .../tests/test_source_enrichment.py | 599 +++++++++++ tools/build_us_spm_role_enrichment.py | 317 ++++++ 16 files changed, 4051 insertions(+), 1 deletion(-) create mode 100644 changelog.d/native-spm-role-source-enrichment.added.md create mode 100644 docs/us-native-spm-role-source-enrichment.md create mode 100644 packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py create mode 100644 packages/microcosm-build/tests/test_us_spm_role_enrichment_builder.py create mode 100644 packages/microcosm-build/tests/test_us_spm_role_source.py create mode 100644 packages/microcosm-data/src/microcosm/data/h5_enrichment.py create mode 100644 packages/microcosm-data/src/microcosm/data/source_enrichment.py create mode 100644 packages/microcosm-data/tests/test_h5_enrichment.py create mode 100644 packages/microcosm-data/tests/test_source_enrichment.py create mode 100644 tools/build_us_spm_role_enrichment.py diff --git a/CLAUDE.md b/CLAUDE.md index ce1b71c3e..7bc947371 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -80,6 +80,18 @@ build recorded staging telemetry that never reached its repo publishes without the flag. Never publish or promote artifacts as a side effect of another task. +The US native-SPM-role source-enrichment lane is a separate release type: +`tools/build_us_spm_role_enrichment.py` creates a local candidate from the exact +reviewed BuildP parent, preserving original variables and inherited schema-5 +calibration evidence. It does not run calibration or relax schema 6 for ordinary +releases. `microcosm.data.source_enrichment` validates candidates and records +actual native-loader compatibility in a separate bundle. The regular publisher +requires `--parent-h5` and the four tested country/Core/wrapper/calculator wheels; `--preflight-only` +runs its real contract without publication. See +[the source-enrichment runbook](docs/us-native-spm-role-source-enrichment.md). +Root's canonical-model acceptance and publication authorization remain separate +from this producer-native-input receipt. + A US release or release-gate preflight that receives a multispine pool through `--base-h5` must authenticate its sibling terminal manifest. A current stacked pool whose terminal battery is red remains fail-closed unless the operator diff --git a/PROGRESS.md b/PROGRESS.md index e528270f7..97313a628 100644 --- a/PROGRESS.md +++ b/PROGRESS.md @@ -1,3 +1,29 @@ +# Native SPM role source enrichment — 2026-09-09 + +Source implementation prepared for a draft PR. The candidate was built locally; +no new H5, release tag, calibration or production deployment was published. +This journal records development evidence; check GitHub for current PR state. + +The new source-enrichment contract reconstructs the independence role from +complete pinned ASEC files, verifies all 166,321 BuildP people and 59,900 units, +and preserves existing native HDF arrays, membership and weights. Calibration +evidence remains the pinned parent's schema 5; ordinary releases still require +schema 6. The source and HDF builder passed 393 focused tests; the subsequent +four-wheel ownership correction passed 337 relevant tests and 17 real runtime +checks. Full wrapper qualification is pending its native household-weight loader +correction. No compatibility or external-publication success is fabricated. + +The candidate's H5 SHA256 is +`6496cc4393d4d3c6574f76eca231de5898c803b9067645591fd5c4d3e65aee84`. +See [the runbook](docs/us-native-spm-role-source-enrichment.md) for exact source +and qualification contracts. Independent code/Fable review, clean-source +qualification, actual registry evidence and numerical acceptance precede promotion. + +--- + +> Historical journal below, retained on 2026-09-09. Earlier State/Next claims +> describe prior work and are not current instructions or branch status. + # F1 portable worker identity — CI crawl fix ## State diff --git a/README.md b/README.md index 964fb2cdb..12c9c1bde 100644 --- a/README.md +++ b/README.md @@ -111,6 +111,12 @@ normal `uv run pytest` suite; the real-H5 mode above is a local/runbook step. ## Releasing & alerts +The [native SPM role source-enrichment lane](docs/us-native-spm-role-source-enrichment.md) +creates a new US H5 from the exact reviewed BuildP parent, preserves its original +variables and schema-5 calibration evidence, and requires fresh country/wrapper +compatibility checks. It has a local candidate builder and uses the regular +publisher's contract with `--parent-h5` and `--preflight-only`. + Standard publication uploads the locally built `releases//` artifacts to the Hugging Face dataset, tags the release, and updates `latest.json`. It runs on the build machine (it needs the freshly built H5), so it isn't a CI step: diff --git a/changelog.d/native-spm-role-source-enrichment.added.md b/changelog.d/native-spm-role-source-enrichment.added.md new file mode 100644 index 000000000..92eb8e623 --- /dev/null +++ b/changelog.d/native-spm-role-source-enrichment.added.md @@ -0,0 +1,4 @@ +Add a reproducible native SPM role enrichment of the reviewed US BuildP H5, +with exact preservation of existing variables, pinned Census reconstruction, +explicit inherited schema-5 calibration evidence, and replayed country/wrapper +compatibility gates in the release publisher. diff --git a/docs/us-native-spm-role-source-enrichment.md b/docs/us-native-spm-role-source-enrichment.md new file mode 100644 index 000000000..aad09b0f8 --- /dev/null +++ b/docs/us-native-spm-role-source-enrichment.md @@ -0,0 +1,186 @@ +# Native SPM role source-enrichment release + +This lane creates a **new H5** from the reviewed BuildP population by adding +`is_spm_independent_minor_role` to its native person table. It performs no +calibration, population aging, weight adjustment, geography assignment, or +membership reconstruction. The country and wrapper use their existing H5 +loaders. The keyed CSV is immutable source evidence, not a runtime join. + +The supported parent is +`populace-us-2024-buildp-sparse-rmloss100-cae8640-20260728T011454Z`, H5 SHA256 +`48b9d479fb4fd1c3537f9383ce4697d130b6f618658409d74f6233c43b994c7e`. +The data contract pins its original release/build manifests, schema-5 +calibration diagnostics, and source coverage evidence by SHA256. This is +explicit inheritance of those measurements. It does not upgrade diagnostics +to schema 6 or assert that they measured a new model's results. Ordinary +calibration releases still require schema 6. A different parent requires a +separately reviewed contract; there is no caller-supplied legacy-schema waiver. + +## Source reconstruction and exact preservation + +`spm_role_source.py` reuses Microcosm's existing Census ASEC archive/member +pins from `education_assistance_source.py`. The complete source person CSVs +are `pppub23.csv`, `pppub24.csv`, and `pppub25.csv` (income years 2022–2024). +The source role is: + +```python +(SPM_HEAD == 1) | (A_FAMTYP.isin([1, 4]) & A_FAMREL.isin([1, 2])) +``` + +The stored Boolean is the role **before** the age gate. Reconciliation uses +`age >= 18 OR (age >= 15 AND role)` to reproduce `SPM_NUMADULTS`, +`SPM_NUMKIDS`, and `SPM_NUMPER` for all 176,039 source units and 59,900 native +BuildP units. This is an inference from documented relationships, exhaustively +checked for these sources; it is not a claim to possess the Census production +program. New source vintages must pass reconciliation independently. + +The join uses income/source year and exact 22-digit string `PERIDNUM`. +Repeated source people across support clones are allowed; missing matches, +ambiguous source IDs, repeated people within one native SPM unit, partial +source units, and inconsistent raw fields are refused. Older missing +relationship cells remain missing. Existing model ages must equal observed +source ages. No weights enter the reconciliation. + +BuildP uses a PyTables compound dataset at `/person/table`. Adding a native +column necessarily extends that compound type; a new unrelated H5 group would +be ignored by the existing country loader. The only permitted physical schema +changes are: + +- Append one HDF bitfield byte (pandas Boolean) after every original record's + bytes. All original field names, order, offsets, types, shapes, and bytes + remain exact, including NaN payloads and signed zero. +- Append the new column's registration to the four `/person` attributes + `data_columns`, `values_cols`, `non_index_axes`, and `info`, and add its five + PyTables field attributes. Every existing registration entry remains intact. + +All other H5 groups, datasets, indexes, types, shapes, attributes, and bytes +are compared against the actual parent. Thus logical pre-existing variable +identity is exact; the enclosing compound datatype is explicitly extended. +The builder copies the parent and does not repack it, so the new file can +retain unused space from the replaced person table. New HDF object timestamps +are disabled for deterministic output within the recorded runtime. + +The generated three-column evidence CSV must also match independently reviewed +SHA256 `22b5968d90fecfeef7614583e493fe10cc16bda8b5be82e6f49a5bc2102d3ce5`. +The CSV is never used to derive the role. Parent and source evidence are copied +unchanged into the release bundle; H5 and evidence files are read-only on exit. + +## Build and validate a local candidate + +Use the repository's installed environment. These commands do not fetch data, +calibrate, or publish. Populate the source cache through the existing pinned +Census source acquisition pipeline, or supply its directory explicitly. +The parent evidence directory contains the original `release_manifest.json`, +`build_manifest.json`, `calibration_diagnostics.json`, and +`us_source_coverage.json`. + +```bash +python tools/build_us_spm_role_enrichment.py \ + --parent-h5 /path/to/certified/populace_us_2024.h5 \ + --parent-release-dir /path/to/buildp-release-source-receipts \ + --source-cache /path/to/pinned/census-person-csvs \ + --reference-evidence-csv /path/to/source-spm-person-independence-private.csv \ + --output-dir /path/to/new-candidate \ + --release-id populace-us-2024-buildp-spm-role-REVIEWED-NEW-ID + +python -m microcosm.data.source_enrichment \ + --release-dir /path/to/new-candidate/releases/RELEASE_ID \ + --parent-h5 /path/to/certified/populace_us_2024.h5 \ + --artifact-root /path/to/new-candidate/artifacts +``` + +The candidate has `release_type: source_enrichment` and +`compatibility.status: pending`. It includes no inherited built-with or model +compatibility claims. An existing output directory is refused, and a failed +build removes its staging output. Its producer receipt identifies the actual +Git commit, dirty status, and source files. Rebuild from the reviewed clean +producer commit before certification; do not edit a dirty receipt to say clean. +The producer also records Python, HDF5, NumPy, pandas, h5py, PyTables, and data +package versions so serialization can be reproduced in the same runtime. + +## Actual compatibility, preflight, and authorized publication + +Install the actual country, Core, wrapper, and `spm-calculator` wheels in an +isolated qualification environment. These may be local candidate wheels; prior +registry publication or wrapper certification on the new H5 is not required. +Certification runs the real country and wrapper +native H5 loaders, verifies the role, IDs, memberships and existing weights, +checks that the country registers a person Boolean variable, and calculates +only the primitive on a small complete native household to prove explicit +inputs override its default formula. It does not run a population simulation. +Installed Python sources must match all four supplied wheels and their versions. +The country registers the native role class supplied by +`spm_calculator/policyengine_adapter.py`; its source is bound to the calculator +wheel, while country and wrapper loader sources bind to their respective wheels. + +```bash +python -m microcosm.data.source_enrichment --certify \ + --release-dir /path/to/new-candidate/releases/RELEASE_ID \ + --output-dir /path/to/certified/releases/RELEASE_ID \ + --parent-h5 /path/to/certified/populace_us_2024.h5 \ + --artifact-root /path/to/new-candidate/artifacts \ + --compatibility-wheel /path/to/policyengine_us-EXACT.whl \ + --compatibility-wheel /path/to/policyengine_core-EXACT.whl \ + --compatibility-wheel /path/to/policyengine-EXACT.whl \ + --compatibility-wheel /path/to/spm_calculator-EXACT.whl + +microcosm-publish-release /path/to/certified/releases/RELEASE_ID \ + --repo-id policyengine/populace-us \ + --parent-h5 /path/to/certified/populace_us_2024.h5 \ + --artifact-root /path/to/new-candidate/artifacts \ + --compatibility-wheel /path/to/policyengine_us-EXACT.whl \ + --compatibility-wheel /path/to/policyengine_core-EXACT.whl \ + --compatibility-wheel /path/to/policyengine-EXACT.whl \ + --compatibility-wheel /path/to/spm_calculator-EXACT.whl \ + --preflight-only +``` + +Certification creates a separate bundle with measured compatibility; it leaves +the candidate H5 and source evidence unchanged. Both the preflight above and +the real publisher invoke the source-enrichment validator and replay the H5 +and compatibility checks before constructing a Hub client. The evidence-tier +publisher cannot be used as an escape hatch. Pending compatibility and a +recorded dirty producer build are hard publication failures. + +Coordinated order: build local candidate wheels, test them against the existing +immutable candidate H5, then let root commit the clean reviewed producer and +create fresh qualification receipts. Root separately verifies external package +publication and numerical/Fable acceptance before data publication and consumer +promotion. A wrapper candidate can load this local H5 before either is published, +so H5 qualification does not depend on an already certified wrapper release. +Qualification retains `external_package_publication: not_attested` and +`scope: native_input_loading_only`; it never substitutes for numerical acceptance. + +These receipts deliberately state that external package publication and +canonical SPM numerical acceptance are **not attested**. Root owns exact Fable +review, canonical calculator/country/wrapper/Axiom parity, and the immutable +published package proof. Once those gates pass and root authorizes publication, +use `tools/publish_release.sh` with the same directory and arguments, remove +`--preflight-only`, and add `--tag-name RELEASE_ID`. The existing publisher +creates the new immutable Hugging Face tag; its standard invocation also +updates `latest.json`. An inspect-only tag uses `--no-latest --tag-only`. +Neither candidate construction nor certification authorizes either mutation. + +Run certification and publisher commands from the Microcosm checkout. Publication +authenticates the six recorded producer source hashes against the recorded Git +commit, the checkout, and the executing data contract modules. A later docs-only +commit is allowed; an invented clean flag, missing commit, changed source, or +older installed producer module is refused. + +## Local regression checks + +```bash +python -m pytest \ + packages/microcosm-build/tests/test_us_spm_role_source.py \ + packages/microcosm-build/tests/test_us_spm_role_enrichment_builder.py \ + packages/microcosm-data/tests/test_h5_enrichment.py \ + packages/microcosm-data/tests/test_source_enrichment.py \ + packages/microcosm-data/tests/test_contract.py \ + packages/microcosm-data/tests/test_release.py \ + packages/microcosm-data/tests/test_publish_guard.py +python tools/ci_test_groups.py --verify +ruff check . +``` + +These tests use synthetic populations. Their success is not certification of +the full private artifact or compatibility with an unreleased model. diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py b/packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py new file mode 100644 index 000000000..0842e34ae --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py @@ -0,0 +1,395 @@ +"""Reconstruct the native SPM role from pinned, complete Census ASEC files. + +This is a source enrichment of an existing population. The raw relationship +rule is a documented-source inference reconciled against Census unit counts; +it is not a claim to have retrieved the Census production program. No model, +weight, calibration, or period-aging operation is performed here. +""" + +from __future__ import annotations + +import hashlib +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +import numpy as np +import pandas as pd + +from .education_assistance_source import ASEC_EDUCATION_ASSISTANCE_ARCHIVES + +NATIVE_SPM_ROLE = "is_spm_independent_minor_role" +EVIDENCE_SPM_ROLE = "is_spm_independence_role" +SPM_ROLE_RULE = "SPM_HEAD == 1 OR (A_FAMTYP in {1,4} AND A_FAMREL in {1,2})" +SPM_ADULT_RULE = "age >= 18 OR (age >= 15 AND is_spm_independent_minor_role)" +_COUNTS = ("SPM_NUMADULTS", "SPM_NUMKIDS", "SPM_NUMPER") +_REQUIRED_RAW_CHECKS = ("A_AGE", "A_LINENO", "P_SEQ", "SPM_HAGE", *_COUNTS) +_OPTIONAL_RAW_CHECKS = ("SPM_HEAD", "A_FAMTYP", "A_FAMREL", "A_SPOUSE", "PECOHAB") +_SOURCE_COLUMNS = ( + "PERIDNUM", + "SPM_ID", + "PH_SEQ", + *_REQUIRED_RAW_CHECKS, + *_OPTIONAL_RAW_CHECKS, +) + + +@dataclass(frozen=True) +class AsecSpmRoleSource: + """Reviewed identity and population coverage of a complete person CSV.""" + + income_year: int + survey_year: int + csv_sha256: str + csv_size_bytes: int + persons: int + units: int + official_archive_url: str + archive_sha256: str + member: str + + +ASEC_SPM_ROLE_SOURCES: dict[int, AsecSpmRoleSource] = { + year: AsecSpmRoleSource( + income_year=year, + survey_year=pin.survey_year, + csv_sha256=pin.member_sha256, + csv_size_bytes=pin.member_size_bytes, + persons=pin.rows, + units={2022: 59_181, 2023: 58_711, 2024: 58_147}[year], + official_archive_url=pin.zip_url, + archive_sha256=pin.zip_sha256, + member=pin.member, + ) + for year, pin in ASEC_EDUCATION_ASSISTANCE_ARCHIVES.items() +} + + +@dataclass(frozen=True) +class SpmRoleSourceResult: + """The primitive in parent row order, keyed evidence, and public counts.""" + + role: np.ndarray + evidence: pd.DataFrame + provenance: dict[str, Any] + + +def _sha256(path: Path) -> str: + with path.open("rb") as stream: + return hashlib.file_digest(stream, "sha256").hexdigest() + + +def _require(condition: bool, message: str) -> None: + if not condition: + raise ValueError(message) + + +def _require_columns(frame: pd.DataFrame, columns: tuple[str, ...], label: str) -> None: + missing = sorted(set(columns) - set(frame.columns)) + _require(not missing, f"{label} missing required columns: {missing}.") + + +def _exact_person_keys(values: pd.Series, label: str) -> pd.Series: + # Converting numeric IDs to text would hide lossy earlier coercions. + decoded = values.map( + lambda value: value.decode("ascii") if isinstance(value, bytes) else value + ) + valid = decoded.map( + lambda value: ( + isinstance(value, str) + and len(value) == 22 + and value.isascii() + and value.isdigit() + ) + ) + _require(bool(valid.all()), f"{label} PERIDNUM must be exact 22-digit strings.") + return decoded + + +def independent_minor_role(persons: pd.DataFrame) -> pd.Series: + """Return the source role BEFORE any age gate (including for adults).""" + _require_columns(persons, ("SPM_HEAD", "A_FAMTYP", "A_FAMREL"), "ASEC role") + return persons.SPM_HEAD.eq(1) | ( + persons.A_FAMTYP.isin((1, 4)) & persons.A_FAMREL.isin((1, 2)) + ) + + +def _reconcile_units(persons: pd.DataFrame, key: str, label: str) -> pd.DataFrame: + groups = persons.groupby(key, sort=False, dropna=False) + _require( + bool(groups[list(_COUNTS)].nunique(dropna=False).eq(1).all().all()), + f"{label} source counts are not constant within SPM units.", + ) + units = groups.agg( + adults=("_adult", "sum"), + expected_adults=("SPM_NUMADULTS", "first"), + expected_children=("SPM_NUMKIDS", "first"), + expected_persons=("SPM_NUMPER", "first"), + persons=("PERIDNUM", "size"), + heads=("SPM_HEAD", "sum"), + age_only_adults=("_age_only_adult", "sum"), + ) + _require( + bool(units.adults.eq(units.expected_adults).all()) + and bool((units.persons - units.adults).eq(units.expected_children).all()) + and bool(units.persons.eq(units.expected_persons).all()), + f"{label} adult/child/person count reconciliation failed.", + ) + _require( + bool(units.heads.eq(1).all()), + f"{label} must have exactly one SPM head per unit.", + ) + _require( + bool(units.adults.ge(1).all()), + f"{label} has an unresolved zero-adult SPM unit.", + ) + heads = persons.SPM_HEAD.eq(1) + _require( + bool(persons.loc[heads, "A_AGE"].eq(persons.loc[heads, "SPM_HAGE"]).all()), + f"{label} source SPM head age disagrees with SPM_HAGE.", + ) + return units + + +def _load_source( + path: Path, pin: AsecSpmRoleSource +) -> tuple[pd.DataFrame, dict[str, Any]]: + label = f"ASEC {pin.survey_year}" + _require( + pin.survey_year == pin.income_year + 1, f"{label} income/survey year mismatch." + ) + _require( + path.stat().st_size == pin.csv_size_bytes, f"{label} CSV byte length mismatch." + ) + _require(_sha256(path) == pin.csv_sha256, f"{label} CSV SHA-256 mismatch.") + source = pd.read_csv( + path, + usecols=list(_SOURCE_COLUMNS), + dtype={"PERIDNUM": str, "SPM_ID": str}, + low_memory=False, + ) + _require(_sha256(path) == pin.csv_sha256, f"{label} CSV changed while reading.") + _require(len(source) == pin.persons, f"{label} CSV person count mismatch.") + source["PERIDNUM"] = _exact_person_keys(source.PERIDNUM, label) + _require(bool(source.PERIDNUM.is_unique), f"{label} has duplicate PERIDNUM keys.") + _require( + bool(source.notna().all().all()), + f"{label} source columns contain missing values.", + ) + _require( + bool(source.SPM_ID.str.fullmatch(r"[0-9]+").all()), + f"{label} SPM_ID must retain exact digit strings.", + ) + numeric = source.drop(columns=["PERIDNUM", "SPM_ID"]).to_numpy(dtype=float) + _require( + bool((np.isfinite(numeric) & (numeric == np.floor(numeric))).all()), + f"{label} raw role/count/identity fields must be finite integers.", + ) + _require( + bool(source.SPM_HEAD.isin((0, 1)).all()), f"{label} SPM_HEAD must be binary." + ) + _require( + bool(source.A_AGE.ge(0).all()), f"{label} source ages must be nonnegative." + ) + source["_role"] = independent_minor_role(source) + source["_age_only_adult"] = source.A_AGE.ge(18) + source["_adult"] = source._age_only_adult | (source.A_AGE.ge(15) & source._role) + units = _reconcile_units(source, "SPM_ID", label) + _require(len(units) == pin.units, f"{label} CSV SPM unit count mismatch.") + source["source_year"] = pin.income_year + source["_source_row_id"] = np.arange(len(source), dtype=np.int64) + extra_minor = source.A_AGE.between(15, 17) & source._role & source.SPM_HEAD.eq(0) + return source, { + "income_year": pin.income_year, + "survey_year": pin.survey_year, + "csv_sha256": pin.csv_sha256, + "csv_size_bytes": pin.csv_size_bytes, + "official_archive_url": pin.official_archive_url, + "archive_sha256": pin.archive_sha256, + "member": pin.member, + "persons": len(source), + "units": len(units), + "adult_child_person_count_mismatch_units": 0, + "extra_independent_minor_people_vs_head_only": int(extra_minor.sum()), + "extra_independent_minor_units_vs_head_only": int( + source.loc[extra_minor, "SPM_ID"].nunique() + ), + } + + +def derive_spm_role_source( + parent_h5: str | Path, + source_paths: Mapping[int, str | Path], + *, + expected_parent_sha256: str, + source_pins: Mapping[int, AsecSpmRoleSource] | None = None, +) -> SpmRoleSourceResult: + """Derive and reconcile from complete pinned CSVs; never trust a sidecar. + + Local CSV paths are mandatory; the existing education-assistance Census + fetcher can obtain their pinned archives separately. The default pins cover + all three certified BuildP source years. Alternate pins support explicit + review of other populations and small synthetic tests. + + Support clones may repeat a source person across native units. Within every + native unit, the source members must be unique and exhaust one complete + Census unit. Original membership and every existing input remain untouched. + """ + path = Path(parent_h5) + _require(_sha256(path) == expected_parent_sha256, "Parent H5 SHA-256 mismatch.") + pins = ASEC_SPM_ROLE_SOURCES if source_pins is None else source_pins + parent = pd.read_hdf(path, "person") + spm = pd.read_hdf(path, "spm_unit") + required = ( + "person_id", + "person_spm_unit_id", + "source_year", + "PERIDNUM", + "age", + "source_household_id", + "source_person_id", + "source_row_id", + *_REQUIRED_RAW_CHECKS, + ) + _require_columns(parent, required, "Parent person table") + _require_columns(spm, ("spm_unit_id",), "Parent SPM table") + _require(len(parent) > 0, "Parent person table is empty.") + _require(bool(parent.person_id.is_unique), "Parent person_id must be unique.") + _require(bool(spm.spm_unit_id.is_unique), "Parent spm_unit_id must be unique.") + _require( + bool(parent[list(required)].notna().all().all()), + "Parent required identity/age/count fields contain missing values.", + ) + _require( + bool(spm.spm_unit_id.notna().all()) + and set(parent.person_spm_unit_id) == set(spm.spm_unit_id), + "Parent SPM membership does not exactly cover its SPM table.", + ) + years = pd.to_numeric(parent.source_year, errors="coerce") + _require( + bool((np.isfinite(years) & years.eq(np.floor(years))).all()), + "Parent source_year must be finite integers.", + ) + _require( + set(years) == set(pins) == set(source_paths), + "Parent income years, pinned sources, and explicit CSV paths must match exactly.", + ) + parent = parent.copy() + parent["PERIDNUM"] = _exact_person_keys(parent.PERIDNUM, "Parent") + sources, source_checks = [], [] + for year in sorted(pins): + _require(pins[year].income_year == year, "Source pin income-year key mismatch.") + source, check = _load_source(Path(source_paths[year]), pins[year]) + sources.append(source) + source_checks.append(check) + source = pd.concat(sources, ignore_index=True) + # Raw SPM_ID was globalized by the pooled-source producer; reconstruct its + # source membership through (income year, PERIDNUM), never compare remapped IDs. + joined = parent.merge( + source, + on=["source_year", "PERIDNUM"], + how="left", + sort=False, + validate="many_to_one", + suffixes=("_parent", ""), + indicator=True, + ) + _require( + bool(joined._merge.eq("both").all()), + "Source join has unmatched parent persons.", + ) + _require( + np.array_equal(joined.person_id.to_numpy(), parent.person_id.to_numpy()), + "Source join changed parent person order.", + ) + for column in _REQUIRED_RAW_CHECKS: + _require( + bool(joined[f"{column}_parent"].eq(joined[column]).all()), + f"Parent raw field {column} disagrees with the pinned Census source.", + ) + optional_coverage = {} + for column in _OPTIONAL_RAW_CHECKS: + if column in parent: + available = joined[f"{column}_parent"].notna() + _require( + bool( + joined.loc[available, f"{column}_parent"] + .eq(joined.loc[available, column]) + .all() + ), + f"Parent raw field {column} disagrees with the pinned Census source.", + ) + optional_coverage[column] = int(available.sum()) + for native, raw in ( + ("age", "A_AGE"), + ("source_household_id", "PH_SEQ"), + ("source_person_id", "PERIDNUM"), + ("source_row_id", "_source_row_id"), + ): + _require( + bool(joined[native].eq(joined[raw]).all()), + f"Parent {native} disagrees with the pinned Census source.", + ) + _require( + not bool( + joined.duplicated(["person_spm_unit_id", "source_year", "PERIDNUM"]).any() + ), + "Native SPM unit repeats a source person.", + ) + native_groups = joined.groupby("person_spm_unit_id", sort=False) + _require( + bool(native_groups[["source_year", "SPM_ID"]].nunique().eq(1).all().all()), + "Native SPM membership combines distinct source units.", + ) + units = _reconcile_units(joined, "person_spm_unit_id", "Parent native SPM") + role = joined._role.to_numpy(dtype=np.bool_, copy=True) + evidence = parent[["person_id", "person_spm_unit_id"]].copy() + evidence[EVIDENCE_SPM_ROLE] = role + minor = joined.A_AGE.between(15, 17) & joined._role + _require( + _sha256(path) == expected_parent_sha256, + "Parent H5 changed while deriving roles.", + ) + return SpmRoleSourceResult( + role=role, + evidence=evidence, + provenance={ + "dataset_sha256": expected_parent_sha256, + "primitive_column": NATIVE_SPM_ROLE, + "evidence_column": EVIDENCE_SPM_ROLE, + "primitive_rule": SPM_ROLE_RULE, + "adult_rule": SPM_ADULT_RULE, + "evidence_status": ( + "Inference from documented primitive relationships, reconciled against " + "every SPM count in pinned complete Census ASEC sources; " + "Census production program not retrieved." + ), + "source_join": [ + "source_year (income year)", + "PERIDNUM (exact 22-digit string)", + ], + "source_checks": source_checks, + "total_source_people": len(source), + "total_source_units": sum(check["units"] for check in source_checks), + "persons_joined": len(joined), + "unmatched_persons": 0, + "all_rows_equal_source_columns": list(_REQUIRED_RAW_CHECKS), + "nonmissing_optional_raw_fields_checked": optional_coverage, + "native_spm_units": len(units), + "complete_source_membership_units": len(units), + "adult_child_person_count_mismatch_units": 0, + "independent_minor_persons": int(minor.sum()), + "classification_changed_units_vs_age_only": int( + units.adults.ne(units.age_only_adults).sum() + ), + "minor_only_units_resolved": int(units.age_only_adults.eq(0).sum()), + "true_role_persons": int(role.sum()), + "table_rows": len(evidence), + "table_key": ["person_id", "person_spm_unit_id"], + "table_order": "exact original HDF person order; verified", + "weights_used": False, + "ages_changed": False, + "period_handling": "Store the role before the age gate; model age supplies the requested period.", + }, + ) diff --git a/packages/microcosm-build/tests/test_us_spm_role_enrichment_builder.py b/packages/microcosm-build/tests/test_us_spm_role_enrichment_builder.py new file mode 100644 index 000000000..620dce135 --- /dev/null +++ b/packages/microcosm-build/tests/test_us_spm_role_enrichment_builder.py @@ -0,0 +1,359 @@ +"""Exercise private candidate assembly through the real enrichment contract.""" + +from __future__ import annotations + +import hashlib +import json +from importlib import metadata +from pathlib import Path +from types import SimpleNamespace + +import numpy as np +import pandas as pd +import pytest + +from microcosm.build.us_runtime.spm_role_source import SpmRoleSourceResult +from microcosm.data import source_enrichment as contract +from microcosm.data.h5_enrichment import file_sha256 +from microcosm.data.publish_cli import _staging_undelivered +from tools import build_us_spm_role_enrichment as builder + +pytest.importorskip("tables") + + +@pytest.fixture +def inputs(tmp_path, monkeypatch): + parent = tmp_path / "parent.h5" + persons = pd.DataFrame( + { + "person_id": [1, 2, 3], + "person_spm_unit_id": [1, 1, 2], + "age": [12.0, 16.0, 45.0], + "person_weight": [2.0, 2.0, 1.0], + } + ) + with pd.HDFStore(parent, "w") as store: + store.put("person", persons, format="table", data_columns=True) + store.put( + "spm_unit", + pd.DataFrame({"spm_unit_id": [1, 2]}), + format="table", + data_columns=True, + ) + parent_sha = file_sha256(parent) + parent_release = tmp_path / "parent-release" + parent_release.mkdir() + parent_payloads = { + "parent_release_manifest.json": { + "schema_version": 1, + "compatible_model_packages": [{"name": "old-model", "specifier": "==0.0"}], + }, + "parent_build_manifest.json": { + "build_id": contract.PARENT_BUILD_ID, + "code": {"git_commit": "parent-only-claim"}, + }, + "calibration_diagnostics.json": {"schema_version": 5, "targets": []}, + "us_source_coverage.json": {"schema_version": 1}, + } + pins = {} + for name, payload in parent_payloads.items(): + path = parent_release / name.removeprefix("parent_") + path.write_text(json.dumps(payload, sort_keys=True) + "\n") + pins[name] = file_sha256(path) + for module in (builder, contract): + monkeypatch.setattr(module, "PARENT_DATASET_SHA256", parent_sha) + monkeypatch.setattr(module, "PARENT_FILES", pins) + roles = np.array([False, True, True], dtype=bool) + evidence = persons[["person_id", "person_spm_unit_id"]].copy() + evidence["is_spm_independence_role"] = roles + reference = tmp_path / "reference.csv" + reference.write_text(evidence.to_csv(index=False)) + evidence_sha = file_sha256(reference) + monkeypatch.setattr(builder, "REFERENCE_EVIDENCE_SHA256", evidence_sha) + monkeypatch.setattr(contract, "SOURCE_EVIDENCE_SHA256", evidence_sha) + counts = { + "persons_joined": 3, + "native_spm_units": 2, + "total_source_people": 3, + "total_source_units": 2, + "minor_only_units_resolved": 1, + "classification_changed_units_vs_age_only": 1, + } + monkeypatch.setattr(contract, "EXPECTED_COUNTS", counts) + monkeypatch.setattr(contract, "CENSUS_PERSON_PINS", {2025: "a" * 64}) + provenance = { + **counts, + "dataset_sha256": parent_sha, + "unmatched_persons": 0, + "adult_child_person_count_mismatch_units": 0, + "complete_source_membership_units": 2, + "weights_used": False, + "ages_changed": False, + "primitive_column": contract.ROLE_VARIABLE, + "evidence_column": contract.EVIDENCE_COLUMN, + "source_checks": [ + { + "income_year": 2024, + "survey_year": 2025, + "csv_sha256": "a" * 64, + "persons": 3, + "units": 2, + "adult_child_person_count_mismatch_units": 0, + "official_archive_url": "https://www2.census.gov/fixture.zip", + "archive_sha256": "b" * 64, + } + ], + } + reconstructed = SpmRoleSourceResult(roles, evidence, provenance) + source_paths = {2024: tmp_path / "synthetic-pinned-source.csv"} + source_paths[2024].write_text("Synthetic source derivation is separately tested.\n") + calls = [] + + def derive(actual_parent, actual_sources, *, expected_parent_sha256): + calls.append((actual_parent, actual_sources, expected_parent_sha256)) + assert actual_parent == parent + assert actual_sources == source_paths + assert expected_parent_sha256 == parent_sha + return reconstructed + + monkeypatch.setattr(builder, "derive_spm_role_source", derive) + code = { + "git_commit": "c" * 40, + "git_dirty": True, + "source_files_sha256": {"synthetic-producer.py": "d" * 64}, + } + monkeypatch.setattr(builder, "_producer_identity", lambda: code) + return SimpleNamespace( + parent=parent, + parent_sha=parent_sha, + parent_release=parent_release, + source_paths=source_paths, + reference=reference, + reconstructed=reconstructed, + output=tmp_path / "candidate", + code=code, + calls=calls, + release_id="populace-us-2024-synthetic-source-enrichment", + ) + + +def _build(inputs, **overrides): + kwargs = { + "parent_h5": inputs.parent, + "parent_release_dir": inputs.parent_release, + "source_paths": inputs.source_paths, + "output_dir": inputs.output, + "release_id": inputs.release_id, + "reference_evidence_csv": inputs.reference, + } + return builder.build_candidate(**(kwargs | overrides)) + + +def _release(inputs): + return inputs.output / "releases" / inputs.release_id + + +def _assert_clean_failure(inputs): + assert not inputs.output.exists() + assert list(inputs.output.parent.glob(".spm-enrichment-*")) == [] + + +def test_builder_assembles_new_candidate_through_real_contract(inputs): + before = inputs.parent.read_bytes() + report = _build(inputs) + release = _release(inputs) + candidate = inputs.output / "artifacts" / "populace_us_2024.h5" + assert inputs.parent.read_bytes() == before + assert inputs.calls == [(inputs.parent, inputs.source_paths, inputs.parent_sha)] + assert report["compatibility"] == {"status": "pending"} + assert report["dataset"]["sha256"] == file_sha256(candidate) + enriched = pd.read_hdf(candidate, "person") + pd.testing.assert_frame_equal( + enriched.drop(columns=[contract.ROLE_VARIABLE]), + pd.read_hdf(inputs.parent, "person"), + ) + assert enriched[contract.ROLE_VARIABLE].tolist() == [False, True, True] + assert enriched[contract.ROLE_VARIABLE].dtype == np.dtype(bool) + assert ( + release / contract.SOURCE_EVIDENCE_FILE + ).read_bytes() == inputs.reference.read_bytes() + for name in contract.PARENT_FILES: + assert (release / name).read_bytes() == ( + inputs.parent_release / name.removeprefix("parent_") + ).read_bytes() + build = json.loads((release / "build_manifest.json").read_text()) + assert build["code"] == inputs.code + assert build["calibration"]["mode"] == "inherited" + assert build["calibration"]["diagnostics_schema_version"] == 5 + assert build["staging"]["enabled"] is False + assert _staging_undelivered(release) is False + manifest = json.loads((release / "release_manifest.json").read_text()) + assert manifest["compatible_core_packages"] == [] + assert manifest["compatible_model_packages"] == [] + assert manifest["build"] == {"build_id": inputs.release_id} + assert manifest["data_package"] == { + "name": "microcosm-data", + "version": metadata.version("microcosm-data"), + } + assert {entry["revision"] for entry in manifest["artifacts"].values()} == { + inputs.release_id + } + for path in ( + candidate, + release / contract.SOURCE_EVIDENCE_FILE, + release / contract.SOURCE_PROVENANCE_FILE, + ): + assert path.stat().st_mode & 0o777 == 0o400 + assert list(inputs.output.parent.glob(".spm-enrichment-*")) == [] + + +def test_reference_csv_is_optional_but_reconstruction_is_mandatory(inputs): + _build(inputs, reference_evidence_csv=None) + assert len(inputs.calls) == 1 + assert ( + _release(inputs) / contract.SOURCE_EVIDENCE_FILE + ).read_bytes() == inputs.reference.read_bytes() + + +def test_repeat_build_has_identical_artifacts_and_receipts(inputs): + first = _build(inputs) + second_output = inputs.output.with_name("candidate-repeat") + second = _build(inputs, output_dir=second_output) + assert first == second + first_files = { + str(path.relative_to(inputs.output)): path.read_bytes() + for path in inputs.output.rglob("*") + if path.is_file() + } + second_files = { + str(path.relative_to(second_output)): path.read_bytes() + for path in second_output.rglob("*") + if path.is_file() + } + assert first_files == second_files + + +@pytest.mark.parametrize( + "release_id", + [ + "wrong-country", + contract.PARENT_BUILD_ID, + "populace-us-2024-sub/path", + "populace-us-2024-sub\\path", + ], +) +def test_rejects_parent_or_unsafe_release_identifier(inputs, release_id): + with pytest.raises(ValueError, match="new bare US 2024"): + _build(inputs, release_id=release_id) + assert inputs.calls == [] + _assert_clean_failure(inputs) + + +@pytest.mark.parametrize("kind", ["directory", "dangling_symlink"]) +def test_existing_output_is_never_replaced(inputs, kind): + if kind == "directory": + inputs.output.mkdir() + (inputs.output / "sentinel").write_text("keep") + else: + inputs.output.symlink_to(inputs.output.parent / "missing-target") + with pytest.raises(FileExistsError, match="already exists"): + _build(inputs) + if kind == "directory": + assert (inputs.output / "sentinel").read_text() == "keep" + else: + assert inputs.output.is_symlink() + assert inputs.calls == [] + + +@pytest.mark.parametrize( + "tamper", ["parent", "parent_evidence", "reference", "reconstruction"] +) +def test_wrong_input_identity_refuses_before_output_creation(inputs, tamper): + if tamper == "parent": + with inputs.parent.open("ab") as stream: + stream.write(b"tamper") + message = "exact reviewed BuildP" + elif tamper == "parent_evidence": + (inputs.parent_release / "calibration_diagnostics.json").write_text( + '{"schema_version":6}' + ) + message = "Parent evidence SHA-256 mismatch" + elif tamper == "reference": + inputs.reference.write_text("wrong independent oracle") + message = "Reference evidence CSV SHA-256 mismatch" + else: + inputs.reconstructed.evidence.loc[0, "is_spm_independence_role"] = True + message = "Reconstructed source evidence differs" + with pytest.raises(ValueError, match=message): + _build(inputs) + _assert_clean_failure(inputs) + + +def test_real_contract_rejects_role_array_disagreeing_with_evidence_and_cleans_up( + inputs, +): + inputs.reconstructed.role[0] = True + with pytest.raises( + ValueError, match="source person evidence does not match native" + ): + _build(inputs) + _assert_clean_failure(inputs) + + +def test_gate_failure_removes_staged_artifacts(inputs, monkeypatch): + def gate(*args, **kwargs): + assert Path(args[0], "build_manifest.json").is_file() + raise ValueError("synthetic final release gate refused") + + monkeypatch.setattr(builder, "validate_source_enrichment_candidate", gate) + with pytest.raises(ValueError, match="synthetic final release gate refused"): + _build(inputs) + _assert_clean_failure(inputs) + + +def test_output_appearing_during_build_is_preserved(inputs, monkeypatch): + real_gate = builder.validate_source_enrichment_candidate + + def gate(*args, **kwargs): + result = real_gate(*args, **kwargs) + inputs.output.mkdir() + (inputs.output / "sentinel").write_text("concurrent owner") + return result + + monkeypatch.setattr(builder, "validate_source_enrichment_candidate", gate) + with pytest.raises(FileExistsError, match="Output appeared"): + _build(inputs) + assert (inputs.output / "sentinel").read_text() == "concurrent owner" + assert list(inputs.output.parent.glob(".spm-enrichment-*")) == [] + + +def test_changed_producer_source_refuses_after_validation(inputs, monkeypatch): + identities = iter( + [ + inputs.code, + inputs.code | {"source_files_sha256": {"synthetic-producer.py": "e" * 64}}, + ] + ) + monkeypatch.setattr(builder, "_producer_identity", lambda: next(identities)) + with pytest.raises(ValueError, match="Producer source files changed"): + _build(inputs) + _assert_clean_failure(inputs) + + +def test_producer_receipt_hashes_the_loaded_checkout(): + identity = builder._producer_identity() + root = Path(builder.__file__).resolve().parents[1] + assert identity["source_files_sha256"] == { + name: hashlib.sha256((root / name).read_bytes()).hexdigest() + for name in builder._PRODUCER_FILES + } + assert len(identity["git_commit"]) == 40 + + +def test_producer_rejects_imported_code_from_another_checkout(monkeypatch): + def foreign_derivation(*args, **kwargs): + pytest.fail("foreign source reconstruction must not execute") + + monkeypatch.setattr(builder, "derive_spm_role_source", foreign_derivation) + with pytest.raises(ValueError, match="execute this checkout"): + builder._producer_identity() diff --git a/packages/microcosm-build/tests/test_us_spm_role_source.py b/packages/microcosm-build/tests/test_us_spm_role_source.py new file mode 100644 index 000000000..76e9e644f --- /dev/null +++ b/packages/microcosm-build/tests/test_us_spm_role_source.py @@ -0,0 +1,263 @@ +"""Pinned source roles must explain complete, unchanged native SPM units.""" + +import hashlib +from dataclasses import replace + +import numpy as np +import pandas as pd +import pytest + +from microcosm.build.us_runtime.spm_role_source import ( + EVIDENCE_SPM_ROLE, + AsecSpmRoleSource, + derive_spm_role_source, + independent_minor_role, +) + +pytest.importorskip("tables") + + +def _digest(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + +@pytest.fixture +def population(tmp_path): + # A minor spouse is independent, while a same-age dependent child is not. + # Unit 100 is cloned into native units 10 and 40 with the same source IDs. + source = pd.DataFrame( + { + "PERIDNUM": [f"{number:022}" for number in range(1, 7)], + "SPM_ID": [100, 100, 100, 200, 200, 300], + "PH_SEQ": [1, 1, 1, 2, 2, 3], + "P_SEQ": [1, 2, 3, 1, 2, 1], + "A_LINENO": [1, 2, 3, 1, 2, 1], + "A_AGE": [17, 16, 10, 40, 16, 15], + "SPM_HAGE": [17, 17, 17, 40, 40, 15], + "SPM_HEAD": [1, 0, 0, 1, 0, 1], + "SPM_NUMADULTS": [2, 2, 2, 1, 1, 1], + "SPM_NUMKIDS": [1, 1, 1, 1, 1, 0], + "SPM_NUMPER": [3, 3, 3, 2, 2, 1], + "A_FAMTYP": [1, 1, 1, 1, 1, 4], + "A_FAMREL": [1, 2, 3, 1, 3, 1], + "A_SPOUSE": [2, 1, 0, 0, 0, 0], + "PECOHAB": [0, 0, 0, 0, 0, 0], + } + ) + source_path = tmp_path / "pppub25.csv" + source.to_csv(source_path, index=False) + parent = pd.concat([source, source.iloc[:3]], ignore_index=True) + parent["source_row_id"] = [0, 1, 2, 3, 4, 5, 0, 1, 2] + parent["source_year"] = 2024 + parent["source_person_id"] = parent.PERIDNUM + parent["source_household_id"] = parent.PH_SEQ + parent["person_id"] = np.arange(1001, 1010) + parent["person_spm_unit_id"] = [10, 10, 10, 20, 20, 30, 40, 40, 40] + parent["age"] = parent.A_AGE.astype(float) + # The pooled producer globalizes SPM_ID, so raw ID equality is incorrect. + parent["SPM_ID"] = parent.SPM_ID.map({100: 1, 200: 2, 300: 3}) + # Frozen older source HDFs lack relationship columns; optional equality + # checks must not fill those cells just because the CSV carries the data. + parent = parent.drop(columns=["SPM_HEAD"]) + parent["A_FAMREL"] = parent.A_FAMREL.astype(float) + parent.loc[:2, "A_FAMREL"] = np.nan + parent_path = tmp_path / "parent.h5" + parent.to_hdf(parent_path, key="person") + pd.DataFrame({"spm_unit_id": [10, 20, 30, 40]}).to_hdf(parent_path, key="spm_unit") + pin = AsecSpmRoleSource( + income_year=2024, + survey_year=2025, + csv_sha256=_digest(source_path), + csv_size_bytes=source_path.stat().st_size, + persons=6, + units=3, + official_archive_url="https://example.invalid/fixture.zip", + archive_sha256="a" * 64, + member="pppub25.csv", + ) + return parent_path, source_path, pin + + +def _derive(population): + parent, source, pin = population + return derive_spm_role_source( + parent, + {2024: source}, + expected_parent_sha256=_digest(parent), + source_pins={2024: pin}, + ) + + +def _change_parent(population, change): + parent_path, _, _ = population + parent = pd.read_hdf(parent_path, "person") + change(parent) + parent.to_hdf(parent_path, key="person") + + +def _change_source(population, change): + parent_path, source_path, pin = population + source = pd.read_csv(source_path, dtype={"PERIDNUM": str}) + change(source) + source.to_csv(source_path, index=False) + return ( + parent_path, + source_path, + replace( + pin, + csv_sha256=_digest(source_path), + csv_size_bytes=source_path.stat().st_size, + ), + ) + + +def test_role_is_primitive_before_age_gate(): + people = pd.DataFrame( + { + "A_AGE": [14, 16, 17, 40, 16, 16], + "SPM_HEAD": [1, 0, 0, 1, 0, 0], + "A_FAMTYP": [0, 1, 4, 1, 1, 2], + "A_FAMREL": [0, 2, 1, 1, 3, 1], + } + ) + role = independent_minor_role(people) + assert role.tolist() == [True, True, True, True, False, False] + adult = people.A_AGE.ge(18) | (people.A_AGE.ge(15) & role) + assert adult.tolist() == [False, True, True, True, False, False] + + +def test_derivation_joins_clones_without_changing_parent(population): + parent, _, _ = population + before = parent.read_bytes() + result = _derive(population) + assert parent.read_bytes() == before + assert result.role.dtype == np.bool_ + assert result.role.tolist() == [ + True, + True, + False, + True, + False, + True, + True, + True, + False, + ] + assert list(result.evidence) == [ + "person_id", + "person_spm_unit_id", + EVIDENCE_SPM_ROLE, + ] + assert result.evidence.person_id.tolist() == list(range(1001, 1010)) + assert result.provenance["total_source_units"] == 3 + assert result.provenance["native_spm_units"] == 4 + assert result.provenance["complete_source_membership_units"] == 4 + assert result.provenance["minor_only_units_resolved"] == 3 + assert result.provenance["independent_minor_persons"] == 5 + assert result.provenance["adult_child_person_count_mismatch_units"] == 0 + assert ( + result.provenance["source_checks"][0][ + "extra_independent_minor_people_vs_head_only" + ] + == 1 + ) + assert result.provenance["nonmissing_optional_raw_fields_checked"]["A_FAMREL"] == 6 + + +def test_rejects_wrong_parent_identity(population): + parent, source, pin = population + with pytest.raises(ValueError, match="Parent H5 SHA-256"): + derive_spm_role_source( + parent, + {2024: source}, + expected_parent_sha256="0" * 64, + source_pins={2024: pin}, + ) + + +def test_rejects_unpinned_csv(population): + parent, source, pin = population + with pytest.raises(ValueError, match="CSV SHA-256"): + _derive((parent, source, replace(pin, csv_sha256="0" * 64))) + + +def test_rejects_missing_source_year(population): + parent, _, pin = population + with pytest.raises(ValueError, match="must match exactly"): + derive_spm_role_source( + parent, {}, expected_parent_sha256=_digest(parent), source_pins={2024: pin} + ) + + +@pytest.mark.parametrize( + "column", ["A_AGE", "P_SEQ", "SPM_NUMADULTS", "A_FAMTYP", "source_row_id"] +) +def test_rejects_parent_source_disagreement(population, column): + _change_parent( + population, lambda frame: frame.__setitem__(column, frame[column] + 1) + ) + with pytest.raises(ValueError, match=f"{column} disagrees"): + _derive(population) + + +def test_rejects_unmatched_parent_person(population): + _change_parent( + population, lambda frame: frame.loc.__setitem__((0, "PERIDNUM"), "9" * 22) + ) + with pytest.raises(ValueError, match="unmatched parent persons"): + _derive(population) + + +def test_rejects_lossy_person_id_coercion(population): + _change_parent( + population, + lambda frame: frame.__setitem__("PERIDNUM", frame.PERIDNUM.astype(float)), + ) + with pytest.raises(ValueError, match="exact 22-digit strings"): + _derive(population) + + +def test_rejects_native_membership_mixing_even_when_counts_match(population): + # Units 10 and 40 are source clones; switching their third members is valid, + # but switching a source member between different source units is not. + def exchange(frame): + frame.loc[2, "person_spm_unit_id"] = 20 + frame.loc[4, "person_spm_unit_id"] = 10 + + _change_parent(population, exchange) + with pytest.raises(ValueError, match="combines distinct source units"): + _derive(population) + + +def test_rejects_duplicate_source_member_inside_native_unit(population): + def duplicate(frame): + # Move a complete clone into its source unit, keeping native IDs present. + frame.loc[6, "person_spm_unit_id"] = 10 + + _change_parent(population, duplicate) + with pytest.raises(ValueError, match="repeats a source person"): + _derive(population) + + +def test_reconciles_full_source_not_just_selected_parent_units(population): + parent, source, pin = population + # Remove unit 300 from parent; its deliberately false raw count must still + # fail full-source reconciliation even though the parent never joins it. + person = pd.read_hdf(parent, "person").query("person_spm_unit_id != 30") + person.to_hdf(parent, key="person") + pd.DataFrame({"spm_unit_id": [10, 20, 40]}).to_hdf(parent, key="spm_unit") + changed = _change_source( + (parent, source, pin), + lambda frame: frame.loc.__setitem__((5, "SPM_NUMADULTS"), 0), + ) + with pytest.raises(ValueError, match="adult/child/person count reconciliation"): + _derive(changed) + + +def test_rejects_duplicate_source_keys(population): + changed = _change_source( + population, + lambda frame: frame.loc.__setitem__((1, "PERIDNUM"), frame.loc[0, "PERIDNUM"]), + ) + with pytest.raises(ValueError, match="duplicate PERIDNUM"): + _derive(changed) diff --git a/packages/microcosm-data/src/microcosm/data/contract.py b/packages/microcosm-data/src/microcosm/data/contract.py index 719c4cbca..89d573d86 100644 --- a/packages/microcosm-data/src/microcosm/data/contract.py +++ b/packages/microcosm-data/src/microcosm/data/contract.py @@ -4771,7 +4771,13 @@ def release_dataset_role(release_dir: Path | str) -> str: return role if isinstance(role, str) and role else NATIONAL_DEFAULT_DATASET_ROLE -def validate_release_dir(release_dir: Path | str) -> None: +def validate_release_dir( + release_dir: Path | str, + *, + parent_h5: Path | str | None = None, + artifact_root: Path | str | None = None, + compatibility_wheels: tuple[Path | str, ...] = (), +) -> None: """Check a local release directory against its dataset-role contract. The directory name is the build id (``populace-us-2024--``) @@ -4789,6 +4795,13 @@ def validate_release_dir(release_dir: Path | str) -> None: national critical-target set deliberately does not apply: the artifact is calibrated to a local surface by design. + ``release_type=source_enrichment`` selects the separately reviewed BuildP + inheritance contract before role dispatch: exact schema-5 calibration bytes, + complete H5 preservation, pinned source role evidence, and replayed native + loader checks. It requires ``parent_h5``, ``artifact_root``, and the tested + ``compatibility_wheels``. Pending local candidates have a separate validator + and cannot pass this publication gate. Ordinary calibration stays on schema 6. + TODO(#578 H5 household-count reconciliation): when the first modern UK exact-k release is actually cut, bind these manifest/diagnostic counts to the household row count read from its shipped H5. There is deliberately no @@ -4821,6 +4834,29 @@ def validate_release_dir(release_dir: Path | str) -> None: manifest_probe = json.loads(manifest_probe_path.read_text()) except (OSError, ValueError): manifest_probe = None + if isinstance(manifest_probe, Mapping) and "release_type" in manifest_probe: + from microcosm.data.source_enrichment import ( + SOURCE_ENRICHMENT_RELEASE_TYPE, + validate_source_enrichment_candidate, + ) + + if manifest_probe["release_type"] == SOURCE_ENRICHMENT_RELEASE_TYPE: + validate_source_enrichment_candidate( + release_dir, + parent_h5=parent_h5, + artifact_root=artifact_root, + require_compatibility=True, + compatibility_wheels=compatibility_wheels, + ) + return + if manifest_probe["release_type"] != "calibration": + raise ReleaseContractError( + release_dir, + [ + "release_manifest.json declares unknown release_type " + f"{manifest_probe['release_type']!r}." + ], + ) if isinstance(manifest_probe, Mapping) and "dataset_role" in manifest_probe: declared_role = manifest_probe["dataset_role"] if declared_role not in ( @@ -5293,6 +5329,14 @@ def validate_evidence_release_dir(release_dir: Path | str) -> None: manifest_probe = json.loads(manifest_probe_path.read_text()) except (OSError, ValueError): manifest_probe = None + if ( + isinstance(manifest_probe, Mapping) + and manifest_probe.get("release_type", "calibration") != "calibration" + ): + raise ReleaseContractError( + release_dir, + ["evidence tier does not accept source-enrichment releases"], + ) if isinstance(manifest_probe, Mapping) and "dataset_role" in manifest_probe: declared_role = manifest_probe["dataset_role"] if declared_role != NATIONAL_DEFAULT_DATASET_ROLE: diff --git a/packages/microcosm-data/src/microcosm/data/h5_enrichment.py b/packages/microcosm-data/src/microcosm/data/h5_enrichment.py new file mode 100644 index 000000000..fb1d81e1a --- /dev/null +++ b/packages/microcosm-data/src/microcosm/data/h5_enrichment.py @@ -0,0 +1,409 @@ +"""Append a native pandas person input without rewriting existing variables. + +BuildP uses a compound ``person/table``, not variable/period HDF groups. Its +record type must grow by one field. All original field types, offsets and +bytes stay exact; only four pandas column-registration attributes change. +Every other HDF object and attribute is compared without exemptions. +""" + +from __future__ import annotations + +import hashlib +import io +import os +import pickle +import shutil +from pathlib import Path + +import h5py +import numpy as np + +ROLE = "is_spm_independent_minor_role" +_TABLE = "person/table" +_REGISTRATION = {"data_columns", "values_cols", "non_index_axes", "info"} +_CHUNK_ROWS = 4096 + + +def file_sha256(path: str | Path) -> str: + digest = hashlib.sha256() + with Path(path).open("rb") as stream: + for block in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(block) + return digest.hexdigest() + + +class _MetadataUnpickler(pickle.Unpickler): + def find_class(self, module, name): + raise ValueError("HDF column metadata must contain only primitive containers") + + +def _registration_value(name: str, raw: object) -> bytes: + value = _MetadataUnpickler(io.BytesIO(bytes(raw))).load() + if name in {"data_columns", "values_cols"}: + if not isinstance(value, list) or ROLE in value: + raise ValueError(f"Unsupported person {name}") + value.append(ROLE) + elif name == "non_index_axes": + if ( + not isinstance(value, list) + or len(value) != 1 + or value[0][0] != 1 + or not isinstance(value[0][1], list) + or ROLE in value[0][1] + ): + raise ValueError("Unsupported person non_index_axes") + value[0][1].append(ROLE) + else: + if not isinstance(value, dict) or ROLE in value: + raise ValueError("Unsupported person info") + value[ROLE] = {} + return pickle.dumps(value, protocol=0) + + +def _copy_attribute(source, destination, name: str) -> None: + attribute = source.attrs.get_id(name) + copied = h5py.h5a.create( + destination.id, name.encode(), attribute.get_type(), attribute.get_space() + ) + if attribute.shape is not None: + value = np.empty(attribute.shape, dtype=attribute.dtype) + attribute.read(value, mtype=attribute.get_type()) + copied.write(value, mtype=attribute.get_type()) + + +def _new_table_attributes(field_count: int) -> dict[str, object]: + return { + f"FIELD_{field_count}_NAME": ROLE, + f"FIELD_{field_count}_FILL": np.uint8(0), + f"{ROLE}_dtype": "bool", + f"{ROLE}_kind": np.bytes_(pickle.dumps([ROLE], protocol=0)), + f"{ROLE}_meta": np.bytes_(b"N."), + } + + +def _check_supported_table(table: h5py.Dataset) -> None: + if ( + table.ndim != 1 + or not table.dtype.names + or "person_id" not in table.dtype.names + or table.chunks is None + or table.compression is not None + or table.shuffle + or table.fletcher32 + or table.scaleoffset is not None + or table.is_virtual + or table.external + ): + raise ValueError("Expected the unfiltered BuildP native pandas person table") + if ROLE in table.dtype.names: + raise ValueError("Parent already contains the native SPM role") + if any(table.dtype[name].hasobject for name in table.dtype.names): + raise ValueError("Object/reference fields are not supported") + + +def append_native_spm_role( + parent_h5: str | Path, + candidate_h5: str | Path, + role: np.ndarray, + *, + expected_parent_sha256: str, +) -> dict: + """Create a new file, append one native field, and exhaustively verify it. + + Destination must not exist. The caller owns a private staging directory; + failure removes only the file this call created. The parent is read-only. + No HDF repacking, pandas table rewrite, or weight operation occurs. + """ + parent_h5, candidate_h5 = Path(parent_h5), Path(candidate_h5) + if file_sha256(parent_h5) != expected_parent_sha256: + raise ValueError("Parent H5 SHA-256 mismatch") + role = np.asarray(role) + if role.dtype != np.dtype(bool) or role.ndim != 1: + raise ValueError("Role must be a one-dimensional Boolean array") + with h5py.File(parent_h5, "r") as parent: + _check_supported_table(parent[_TABLE]) + if role.shape != parent[_TABLE].shape: + raise ValueError("Role coverage does not match the parent person rows") + # Exclusive creation also refuses aliases/symlinks to the parent. + descriptor = os.open(candidate_h5, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + try: + with os.fdopen(descriptor, "wb") as destination, parent_h5.open("rb") as source: + shutil.copyfileobj(source, destination, 1024 * 1024) + if file_sha256(candidate_h5) != expected_parent_sha256: + raise ValueError("Parent changed during the verified copy") + with h5py.File(candidate_h5, "r+") as candidate: + old = candidate[_TABLE] + compound = old.id.get_type().copy() + old_size = compound.get_size() + compound.set_size(old_size + 1) + # PyTables Boolean columns are HDF bitfields, not enums/integers. + compound.insert(ROLE.encode(), old_size, h5py.h5t.NATIVE_B8) + creation = h5py.h5p.create(h5py.h5p.DATASET_CREATE) + creation.set_chunk(old.chunks) + creation.set_obj_track_times(False) + group = candidate["person"] + temporary = "_spm_role_enriched_table" + if temporary in group: + raise ValueError("Unexpected temporary HDF object in parent") + new = h5py.Dataset( + h5py.h5d.create( + group.id, + temporary.encode(), + compound, + old.id.get_space(), + dcpl=creation, + ) + ) + for start in range(0, len(old), _CHUNK_ROWS): + stop = min(start + _CHUNK_ROWS, len(old)) + original = _raw_rows(old, start, stop) + extended = np.empty(stop - start, dtype=new.dtype) + extended.view(np.uint8).reshape(-1, old_size + 1)[:, :old_size] = ( + original.view(np.uint8).reshape(-1, old_size) + ) + extended[ROLE] = role[start:stop] + memory = h5py.h5s.create_simple((stop - start,)) + selection = new.id.get_space() + selection.select_hyperslab((start,), (stop - start,)) + new.id.write(memory, selection, extended, mtype=compound) + for name in old.attrs: + _copy_attribute(old, new, name) + for name, value in _new_table_attributes(len(old.dtype.names)).items(): + if isinstance(value, str): + new.attrs.create( + name, value, dtype=h5py.string_dtype("utf-8", len(value)) + ) + else: + new.attrs[name] = value + for name in sorted(_REGISTRATION): + group.attrs[name] = np.bytes_( + _registration_value(name, group.attrs[name]) + ) + del candidate[_TABLE] + group.move(temporary, "table") + report = compare_h5_enrichment(parent_h5, candidate_h5) + if report["parent_sha256"] != expected_parent_sha256: + raise ValueError("Parent changed during enrichment") + if report["role_sha256"] != hashlib.sha256(role.tobytes()).hexdigest(): + raise ValueError("Written native role differs from the source derivation") + return report + except BaseException: + candidate_h5.unlink(missing_ok=True) + raise + + +def _objects(root: h5py.File) -> dict: + objects = {"": root} + addresses = {h5py.h5o.get_info(root.id).addr} + + def descend(group, prefix=""): + for name in group: + path = f"{prefix}/{name}".lstrip("/") + if not isinstance(group.get(name, getlink=True), h5py.HardLink): + raise ValueError(f"Nonlocal HDF link at {path}") + child = group[name] + address = h5py.h5o.get_info(child.id).addr + if address in addresses: + raise ValueError(f"Aliased HDF object at {path}") + addresses.add(address) + objects[path] = child + if isinstance(child, h5py.Group): + descend(child, path) + + descend(root) + return objects + + +def _value_bytes(value: object) -> bytes: + if isinstance(value, h5py.Empty): + return b"" + array = np.asarray(value) + if array.dtype.hasobject: + raise ValueError("Object/reference HDF values are not supported") + return array.tobytes() + + +def _raw_rows(dataset: h5py.Dataset, start: int, stop: int) -> np.ndarray: + """Read with the file type to prevent HDF string-padding conversions.""" + shape = (stop - start, *dataset.shape[1:]) + result = np.empty(shape, dtype=dataset.dtype) + memory = h5py.h5s.create_simple(shape) + selection = dataset.id.get_space() + selection.select_hyperslab((start, *(0 for _ in dataset.shape[1:])), shape) + dataset.id.read(memory, selection, result, mtype=dataset.id.get_type()) + return result + + +def _raw_selection(dataset: h5py.Dataset, selection) -> np.ndarray: + if selection != (): + return _raw_rows(dataset, selection.start, selection.stop) + result = np.empty((), dtype=dataset.dtype) + dataset.id.read(h5py.h5s.ALL, h5py.h5s.ALL, result, mtype=dataset.id.get_type()) + return result + + +def _attribute_bytes(attribute) -> bytes: + if attribute.shape is None: + return b"" + value = np.empty(attribute.shape, dtype=attribute.dtype) + attribute.read(value, mtype=attribute.get_type()) + return _value_bytes(value) + + +def _compare_attributes(path: str, old, new) -> int: + additions = _new_table_attributes(len(old.dtype.names)) if path == _TABLE else {} + if set(new.attrs) != set(old.attrs) | set(additions): + raise ValueError(f"HDF attributes changed at {path}") + for name in old.attrs: + if path == "person" and name in _REGISTRATION: + expected = _registration_value(name, old.attrs[name]) + attribute = new.attrs.get_id(name) + if ( + attribute.shape != () + or attribute.get_type().get_class() != h5py.h5t.STRING + or attribute.dtype != np.dtype(f"S{len(expected)}") + or attribute.get_type().get_cset() != h5py.h5t.CSET_ASCII + or bytes(new.attrs[name]) != expected + ): + raise ValueError( + f"Unexpected pandas column-registration change: {name}" + ) + continue + a, b = old.attrs.get_id(name), new.attrs.get_id(name) + if ( + a.get_type() != b.get_type() + or a.shape != b.shape + or _attribute_bytes(a) != _attribute_bytes(b) + ): + raise ValueError(f"HDF attribute changed at {path}:{name}") + for name, expected in additions.items(): + actual = new.attrs[name] + attribute = new.attrs.get_id(name) + if isinstance(expected, str): + expected = expected.encode() + correct_type = ( + attribute.get_type().get_class() == h5py.h5t.STRING + and attribute.dtype == np.dtype(f"S{len(expected)}") + and attribute.get_type().get_cset() == h5py.h5t.CSET_UTF8 + ) + elif isinstance(expected, np.bytes_): + correct_type = ( + attribute.get_type().get_class() == h5py.h5t.STRING + and attribute.dtype == np.asarray(expected).dtype + and attribute.get_type().get_cset() == h5py.h5t.CSET_ASCII + ) + else: + correct_type = attribute.get_type() == h5py.h5t.NATIVE_UINT8 + if ( + not correct_type + or attribute.shape != () + or _value_bytes(actual) != _value_bytes(expected) + ): + raise ValueError(f"Invalid added role metadata: {name}") + return len(old.attrs) + + +def compare_h5_enrichment(parent_h5: str | Path, candidate_h5: str | Path) -> dict: + """Prove exact old-variable identity and the single allowed schema extension. + + Comparisons include HDF datatype identities (including bitfields), NaN + payloads, signed zero, all indexes, attributes, and row order. The result + contains only aggregate counts and hashes, never person records. + """ + parent_sha = file_sha256(parent_h5) + candidate_sha = file_sha256(candidate_h5) + old_fields = datasets = attributes = groups = 0 + role_digest = hashlib.sha256() + with h5py.File(parent_h5, "r") as parent, h5py.File(candidate_h5, "r") as candidate: + old_objects, new_objects = _objects(parent), _objects(candidate) + if old_objects.keys() != new_objects.keys(): + raise ValueError("HDF groups/datasets added or removed") + _check_supported_table(parent[_TABLE]) + for path, old in old_objects.items(): + new = new_objects[path] + if type(old) is not type(new): + raise ValueError(f"HDF object kind changed at {path}") + attributes += _compare_attributes(path, old, new) + if isinstance(old, h5py.Group): + groups += 1 + continue + datasets += 1 + if ( + old.shape != new.shape + or old.maxshape != new.maxshape + or old.chunks != new.chunks + or old.compression != new.compression + or old.compression_opts != new.compression_opts + or old.shuffle != new.shuffle + or old.fletcher32 != new.fletcher32 + or old.scaleoffset != new.scaleoffset + or new.is_virtual + or new.external + ): + raise ValueError(f"HDF shape/storage changed at {path}") + if path == _TABLE: + before_type, after_type = old.id.get_type(), new.id.get_type() + expected_type = before_type.copy() + expected_type.set_size(before_type.get_size() + 1) + expected_type.insert( + ROLE.encode(), before_type.get_size(), h5py.h5t.NATIVE_B8 + ) + if expected_type != after_type or new.dtype.names != ( + *old.dtype.names, + ROLE, + ): + raise ValueError( + "Person dtype differs beyond the appended role field" + ) + old_fields = len(old.dtype.names) + elif old.id.get_type() != new.id.get_type() or old.dtype != new.dtype: + raise ValueError(f"HDF dtype changed at {path}") + slices = ( + [()] + if old.ndim == 0 + else [ + slice(i, min(i + _CHUNK_ROWS, len(old))) + for i in range(0, len(old), _CHUNK_ROWS) + ] + ) + for selection in slices: + if path == _TABLE: + a = _raw_rows(old, selection.start, selection.stop) + b = _raw_rows(new, selection.start, selection.stop) + for name in old.dtype.names: + if a[name].tobytes() != b[name].tobytes(): + raise ValueError(f"Existing person field changed: {name}") + role = b[ROLE] + if not np.isin(role, [0, 1]).all(): + raise ValueError("Native SPM role is not Boolean") + role_digest.update(role.tobytes()) + elif _value_bytes(_raw_selection(old, selection)) != _value_bytes( + _raw_selection(new, selection) + ): + raise ValueError(f"Existing HDF dataset values changed: {path}") + persons = len(parent[_TABLE]) + if ( + file_sha256(parent_h5) != parent_sha + or file_sha256(candidate_h5) != candidate_sha + ): + raise ValueError("HDF changed during the exhaustive comparison") + return { + "schema_version": 1, + "parent_sha256": parent_sha, + "candidate_sha256": candidate_sha, + "all_preexisting_variables_exact": True, + "groups_checked_including_root": groups, + "datasets_checked": datasets, + "preexisting_attributes_checked": attributes, + "person_fields_checked_including_index": old_fields, + "persons": persons, + "role_sha256": role_digest.hexdigest(), + "physical_schema_extension": { + "dataset": _TABLE, + "appended_field": ROLE, + "storage": "HDF bitfield8; pandas bool", + "record_size_increase_bytes": 1, + "column_registration_attributes": sorted(_REGISTRATION), + "all_other_objects_and_attributes_exact": True, + }, + } diff --git a/packages/microcosm-data/src/microcosm/data/publish_cli.py b/packages/microcosm-data/src/microcosm/data/publish_cli.py index 25e394e6d..f70d2822b 100644 --- a/packages/microcosm-data/src/microcosm/data/publish_cli.py +++ b/packages/microcosm-data/src/microcosm/data/publish_cli.py @@ -7,6 +7,7 @@ import sys from pathlib import Path +from microcosm.data.contract import validate_evidence_release_dir, validate_release_dir from microcosm.data.release import publish_release @@ -68,6 +69,23 @@ def main(argv: list[str] | None = None) -> int: "for example populace_us_2024.h5." ), ) + parser.add_argument( + "--parent-h5", + type=Path, + help="Exact certified parent H5 required for source-enrichment publication.", + ) + parser.add_argument( + "--compatibility-wheel", + action="append", + type=Path, + default=[], + help="Exact installed country/Core/wrapper/calculator wheel; repeat for all four packages. Candidate wheels may be tested before publication.", + ) + parser.add_argument( + "--preflight-only", + action="store_true", + help="Run the actual publication contract, including enrichment loader tests, without publishing.", + ) parser.add_argument( "--create-tag", action="store_true", @@ -183,10 +201,25 @@ def main(argv: list[str] | None = None) -> int: ) return 1 + if args.preflight_only: + if args.evidence: + validate_evidence_release_dir(Path(args.release_dir)) + else: + validate_release_dir( + Path(args.release_dir), + parent_h5=args.parent_h5, + artifact_root=args.artifact_root, + compatibility_wheels=tuple(args.compatibility_wheel), + ) + print(json.dumps({"valid": True, "published": False})) + return 0 + pointer = publish_release( Path(args.release_dir), args.repo_id, artifact_root=Path(args.artifact_root) if args.artifact_root else None, + parent_h5=args.parent_h5, + compatibility_wheels=tuple(args.compatibility_wheel), create_tag=args.create_tag, tag_name=args.tag_name, extra_files=tuple(args.extra_file), diff --git a/packages/microcosm-data/src/microcosm/data/release.py b/packages/microcosm-data/src/microcosm/data/release.py index 4bab1172b..d3def80ba 100644 --- a/packages/microcosm-data/src/microcosm/data/release.py +++ b/packages/microcosm-data/src/microcosm/data/release.py @@ -164,6 +164,8 @@ def publish_release( *, api=None, artifact_root: Path | str | None = None, + parent_h5: Path | str | None = None, + compatibility_wheels: tuple[Path | str, ...] = (), create_tag: bool = True, tag_name: str | None = None, extra_files: tuple[str, ...] = (), @@ -195,6 +197,13 @@ def publish_release( Contract files are always read from ``release_dir`` and uploaded under ``releases//``; artifact paths are uploaded to their manifest-declared repo paths. + parent_h5: Exact certified parent H5 for source-enrichment releases. + The publication gate rereads it and the candidate to prove that + every pre-existing logical variable and metadata field is preserved. + compatibility_wheels: Exact installed country, Core, wrapper, and calculator wheels + used to replay source-enrichment native-loader compatibility checks. + External package publication and numerical model acceptance remain + the release operator's gates. create_tag: Create an immutable Hub tag for the release snapshot before updating main. The tag defaults to the release id. This is required when artifact revisions in ``release_manifest.json`` are pinned to @@ -242,6 +251,13 @@ def publish_release( release_dir = Path(release_dir) if evidence: validate_evidence_release_dir(release_dir) + elif parent_h5 is not None or compatibility_wheels: + validate_release_dir( + release_dir, + parent_h5=parent_h5, + artifact_root=artifact_root, + compatibility_wheels=compatibility_wheels, + ) else: validate_release_dir(release_dir) release_id = release_dir.name diff --git a/packages/microcosm-data/src/microcosm/data/source_enrichment.py b/packages/microcosm-data/src/microcosm/data/source_enrichment.py new file mode 100644 index 000000000..d327f42fd --- /dev/null +++ b/packages/microcosm-data/src/microcosm/data/source_enrichment.py @@ -0,0 +1,978 @@ +"""Additive native-input releases that inherit an immutable calibration. + +This release type does not certify a new calibration or upgrade its schema. +Its authority is the reviewed parent byte identity, an exhaustive H5 comparison, +Census source reconciliation, and separately measured native-loader compatibility. +Publication replays the latter checks before the Hub client is constructed. +""" + +from __future__ import annotations + +import csv +import hashlib +import json +import re +from collections.abc import Mapping +from pathlib import Path + +from microcosm.data.contract import ReleaseContractError + +SOURCE_ENRICHMENT_RELEASE_TYPE = "source_enrichment" +SOURCE_ENRICHMENT_FILE = "source_enrichment.json" +ROLE_VARIABLE = "is_spm_independent_minor_role" +EVIDENCE_COLUMN = "is_spm_independence_role" +SOURCE_EVIDENCE_FILE = "source_spm_person_independence.csv" +SOURCE_EVIDENCE_SHA256 = ( + "22b5968d90fecfeef7614583e493fe10cc16bda8b5be82e6f49a5bc2102d3ce5" +) +SOURCE_PROVENANCE_FILE = "source_spm_independence_provenance.json" +COMPATIBILITY_FILE = "source_enrichment_compatibility.json" +COMPATIBILITY_PACKAGES = ( + "policyengine-us", + "policyengine-core", + "policyengine", + "spm-calculator", +) +LOADED_SOURCE_PACKAGES = { + "country_loader": "policyengine-us", + "wrapper_loader": "policyengine", + "native_role": "spm-calculator", +} +PARENT_BUILD_ID = "populace-us-2024-buildp-sparse-rmloss100-cae8640-20260728T011454Z" +PARENT_DATASET_SHA256 = ( + "48b9d479fb4fd1c3537f9383ce4697d130b6f618658409d74f6233c43b994c7e" +) +# Reviewed immutable BuildP evidence. Callers cannot grant another schema-5 +# calibration permission to use the inheritance lane by supplying their own pins. +PARENT_FILES = { + "parent_release_manifest.json": ( + "dd949ba3c4c7a56aff8442c6db5a031d2c84c3da59f5d14832f468c358604506" + ), + "parent_build_manifest.json": ( + "1990e8fd37ce66f499b8e6600c5896701bb07015be132a12de64e321a2edc67e" + ), + "calibration_diagnostics.json": ( + "870449b44e86b13b25bcea1a57f0e7af37f4d4db18be815eea3acdf9fe6eb40e" + ), + "us_source_coverage.json": ( + "6406c8686c292015a5bc7265a42402f89935f9395a8b63510efdd4d9562e7ff5" + ), +} +CENSUS_PERSON_PINS = { + 2023: "19b56537e50e7663f954361ef2bb5ce9cef8d9d45f156fe1a69a99b654198ffe", + 2024: "21a2b9e0e4b08534563578a45acad77868af4ae9a7d46f23776b707d4a559aa7", + 2025: "06921fe83fc66c907e6c7b86b82255dc70458ee7d76258fc48297cb34f0c06b5", +} +EXPECTED_COUNTS = { + "persons_joined": 166321, + "native_spm_units": 59900, + "total_source_people": 432523, + "total_source_units": 176039, + "minor_only_units_resolved": 28, + "classification_changed_units_vs_age_only": 132, +} +_SHA256_RE = re.compile(r"[0-9a-f]{64}") +PRODUCER_SOURCE_FILES = ( + "tools/build_us_spm_role_enrichment.py", + "packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py", + "packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py", + "packages/microcosm-data/src/microcosm/data/h5_enrichment.py", + "packages/microcosm-data/src/microcosm/data/source_enrichment.py", + "packages/microcosm-data/src/microcosm/data/contract.py", +) + + +def sha256_file(path: Path | str) -> str: + digest = hashlib.sha256() + with Path(path).open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _json(path: Path, failures: list[str]) -> dict: + try: + payload = json.loads(path.read_text()) + if not isinstance(payload, dict): + raise ValueError("expected an object") + return payload + except (OSError, ValueError) as exc: + failures.append(f"{path.name}: {exc}") + return {} + + +def _mapping(value: object) -> Mapping: + return value if isinstance(value, Mapping) else {} + + +def _check_person_evidence(path: Path, candidate: Path) -> dict: + """Independently verify a complete, ordered, one-to-one native input join.""" + import h5py + import numpy as np + + with path.open(newline="") as stream: + reader = csv.DictReader(stream) + expected = ["person_id", "person_spm_unit_id", EVIDENCE_COLUMN] + if reader.fieldnames != expected: + raise ValueError(f"{path.name} must have exactly {expected}") + rows = list(reader) + if any(row[EVIDENCE_COLUMN] not in {"False", "True"} for row in rows): + raise ValueError("source person evidence roles must be literal False or True") + ids = np.array([int(row["person_id"]) for row in rows], dtype=np.int64) + units = np.array([int(row["person_spm_unit_id"]) for row in rows], dtype=np.int64) + roles = np.array([row[EVIDENCE_COLUMN] == "True" for row in rows], dtype=bool) + if len(np.unique(ids)) != len(ids): + raise ValueError("source person evidence has duplicate person IDs") + with h5py.File(candidate, "r") as h5: + table = h5["person/table"] + for field, values in ( + ("person_id", ids), + ("person_spm_unit_id", units), + (ROLE_VARIABLE, roles), + ): + if not np.array_equal(table[field], values): + raise ValueError( + f"source person evidence does not match native {field}" + ) + role_index = table.dtype.names.index(ROLE_VARIABLE) + if table.id.get_type().get_member_type(role_index) != h5py.h5t.NATIVE_B8 or ( + table.attrs.get(f"{ROLE_VARIABLE}_dtype") != b"bool" + ): + raise ValueError("native person role must have PyTables bool bitfield type") + return {"persons_joined": len(ids), "native_spm_units": len(np.unique(units))} + + +def _check_source_provenance(provenance: Mapping, failures: list[str]) -> None: + if provenance.get("dataset_sha256") != PARENT_DATASET_SHA256: + failures.append( + "source provenance dataset_sha256 must identify reviewed BuildP" + ) + for key, expected in EXPECTED_COUNTS.items(): + if type(provenance.get(key)) is not int or provenance[key] != expected: + failures.append(f"source provenance {key} must equal {expected}") + for key in ("unmatched_persons", "adult_child_person_count_mismatch_units"): + if type(provenance.get(key)) is not int or provenance[key] != 0: + failures.append(f"source provenance {key} must equal zero") + if ( + provenance.get("complete_source_membership_units") + != EXPECTED_COUNTS["native_spm_units"] + ): + failures.append("source provenance must reconcile every native SPM unit") + for key in ("weights_used", "ages_changed"): + if provenance.get(key) is not False: + failures.append(f"source provenance {key} must be false") + if provenance.get("primitive_column") != ROLE_VARIABLE: + failures.append("source provenance primitive_column must name the native role") + if provenance.get("evidence_column") != EVIDENCE_COLUMN: + failures.append("source provenance evidence_column must name the source role") + checks = provenance.get("source_checks") + if not isinstance(checks, list) or len(checks) != len(CENSUS_PERSON_PINS): + failures.append("source provenance must carry every pinned Census source check") + return + seen = set() + for raw in checks: + row = _mapping(raw) + year = row.get("survey_year") + if type(year) is not int or year not in CENSUS_PERSON_PINS or year in seen: + failures.append("source provenance has an unknown or duplicate Census year") + continue + seen.add(year) + if row.get("csv_sha256") != CENSUS_PERSON_PINS[year]: + failures.append(f"source provenance Census {year} CSV hash differs") + if row.get("adult_child_person_count_mismatch_units") != 0: + failures.append( + f"source provenance Census {year} count reconciliation failed" + ) + if not str(row.get("official_archive_url", "")).startswith( + "https://www2.census.gov/" + ): + failures.append( + f"source provenance Census {year} requires official archive" + ) + archive_sha = row.get("archive_sha256") + if not isinstance(archive_sha, str) or not _SHA256_RE.fullmatch(archive_sha): + failures.append(f"source provenance Census {year} requires archive SHA256") + + +def validate_source_enrichment_candidate( + release_dir: Path | str, + *, + parent_h5: Path | str | None, + artifact_root: Path | str | None, + require_compatibility: bool = False, + compatibility_wheels: tuple[Path | str, ...] = (), +) -> dict: + """Check the inheritance contract; pending candidates cannot be published. + + Unlike the standard release contract, this never treats parent diagnostics + as measurements of the enriched model. Both actual H5 files are mandatory: + a hand-written preservation receipt is not sufficient evidence. + """ + from microcosm.data.h5_enrichment import compare_h5_enrichment + + release_dir = Path(release_dir) + failures: list[str] = [] + manifest = _json(release_dir / "release_manifest.json", failures) + build = _json(release_dir / "build_manifest.json", failures) + report = _json(release_dir / SOURCE_ENRICHMENT_FILE, failures) + if manifest.get("release_type") != SOURCE_ENRICHMENT_RELEASE_TYPE: + failures.append( + "release_manifest.json must declare release_type=source_enrichment" + ) + if manifest.get("schema_version") != 1: + failures.append( + "source enrichment release manifest schema_version must equal 1" + ) + if _mapping(manifest.get("build")).get("build_id") != release_dir.name: + failures.append( + "release manifest build_id must match its new release directory" + ) + if release_dir.name == PARENT_BUILD_ID or not release_dir.name.startswith( + "populace-us-2024-" + ): + failures.append("source enrichment requires a NEW US 2024 release id") + if manifest.get("dataset_role", "national_default") != "national_default": + failures.append("source enrichment supports only the national default dataset") + if build.get("build_id") != release_dir.name: + failures.append("build manifest build_id must match the new release directory") + if build.get("release_type") != SOURCE_ENRICHMENT_RELEASE_TYPE: + failures.append("build manifest must declare source_enrichment") + if report.get("schema_version") != 1 or report.get("release_type") != ( + SOURCE_ENRICHMENT_RELEASE_TYPE + ): + failures.append( + "source_enrichment.json must declare source_enrichment schema 1" + ) + if report.get("operation") != "add_native_spm_independent_minor_role": + failures.append("source enrichment operation must add only the native SPM role") + parent = _mapping(report.get("parent")) + expected_parent = { + "build_id": PARENT_BUILD_ID, + "repo_id": "policyengine/populace-us", + "revision": PARENT_BUILD_ID, + "dataset_sha256": PARENT_DATASET_SHA256, + "calibration_diagnostics_schema_version": 5, + "files": PARENT_FILES, + } + if parent != expected_parent: + failures.append( + "source enrichment parent must match the reviewed BuildP identity" + ) + for name, expected in PARENT_FILES.items(): + path = release_dir / name + if not path.is_file() or sha256_file(path) != expected: + failures.append(f"inherited {name} must preserve the reviewed parent bytes") + diagnostics = _json(release_dir / "calibration_diagnostics.json", failures) + if ( + type(diagnostics.get("schema_version")) is not int + or diagnostics.get("schema_version") != 5 + ): + failures.append("inherited calibration_diagnostics must retain schema 5") + if report.get("added_variable") != { + "name": ROLE_VARIABLE, + "entity": "person", + "dtype": "bool", + "age_gate_applied": False, + }: + failures.append( + "added_variable must declare the person bool primitive before age gate" + ) + dataset = _mapping(report.get("dataset")) + filename = dataset.get("filename") + if ( + not isinstance(filename, str) + or Path(filename).name != filename + or not (filename.endswith(".h5")) + ): + failures.append("source enrichment dataset.filename must be a bare H5 filename") + filename = None + candidate = Path(artifact_root) / filename if artifact_root and filename else None + if parent_h5 is None or candidate is None: + failures.append( + "source enrichment requires parent_h5 and artifact_root for exact H5 verification" + ) + elif not Path(parent_h5).is_file() or not candidate.is_file(): + failures.append("source enrichment parent and candidate H5 files must exist") + else: + if sha256_file(parent_h5) != PARENT_DATASET_SHA256: + failures.append("parent H5 SHA256 differs from reviewed BuildP") + if candidate.resolve() == Path(parent_h5).resolve(): + failures.append("source enrichment must create a NEW H5") + if sha256_file(candidate) != dataset.get("sha256"): + failures.append("candidate H5 SHA256 differs from source enrichment report") + try: + comparison = compare_h5_enrichment(Path(parent_h5), candidate) + if report.get("preservation") != comparison: + failures.append( + "preservation report differs from actual exhaustive H5 comparison" + ) + except (OSError, ValueError, KeyError, TypeError) as exc: + failures.append(f"exact H5 preservation failed: {exc}") + if _mapping(build.get("dataset")) != dataset: + failures.append("build manifest dataset must match the new H5 identity") + calibration = _mapping(build.get("calibration")) + if calibration != { + "mode": "inherited", + "parent_build_id": PARENT_BUILD_ID, + "diagnostics_sha256": PARENT_FILES["calibration_diagnostics.json"], + "diagnostics_schema_version": 5, + }: + failures.append( + "build calibration must explicitly inherit schema-5 parent evidence" + ) + source = _mapping(report.get("source")) + for field, expected_name in ( + ("person_evidence", SOURCE_EVIDENCE_FILE), + ("provenance", SOURCE_PROVENANCE_FILE), + ): + name = source.get(f"{field}_filename") + path = release_dir / expected_name + if ( + name != expected_name + or not path.is_file() + or sha256_file(path) != source.get(f"{field}_sha256") + ): + failures.append( + f"source enrichment {field} must bind the immutable {expected_name}" + ) + evidence_path = release_dir / SOURCE_EVIDENCE_FILE + if ( + not evidence_path.is_file() + or sha256_file(evidence_path) != SOURCE_EVIDENCE_SHA256 + ): + failures.append( + "source evidence must match the independently reviewed Census-derived person table" + ) + provenance = _json(release_dir / SOURCE_PROVENANCE_FILE, failures) + _check_source_provenance(provenance, failures) + if report.get("reconciliation") != provenance: + failures.append("source enrichment reconciliation must equal source provenance") + if ( + candidate + and candidate.is_file() + and (release_dir / SOURCE_EVIDENCE_FILE).is_file() + ): + try: + counts = _check_person_evidence( + release_dir / SOURCE_EVIDENCE_FILE, candidate + ) + if any(provenance.get(key) != value for key, value in counts.items()): + failures.append( + "source provenance coverage differs from actual person evidence" + ) + except (ValueError, KeyError, OSError, TypeError) as exc: + failures.append(f"native source evidence reconciliation failed: {exc}") + required_artifacts = { + *PARENT_FILES, + SOURCE_ENRICHMENT_FILE, + SOURCE_EVIDENCE_FILE, + SOURCE_PROVENANCE_FILE, + } + artifacts = _mapping(manifest.get("artifacts")) + by_path = {} + for key, raw in artifacts.items(): + entry = _mapping(raw) + path = entry.get("path") + if not isinstance(path, str) or Path(path).name != path: + failures.append( + f"source enrichment artifact {key} must use a bare filename" + ) + continue + if path in by_path: + failures.append(f"source enrichment duplicate artifact path {path}") + by_path[path] = entry + if ( + entry.get("repo_id") != "policyengine/populace-us" + or entry.get("revision") != release_dir.name + ): + failures.append( + f"source enrichment artifact {key} must pin the new repo/tag" + ) + local = candidate if path == filename else release_dir / path + if ( + local is None + or not local.is_file() + or sha256_file(local) != entry.get("sha256") + ): + failures.append(f"source enrichment artifact {key} hash/file mismatch") + if not required_artifacts.issubset(by_path): + failures.append( + "release manifest must list every inherited and source enrichment artifact" + ) + national = _mapping(manifest.get("default_datasets")).get("national") + native_artifact = ( + _mapping(artifacts.get(national)) if isinstance(national, str) else {} + ) + if ( + native_artifact.get("path") != filename + or native_artifact.get("kind") != "microdata" + ): + failures.append("default_datasets.national must select the enriched native H5") + if ( + len([entry for entry in by_path.values() if entry.get("kind") == "microdata"]) + != 1 + ): + failures.append( + "source enrichment must deliver exactly one native H5 microdata artifact" + ) + compatibility = _mapping(report.get("compatibility")) + if compatibility.get("status") == "pending": + if require_compatibility: + failures.append( + "source enrichment compatibility is pending; run real candidate-wheel native loader qualification" + ) + if manifest.get("compatible_core_packages") or manifest.get( + "compatible_model_packages" + ): + failures.append( + "pending source enrichment must not claim model/Core compatibility" + ) + if any( + key in _mapping(manifest.get("build")) + for key in ("built_with_core_package", "built_with_model_package") + ): + failures.append( + "pending source enrichment must not copy built-with package claims" + ) + elif compatibility.get("status") == "passed": + if COMPATIBILITY_FILE not in by_path: + failures.append( + "release manifest must deliver the native compatibility receipt" + ) + _check_compatibility( + release_dir, + manifest, + compatibility, + candidate, + require_compatibility, + compatibility_wheels, + failures, + ) + else: + failures.append( + "source enrichment compatibility status must be pending or passed" + ) + if require_compatibility: + try: + _check_producer_source_identity(_mapping(build.get("code"))) + except (OSError, ValueError) as exc: + failures.append(f"producer source identity failed: {exc}") + if failures: + raise ReleaseContractError(release_dir, failures) + return report + + +def _check_producer_source_identity(code: Mapping) -> None: + """Bind clean build provenance to committed and currently executed sources. + + Publication/certification runs from a Microcosm checkout. A later docs-only + commit is fine; each reviewed producer file must still have the exact bytes + recorded in the build and in its immutable producer commit. + """ + import subprocess + + from microcosm.data import contract, h5_enrichment + + commit = code.get("git_commit") + if ( + code.get("git_dirty") is not False + or not isinstance(commit, str) + or not (re.fullmatch(r"[0-9a-f]{40}", commit)) + ): + raise ValueError("publication requires a recorded clean producer commit") + hashes = _mapping(code.get("source_files_sha256")) + if set(hashes) != set(PRODUCER_SOURCE_FILES) or any( + not isinstance(value, str) or not _SHA256_RE.fullmatch(value) + for value in hashes.values() + ): + raise ValueError("producer must hash the exact reviewed source-file inventory") + + def git(*args: str) -> bytes: + try: + return subprocess.run( + ["git", *args], check=True, capture_output=True + ).stdout + except subprocess.CalledProcessError as exc: + raise ValueError( + "producer commit/source cannot be authenticated in the current Git checkout" + ) from exc + + checkout = Path(git("rev-parse", "--show-toplevel").decode().strip()) + verified_commit = ( + git("rev-parse", "--verify", f"{commit}^{{commit}}").decode().strip() + ) + if verified_commit != commit: + raise ValueError("producer git_commit does not resolve to the recorded commit") + for relative in PRODUCER_SOURCE_FILES: + committed = git("show", f"{commit}:{relative}") + if hashlib.sha256(committed).hexdigest() != hashes[relative]: + raise ValueError( + f"producer committed source differs from recorded hash: {relative}" + ) + current = checkout / relative + if not current.is_file() or sha256_file(current) != hashes[relative]: + raise ValueError( + f"producer checkout source differs from recorded hash: {relative}" + ) + executing = { + "packages/microcosm-data/src/microcosm/data/h5_enrichment.py": h5_enrichment.__file__, + "packages/microcosm-data/src/microcosm/data/source_enrichment.py": __file__, + "packages/microcosm-data/src/microcosm/data/contract.py": contract.__file__, + } + for relative, actual in executing.items(): + if actual is None or sha256_file(actual) != hashes[relative]: + raise ValueError( + f"executing producer contract differs from recorded source: {relative}" + ) + + +def _check_compatibility( + release_dir, + manifest, + compatibility, + candidate, + require_wheel_proof, + wheels, + failures, +): + receipt_path = release_dir / COMPATIBILITY_FILE + receipt = _json(receipt_path, failures) + if ( + compatibility.get("filename") != COMPATIBILITY_FILE + or not receipt_path.is_file() + or (compatibility.get("sha256") != sha256_file(receipt_path)) + ): + failures.append( + "source enrichment compatibility must hash-bind the actual test receipt" + ) + if candidate is None or not candidate.is_file(): + return + try: + actual = run_native_loader_compatibility( + candidate, require_wheels=require_wheel_proof, compatibility_wheels=wheels + ) + if receipt != actual: + failures.append( + "compatibility receipt differs from actual native loader tests/runtime" + ) + from microcosm.data.contract import _check_release_manifest + + _check_release_manifest(manifest, release_dir.name, failures) + for package, field in ( + ("policyengine-us", "model"), + ("policyengine-core", "core"), + ): + version = _mapping(actual.get("packages")).get(package, {}).get("version") + if _mapping( + _mapping(manifest.get("build")).get(f"built_with_{field}_package") + ) != {"name": package, "version": version}: + failures.append( + f"compatibility built-with {package} must match tested runtime" + ) + if manifest.get(f"compatible_{field}_packages") != [ + {"name": package, "specifier": f"=={version}"} + ]: + failures.append( + f"compatibility {package} must pin exactly the tested version" + ) + except (ValueError, OSError, ImportError, KeyError, TypeError) as exc: + failures.append(f"native loader compatibility failed: {exc}") + + +def _wheel_files(path: Path) -> tuple[str, str, dict[str, bytes]]: + from email.parser import BytesParser + from zipfile import ZipFile + + with ZipFile(path) as wheel: + metadata_files = [ + name for name in wheel.namelist() if name.endswith(".dist-info/METADATA") + ] + if len(metadata_files) != 1: + raise ValueError(f"{path.name} must contain one wheel METADATA file") + metadata = BytesParser().parsebytes(wheel.read(metadata_files[0])) + return ( + metadata["Name"], + metadata["Version"], + { + name: wheel.read(name) + for name in wheel.namelist() + if name.endswith(".py") + }, + ) + + +def _runtime_package_identities(compatibility_wheels, *, require_wheels: bool) -> dict: + """Hash actual installed source; optional wheel proof is offline, not PyPI proof.""" + from importlib import metadata + + names = COMPATIBILITY_PACKAGES + wheels = {} + for raw_path in compatibility_wheels: + path = Path(raw_path) + name, version, files = _wheel_files(path) + name = name.lower().replace("_", "-") + if name not in names or name in wheels: + raise ValueError( + "compatibility wheels must uniquely name country, Core, wrapper, and calculator" + ) + wheels[name] = (path, version, files) + if require_wheels and set(wheels) != set(names): + raise ValueError( + "compatibility requires exact installed policyengine-us, policyengine-core, policyengine, and spm-calculator wheels; candidate wheels may be tested before publication, whose external proof remains the publisher's gate" + ) + result = {} + for name in names: + dist = metadata.distribution(name) + direct_url = dist.read_text("direct_url.json") + source_files = sorted( + str(file) for file in (dist.files or ()) if str(file).endswith(".py") + ) + if not source_files: + raise ValueError( + f"cannot attest actual installed Python sources for {name}" + ) + digest = hashlib.sha256() + for relative in source_files: + path = Path(dist.locate_file(relative)) + digest.update(relative.encode() + b"\0" + path.read_bytes() + b"\0") + identity = { + "version": dist.version, + "source_sha256": digest.hexdigest(), + "direct_url": json.loads(direct_url) if direct_url else None, + "wheel_sha256": None, + } + if name in wheels: + path, version, files = wheels[name] + if version != dist.version: + raise ValueError( + f"{name} wheel version differs from the tested install" + ) + for relative, expected in files.items(): + installed = Path(dist.locate_file(relative)) + if not installed.is_file() or installed.read_bytes() != expected: + raise ValueError( + f"{name} installed source differs from wheel: {relative}" + ) + if set(source_files) != set(files): + raise ValueError( + f"{name} installed source inventory differs from wheel" + ) + if direct_url and json.loads(direct_url).get("dir_info", {}).get( + "editable" + ): + raise ValueError( + f"{name} editable installation is not a publishable wheel identity" + ) + identity["wheel_sha256"] = sha256_file(path) + result[name] = identity + return result + + +def run_native_loader_compatibility( + candidate_h5: Path | str, + *, + require_wheels: bool = False, + compatibility_wheels: tuple[Path | str, ...] = (), +) -> dict: + """Run fixed native-loader tests and one complete-household input-precedence probe. + + This proves native delivery, not canonical SPM numerical-model acceptance or + publication of package versions. Those remain the root release gates. An + installed wheel is checked against its source bytes, never a claimed bool. + """ + import inspect + + import h5py + import numpy as np + from policyengine.tax_benefit_models.us.datasets import PolicyEngineUSDataset + from policyengine_us.data import USSingleYearDataset + from policyengine_us.system import system + + candidate_h5 = Path(candidate_h5) + packages = _runtime_package_identities( + compatibility_wheels, require_wheels=require_wheels + ) + variable = system.variables.get(ROLE_VARIABLE) + if ( + variable is None + or variable.entity.key != "person" + or variable.value_type is not bool + ): + raise ValueError( + f"tested country model must register {ROLE_VARIABLE} as a native person bool" + ) + source_paths = { + "country_loader": inspect.getsourcefile(USSingleYearDataset), + "wrapper_loader": inspect.getsourcefile(PolicyEngineUSDataset), + "native_role": inspect.getsourcefile(type(variable)), + } + if require_wheels: + _check_loaded_source_ownership(source_paths) + country = USSingleYearDataset(file_path=str(candidate_h5)) + wrapper = PolicyEngineUSDataset( + name="native_spm_role_compatibility", + description="Read-only native SPM role compatibility probe", + filepath=str(candidate_h5), + year=2024, + ) + checked = [] + with h5py.File(candidate_h5, "r") as h5: + for entity in ( + "person", + "household", + "tax_unit", + "spm_unit", + "family", + "marital_unit", + ): + table = h5[f"{entity}/table"] + columns = [f"{entity}_id"] + weight_column = f"{entity}_weight" + if weight_column in table.dtype.names: + columns.append(weight_column) + if entity == "person": + columns += [ROLE_VARIABLE] + [ + name + for name in table.dtype.names + if name.startswith("person_") and name.endswith("_id") + ] + for column in dict.fromkeys(columns): + expected = table[column] + if column == ROLE_VARIABLE: + expected = expected.astype(bool) + for loader, frame in ( + ("country", getattr(country, entity)), + ("wrapper", getattr(wrapper.data, entity)), + ): + observed = frame[column].to_numpy() + if ( + observed.dtype != expected.dtype + or observed.shape != expected.shape + or observed.tobytes() != expected.tobytes() + ): + raise ValueError( + f"{loader} native loader changed {entity}.{column}" + ) + checked.append(f"{loader}:{entity}.{column}") + if not np.asarray(country.person[ROLE_VARIABLE]).dtype == np.dtype(bool): + raise ValueError("country native role dtype must remain bool") + # The country engine may provide a household-role fallback for surveys + # without this primitive. Opposite supplied values on one complete native + # household must both reach Core unchanged, overriding any fixed fallback. + _check_native_input_precedence(country) + return { + "schema_version": 1, + "status": "passed", + "scope": "native_input_loading_only", + "external_package_publication": "not_attested", + "canonical_spm_model_acceptance": "not_attested", + "dataset_sha256": sha256_file(candidate_h5), + "runner_sha256": sha256_file(__file__), + "packages": packages, + "loaded_source_packages": LOADED_SOURCE_PACKAGES.copy(), + "loaded_source_sha256": { + key: sha256_file(path) for key, path in source_paths.items() + }, + "checks": [ + "country:registered_person_bool_input", + "country:complete_household_native_input_overrides_default", + *checked, + ], + } + + +def _check_loaded_source_ownership(source_paths: Mapping) -> None: + """The country registers the native role supplied by the calculator wheel.""" + from importlib import import_module, metadata + + for package in COMPATIBILITY_PACKAGES: + module_name = package.replace("-", "_") + module = import_module(module_name) + expected_root = Path( + metadata.distribution(package).locate_file(module_name) + ).resolve() + if not Path(module.__file__).resolve().is_relative_to(expected_root): + raise ValueError( + f"tested {package} import does not come from the verified installed wheel" + ) + for label, package in LOADED_SOURCE_PACKAGES.items(): + expected_root = Path( + metadata.distribution(package).locate_file(package.replace("-", "_")) + ).resolve() + source = source_paths.get(label) + if source is None or not Path(source).resolve().is_relative_to(expected_root): + raise ValueError( + f"tested {label} source does not come from the verified {package} wheel" + ) + + +def _check_native_input_precedence(country) -> None: + import numpy as np + from policyengine_us import Microsimulation + from policyengine_us.data import USSingleYearDataset + + household_id = country.household["household_id"].iloc[0] + person = country.person.loc[ + country.person["person_household_id"] == household_id + ].copy() + if person.empty: + raise ValueError("native compatibility fixture has no complete household") + tables = {"person": person, "time_period": 2024} + for entity in ("household", "tax_unit", "spm_unit", "family", "marital_unit"): + ids = person[f"person_{entity}_id"].unique() + table = getattr(country, entity) + tables[entity] = table.loc[table[f"{entity}_id"].isin(ids)].copy() + if entity == "spm_unit": + if ( + not country.person.loc[ + country.person["person_spm_unit_id"].isin(ids), + "person_household_id", + ] + .eq(household_id) + .all() + ): + raise ValueError("native compatibility household cuts an SPM unit") + for supplied_value in (False, True): + supplied = np.full(len(person), supplied_value, dtype=bool) + tables["person"] = person.assign(**{ROLE_VARIABLE: supplied}) + simulation = Microsimulation(dataset=USSingleYearDataset(**tables)) + observed = np.asarray(simulation.calculate(ROLE_VARIABLE, 2024)) + if observed.dtype != supplied.dtype or not np.array_equal(observed, supplied): + raise ValueError("country Core discarded the supplied native person role") + + +def certify_source_enrichment( + release_dir: Path | str, + output_dir: Path | str, + *, + parent_h5: Path | str, + artifact_root: Path | str, + compatibility_wheels: tuple[Path | str, ...], +) -> Path: + """Create a separate certified bundle only after measured loader checks pass. + + H5 and source evidence are never modified. The caller still owns canonical + model acceptance, package publication proof, and publisher authorization. + """ + import shutil + import tempfile + from importlib import metadata + + release_dir, output_dir = Path(release_dir), Path(output_dir) + if output_dir.exists() or output_dir.name != release_dir.name: + raise ValueError( + "output_dir must be new and keep the candidate release id as its basename" + ) + report = validate_source_enrichment_candidate( + release_dir, parent_h5=parent_h5, artifact_root=artifact_root + ) + candidate = Path(artifact_root) / report["dataset"]["filename"] + receipt = run_native_loader_compatibility( + candidate, require_wheels=True, compatibility_wheels=compatibility_wheels + ) + output_dir.parent.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory( + prefix=".source-enrichment-", dir=output_dir.parent + ) as staging: + staged = Path(staging) / release_dir.name + shutil.copytree(release_dir, staged) + + def write(name, value): + (staged / name).write_text( + json.dumps(value, indent=2, sort_keys=True) + "\n" + ) + + write(COMPATIBILITY_FILE, receipt) + report["compatibility"] = { + "status": "passed", + "filename": COMPATIBILITY_FILE, + "sha256": sha256_file(staged / COMPATIBILITY_FILE), + } + write(SOURCE_ENRICHMENT_FILE, report) + manifest = json.loads((staged / "release_manifest.json").read_text()) + for package, field in ( + ("policyengine-us", "model"), + ("policyengine-core", "core"), + ): + version = receipt["packages"][package]["version"] + manifest["build"][f"built_with_{field}_package"] = { + "name": package, + "version": version, + } + manifest[f"compatible_{field}_packages"] = [ + {"name": package, "specifier": f"=={version}"} + ] + manifest["data_package"] = { + "name": "microcosm-data", + "version": metadata.version("microcosm-data"), + } + for entry in manifest["artifacts"].values(): + if entry["path"] == SOURCE_ENRICHMENT_FILE: + entry["sha256"] = sha256_file(staged / SOURCE_ENRICHMENT_FILE) + manifest["artifacts"]["source_enrichment_compatibility"] = { + "kind": "diagnostics", + "path": COMPATIBILITY_FILE, + "repo_id": "policyengine/populace-us", + "revision": release_dir.name, + "sha256": sha256_file(staged / COMPATIBILITY_FILE), + } + write("release_manifest.json", manifest) + validate_source_enrichment_candidate( + staged, + parent_h5=parent_h5, + artifact_root=artifact_root, + require_compatibility=True, + compatibility_wheels=compatibility_wheels, + ) + staged.rename(output_dir) + return output_dir + + +def main(argv: list[str] | None = None) -> int: + import argparse + import sys + + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--release-dir", required=True, type=Path) + parser.add_argument("--parent-h5", required=True, type=Path) + parser.add_argument("--artifact-root", required=True, type=Path) + parser.add_argument("--certify", action="store_true") + parser.add_argument("--output-dir", type=Path) + parser.add_argument("--require-compatibility", action="store_true") + parser.add_argument("--compatibility-wheel", action="append", type=Path, default=[]) + args = parser.parse_args(argv) + if args.certify and args.output_dir is None: + parser.error( + "--certify requires a new --output-dir ending in the same release id" + ) + try: + if args.certify: + result = certify_source_enrichment( + args.release_dir, + args.output_dir, + parent_h5=args.parent_h5, + artifact_root=args.artifact_root, + compatibility_wheels=tuple(args.compatibility_wheel), + ) + print(json.dumps({"certified_bundle": str(result), "published": False})) + else: + report = validate_source_enrichment_candidate( + args.release_dir, + parent_h5=args.parent_h5, + artifact_root=args.artifact_root, + require_compatibility=args.require_compatibility, + compatibility_wheels=tuple(args.compatibility_wheel), + ) + print( + json.dumps( + {"valid": True, "compatibility": report["compatibility"]["status"]} + ) + ) + except (ValueError, OSError, ImportError, KeyError, TypeError) as exc: + print(str(exc), file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/packages/microcosm-data/tests/test_h5_enrichment.py b/packages/microcosm-data/tests/test_h5_enrichment.py new file mode 100644 index 000000000..e6763b18b --- /dev/null +++ b/packages/microcosm-data/tests/test_h5_enrichment.py @@ -0,0 +1,403 @@ +"""Native source enrichment preserves existing HDF inputs byte for byte.""" + +from __future__ import annotations + +import hashlib +import pickle +import time + +import h5py +import numpy as np +import pytest + +from microcosm.data.h5_enrichment import ( + ROLE, + append_native_spm_role, + compare_h5_enrichment, + file_sha256, +) + +pd = pytest.importorskip("pandas") +pytest.importorskip("tables") + + +@pytest.fixture +def parent(tmp_path): + """An indexed pandas table, native entities, and unusual float payloads.""" + path = tmp_path / "parent.h5" + payloads = np.array( + [0x7FF8000000000051, 0x8000000000000000, 0, 0x3FF0000000000000], + dtype=np.uint64, + ).view(np.float64) + frames = { + "person": pd.DataFrame( + { + "person_id": np.array([101, 102, 103, 104], dtype=np.int64), + "person_household_id": np.array([11, 11, 12, 12], dtype=np.int64), + "person_spm_unit_id": np.array([21, 21, 22, 23], dtype=np.int64), + "age": np.array([17, 40, 16, 18], dtype=np.int16), + "original_bool": np.array([True, False, False, True]), + "person_weight": np.array([1.25, 1.25, 2.5, 2.5]), + "float_payload": payloads, + "label": ["alpha", "beta", "gamma", "delta"], + }, + index=pd.Index([9, 3, 8, 1], name="source_index"), + ), + "household": pd.DataFrame( + {"household_id": [11, 12], "household_weight": [1.25, 2.5]} + ), + "spm_unit": pd.DataFrame({"spm_unit_id": [21, 22, 23]}), + "tax_unit": pd.DataFrame({"tax_unit_id": [31, 32]}), + "family": pd.DataFrame({"family_id": [41, 42]}), + "marital_unit": pd.DataFrame({"marital_unit_id": [51, 52]}), + "_time_period": pd.Series([2024]), + } + with pd.HDFStore(path, mode="w") as store: + for key, frame in frames.items(): + store.put(key, frame, format="table", data_columns=True) + with h5py.File(path, "r+") as h5: + h5.attrs["unchanged_attribute"] = np.int16(7) + h5["person"].attrs["empty_attribute"] = h5py.Empty("i8") + auxiliary = h5.create_group("auxiliary") + auxiliary.create_dataset("array", data=np.array([7, 8], dtype=np.int16)) + auxiliary.create_dataset("scalar", data=payloads[0]) + auxiliary.create_dataset("empty", shape=(0,), dtype="f8") + return path + + +@pytest.fixture +def enriched(parent, tmp_path): + candidate = tmp_path / "candidate.h5" + role = np.array([True, False, True, False]) + report = append_native_spm_role( + parent, candidate, role, expected_parent_sha256=file_sha256(parent) + ) + return parent, candidate, role, report + + +def _replace_dataset(group, name, datatype, values): + """Surgically tamper one datatype while retaining storage and attributes.""" + old = group[name] + creation = old.id.get_create_plist().copy() + new = h5py.Dataset( + h5py.h5d.create( + group.id, + b"_tampered", + datatype, + old.id.get_space(), + dcpl=creation, + ) + ) + _write_records(new, values) + for key in old.attrs: + attribute = old.attrs.get_id(key) + copied = h5py.h5a.create( + new.id, key.encode(), attribute.get_type(), attribute.get_space() + ) + value = old.attrs[key] + if not isinstance(value, h5py.Empty): + copied.write(np.asarray(value), mtype=attribute.get_type()) + del group[name] + group.move("_tampered", name) + + +def _write_records(table, values, start=0): + """Mutate only the requested bytes, avoiding fixed-string conversions.""" + values = np.ascontiguousarray(values) + selection = table.id.get_space() + selection.select_hyperslab((start,), (values.size,)) + table.id.write( + h5py.h5s.create_simple(values.shape), + selection, + values, + mtype=table.id.get_type(), + ) + + +def test_native_pandas_bool_and_exact_original_inputs(enriched): + parent, candidate, role, report = enriched + original = pd.read_hdf(parent, "person") + actual = pd.read_hdf(candidate, "person") + assert actual.columns.tolist() == [*original.columns, ROLE] + assert actual[ROLE].dtype == np.dtype(bool) + np.testing.assert_array_equal(actual[ROLE].to_numpy(), role) + pd.testing.assert_frame_equal(actual.drop(columns=ROLE), original, check_exact=True) + for name in ("household", "spm_unit", "tax_unit", "family", "marital_unit"): + pd.testing.assert_frame_equal( + pd.read_hdf(candidate, name), pd.read_hdf(parent, name), check_exact=True + ) + with h5py.File(parent, "r") as before, h5py.File(candidate, "r") as after: + a, b = before["person/table"], after["person/table"] + assert a.dtype.itemsize + 1 == b.dtype.itemsize + assert b.dtype.names == (*a.dtype.names, ROLE) + for name in a.dtype.names: + assert a.dtype.fields[name] == b.dtype.fields[name] + assert a[name].tobytes() == b[name].tobytes() + # Equality must retain both NaN payload and signed-zero bit patterns. + assert b["float_payload"].view(np.uint64).tolist() == [ + 0x7FF8000000000051, + 0x8000000000000000, + 0, + 0x3FF0000000000000, + ] + assert set(after["person/_i_table"]) == set(before["person/_i_table"]) + # Existing pandas indexes remain usable after the record gets one byte wider. + pd.testing.assert_frame_equal( + pd.read_hdf(candidate, "person", where="person_id == 102").drop(columns=ROLE), + pd.read_hdf(parent, "person", where="person_id == 102"), + check_exact=True, + ) + assert file_sha256(parent) == report["parent_sha256"] + assert report["all_preexisting_variables_exact"] is True + assert report["persons"] == len(role) + assert report["role_sha256"] == hashlib.sha256(role.tobytes()).hexdigest() + assert report["candidate_sha256"] == file_sha256(candidate) + + +def test_enrichment_is_byte_reproducible(parent, tmp_path): + first, second = tmp_path / "first.h5", tmp_path / "second.h5" + role = np.array([True, False, True, False]) + parent_sha = file_sha256(parent) + append_native_spm_role(parent, first, role, expected_parent_sha256=parent_sha) + # HDF object timestamps are stored at second precision; cross that boundary. + time.sleep(1.1) + append_native_spm_role(parent, second, role, expected_parent_sha256=parent_sha) + assert file_sha256(first) == file_sha256(second) + + +@pytest.mark.parametrize( + "field", + ["person_id", "person_household_id", "person_spm_unit_id", "person_weight", "age"], +) +def test_old_person_values_cannot_change(enriched, field): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + table = h5["person/table"] + row = table[0] + row[field] += 1 + _write_records(table, np.asarray(row).reshape(1)) + with pytest.raises(ValueError, match=f"Existing person field changed: {field}"): + compare_h5_enrichment(parent, candidate) + + +@pytest.mark.parametrize("mutation", ["dtype", "shape", "scalar_value"]) +def test_nonperson_dtypes_shapes_and_scalar_values_are_exact(enriched, mutation): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + if mutation == "scalar_value": + h5["auxiliary/scalar"][()] = 1.0 + else: + del h5["auxiliary/array"] + h5["auxiliary"].create_dataset( + "array", + data=np.array( + [7, 8] if mutation == "dtype" else [7], + dtype=np.int64 if mutation == "dtype" else np.int16, + ), + ) + with pytest.raises( + ValueError, + match="HDF dtype changed|HDF shape/storage changed|HDF dataset values changed", + ): + compare_h5_enrichment(parent, candidate) + + +@pytest.mark.parametrize("replacement", [0x7FF8000000000052, 0]) +def test_float_payload_and_signed_zero_changes_are_rejected(enriched, replacement): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + table = h5["person/table"] + index = 0 if replacement else 1 + row = table[index] + row["float_payload"] = np.array(replacement, dtype=np.uint64).view(np.float64) + _write_records(table, np.asarray(row).reshape(1), start=index) + with pytest.raises( + ValueError, match="Existing person field changed: float_payload" + ): + compare_h5_enrichment(parent, candidate) + + +@pytest.mark.parametrize( + "target", ["household/table", "person/_i_table/person_id/sorted"] +) +def test_weights_and_existing_indexes_cannot_change(enriched, target): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + dataset = h5[target] + if target == "household/table": + row = dataset[0] + row["household_weight"] += 1 + _write_records(dataset, np.asarray(row).reshape(1)) + else: + # The index is allocated even when the small table has no full slice. + dataset.attrs["DIRTY"] = np.uint8(1) + with pytest.raises(ValueError, match="HDF (dataset values|attributes) changed"): + compare_h5_enrichment(parent, candidate) + + +@pytest.mark.parametrize("mutation", ["value", "dtype", "shape", "removed", "added"]) +def test_original_attributes_cannot_change(enriched, mutation): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + if mutation == "removed": + del h5.attrs["unchanged_attribute"] + elif mutation == "added": + h5.attrs["unapproved"] = 1 + else: + h5.attrs["unchanged_attribute"] = { + "value": np.int16(8), + "dtype": np.int64(7), + "shape": np.array([7], dtype=np.int16), + }[mutation] + with pytest.raises(ValueError, match="HDF attribute"): + compare_h5_enrichment(parent, candidate) + + +@pytest.mark.parametrize( + "attribute", ["data_columns", "values_cols", "non_index_axes", "info"] +) +def test_registration_changes_must_only_append_the_new_role(enriched, attribute): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + value = pickle.loads(h5["person"].attrs[attribute]) + if attribute == "info": + value["person_id"] = {"changed": True} + elif attribute == "non_index_axes": + value[0][1].reverse() + else: + value.reverse() + h5["person"].attrs[attribute] = np.bytes_(pickle.dumps(value, protocol=0)) + with pytest.raises(ValueError, match="pandas column-registration change"): + compare_h5_enrichment(parent, candidate) + + +@pytest.mark.parametrize("attribute", [f"{ROLE}_dtype", "values_cols"]) +def test_metadata_bytes_with_wrong_type_and_shape_are_rejected(enriched, attribute): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + owner = h5["person"] if attribute == "values_cols" else h5["person/table"] + raw = bytes(owner.attrs[attribute]) + owner.attrs[attribute] = np.frombuffer(raw, dtype=np.uint8) + with pytest.raises(ValueError): + compare_h5_enrichment(parent, candidate) + + +@pytest.mark.parametrize("mutation", ["dtype", "order", "role_dtype"]) +def test_compound_hdf_types_and_field_order_are_exact(enriched, mutation): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + table = h5["person/table"] + old_type = table.id.get_type() + datatype = h5py.h5t.create(h5py.h5t.COMPOUND, old_type.get_size()) + indices = list(range(old_type.get_nmembers())) + if mutation == "order": + indices.reverse() + for index in indices: + name = old_type.get_member_name(index) + field_type = old_type.get_member_type(index) + if (mutation == "dtype" and name == b"original_bool") or ( + mutation == "role_dtype" and name == ROLE.encode() + ): + # NumPy still sees uint8, but the required HDF bitfield type is lost. + field_type = h5py.h5t.STD_U8LE + datatype.insert(name, old_type.get_member_offset(index), field_type) + values = table[:] + _replace_dataset(h5["person"], "table", datatype, values) + with pytest.raises(ValueError, match="Person dtype differs"): + compare_h5_enrichment(parent, candidate) + + +@pytest.mark.parametrize( + "mutation", ["extra_group", "extra_dataset", "removed", "soft_link", "alias"] +) +def test_object_inventory_is_exact(enriched, mutation): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + if mutation == "extra_group": + h5.create_group("extra") + elif mutation == "extra_dataset": + h5.create_dataset("extra", data=[1]) + elif mutation == "removed": + del h5["family"] + elif mutation == "soft_link": + h5["extra"] = h5py.SoftLink("/person") + else: + h5["extra"] = h5["person"] + with pytest.raises( + ValueError, match="added or removed|Nonlocal HDF link|Aliased HDF object" + ): + compare_h5_enrichment(parent, candidate) + + +def test_row_order_cannot_change(enriched): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + table = h5["person/table"] + _write_records(table, table[:][::-1]) + with pytest.raises(ValueError, match="Existing person field changed"): + compare_h5_enrichment(parent, candidate) + + +def test_non_boolean_role_values_are_rejected(enriched): + parent, candidate, _, _ = enriched + with h5py.File(candidate, "r+") as h5: + table = h5["person/table"] + row = table[0] + row[ROLE] = 2 + _write_records(table, np.asarray(row).reshape(1)) + with pytest.raises(ValueError, match="Native SPM role is not Boolean"): + compare_h5_enrichment(parent, candidate) + + +@pytest.mark.parametrize( + "role", + [np.ones(4, dtype=np.uint8), np.ones((4, 1), dtype=bool), np.ones(3, dtype=bool)], +) +def test_invalid_role_input_leaves_no_candidate(parent, tmp_path, role): + candidate = tmp_path / "candidate.h5" + parent_sha = file_sha256(parent) + with pytest.raises(ValueError, match="one-dimensional Boolean|Role coverage"): + append_native_spm_role( + parent, candidate, role, expected_parent_sha256=parent_sha + ) + assert not candidate.exists() + assert file_sha256(parent) == parent_sha + + +def test_wrong_parent_hash_leaves_no_candidate(parent, tmp_path): + candidate = tmp_path / "candidate.h5" + with pytest.raises(ValueError, match="Parent H5 SHA-256 mismatch"): + append_native_spm_role( + parent, candidate, np.ones(4, dtype=bool), expected_parent_sha256="0" * 64 + ) + assert not candidate.exists() + + +@pytest.mark.parametrize("destination", ["existing", "parent", "symlink"]) +def test_existing_destination_is_never_overwritten(parent, tmp_path, destination): + candidate = tmp_path / "candidate.h5" + if destination == "existing": + candidate.write_bytes(b"existing artifact") + elif destination == "parent": + candidate = parent + else: + candidate.symlink_to(parent) + original = candidate.read_bytes() + with pytest.raises(FileExistsError): + append_native_spm_role( + parent, + candidate, + np.ones(4, dtype=bool), + expected_parent_sha256=file_sha256(parent), + ) + assert candidate.read_bytes() == original + + +def test_already_enriched_parent_is_rejected(enriched, tmp_path): + _, candidate, role, _ = enriched + second = tmp_path / "second.h5" + with pytest.raises(ValueError, match="already contains"): + append_native_spm_role( + candidate, second, role, expected_parent_sha256=file_sha256(candidate) + ) + assert not second.exists() diff --git a/packages/microcosm-data/tests/test_source_enrichment.py b/packages/microcosm-data/tests/test_source_enrichment.py new file mode 100644 index 000000000..4757e82d8 --- /dev/null +++ b/packages/microcosm-data/tests/test_source_enrichment.py @@ -0,0 +1,599 @@ +"""The inheritance lane preserves calibration and rejects rehashed tampering.""" + +from __future__ import annotations + +import json + +import h5py +import numpy as np +import pytest + +from microcosm.data import source_enrichment as enrichment +from microcosm.data.contract import ReleaseContractError, validate_release_dir +from microcosm.data.h5_enrichment import append_native_spm_role, compare_h5_enrichment +from microcosm.data.publish_cli import main as publish_main +from microcosm.data.release import publish_release + + +def _write(path, payload): + path.write_text(json.dumps(payload, sort_keys=True)) + + +@pytest.fixture +def candidate(tmp_path, monkeypatch): + pd = pytest.importorskip("pandas") + pytest.importorskip("tables") + parent = tmp_path / "parent.h5" + people = pd.DataFrame( + { + "person_id": [1, 2, 3], + "person_spm_unit_id": [1, 1, 2], + "person_household_id": [1, 1, 2], + "person_weight": [2.0, 2.0, 1.0], + "age": [12, 16, 45], + } + ) + with pd.HDFStore(parent, "w") as store: + store.put("person", people, format="table", data_columns=True) + for entity in ("household", "spm_unit", "tax_unit", "family", "marital_unit"): + store.put( + entity, + pd.DataFrame({f"{entity}_id": [1, 2], f"{entity}_weight": [2.0, 1.0]}), + format="table", + data_columns=True, + ) + artifact_root = tmp_path / "artifacts" + artifact_root.mkdir() + h5 = artifact_root / "populace_us_2024.h5" + append_native_spm_role( + parent, + h5, + np.array([False, True, True]), + expected_parent_sha256=enrichment.sha256_file(parent), + ) + release = tmp_path / "populace-us-2024-test-enrichment" + release.mkdir() + parent_payloads = { + "parent_release_manifest.json": {"schema_version": 1}, + "parent_build_manifest.json": {"calibration": {"old": True}}, + "calibration_diagnostics.json": {"schema_version": 5, "targets": []}, + "us_source_coverage.json": {"schema_version": 1}, + } + for name, payload in parent_payloads.items(): + _write(release / name, payload) + pins = {name: enrichment.sha256_file(release / name) for name in parent_payloads} + monkeypatch.setattr(enrichment, "PARENT_FILES", pins) + monkeypatch.setattr( + enrichment, "PARENT_DATASET_SHA256", enrichment.sha256_file(parent) + ) + counts = { + "persons_joined": 3, + "native_spm_units": 2, + "total_source_people": 3, + "total_source_units": 2, + "minor_only_units_resolved": 0, + "classification_changed_units_vs_age_only": 1, + } + monkeypatch.setattr(enrichment, "EXPECTED_COUNTS", counts) + monkeypatch.setattr(enrichment, "CENSUS_PERSON_PINS", {2025: "a" * 64}) + evidence = release / enrichment.SOURCE_EVIDENCE_FILE + evidence.write_text( + "person_id,person_spm_unit_id,is_spm_independence_role\n1,1,False\n2,1,True\n3,2,True\n" + ) + monkeypatch.setattr( + enrichment, "SOURCE_EVIDENCE_SHA256", enrichment.sha256_file(evidence) + ) + provenance = { + **counts, + "dataset_sha256": enrichment.PARENT_DATASET_SHA256, + "unmatched_persons": 0, + "adult_child_person_count_mismatch_units": 0, + "complete_source_membership_units": 2, + "weights_used": False, + "ages_changed": False, + "primitive_column": enrichment.ROLE_VARIABLE, + "evidence_column": enrichment.EVIDENCE_COLUMN, + "source_checks": [ + { + "survey_year": 2025, + "csv_sha256": "a" * 64, + "adult_child_person_count_mismatch_units": 0, + "official_archive_url": "https://www2.census.gov/test.zip", + "archive_sha256": "b" * 64, + } + ], + } + _write(release / enrichment.SOURCE_PROVENANCE_FILE, provenance) + dataset = {"filename": h5.name, "sha256": enrichment.sha256_file(h5)} + build = { + "build_id": release.name, + "release_type": "source_enrichment", + "dataset": dataset, + "code": {"git_commit": "a" * 40, "git_dirty": False}, + "calibration": { + "mode": "inherited", + "parent_build_id": enrichment.PARENT_BUILD_ID, + "diagnostics_sha256": pins["calibration_diagnostics.json"], + "diagnostics_schema_version": 5, + }, + } + _write(release / "build_manifest.json", build) + report = { + "schema_version": 1, + "release_type": "source_enrichment", + "operation": "add_native_spm_independent_minor_role", + "parent": { + "build_id": enrichment.PARENT_BUILD_ID, + "repo_id": "policyengine/populace-us", + "revision": enrichment.PARENT_BUILD_ID, + "dataset_sha256": enrichment.PARENT_DATASET_SHA256, + "calibration_diagnostics_schema_version": 5, + "files": pins, + }, + "dataset": dataset, + "added_variable": { + "name": enrichment.ROLE_VARIABLE, + "entity": "person", + "dtype": "bool", + "age_gate_applied": False, + }, + "preservation": compare_h5_enrichment(parent, h5), + "source": { + "person_evidence_filename": enrichment.SOURCE_EVIDENCE_FILE, + "person_evidence_sha256": enrichment.sha256_file(evidence), + "provenance_filename": enrichment.SOURCE_PROVENANCE_FILE, + "provenance_sha256": enrichment.sha256_file( + release / enrichment.SOURCE_PROVENANCE_FILE + ), + }, + "reconciliation": provenance, + "compatibility": {"status": "pending"}, + } + _write(release / enrichment.SOURCE_ENRICHMENT_FILE, report) + manifest = { + "schema_version": 1, + "release_type": "source_enrichment", + "build": {"build_id": release.name}, + "compatible_core_packages": [], + "compatible_model_packages": [], + "default_datasets": {"national": "dataset"}, + "artifacts": {}, + } + for path in [ + *map(lambda name: release / name, pins), + release / enrichment.SOURCE_ENRICHMENT_FILE, + evidence, + release / enrichment.SOURCE_PROVENANCE_FILE, + h5, + ]: + manifest["artifacts"]["dataset" if path == h5 else path.stem] = { + "kind": "microdata" if path == h5 else "diagnostics", + "path": path.name, + "repo_id": "policyengine/populace-us", + "revision": release.name, + "sha256": enrichment.sha256_file(path), + } + _write(release / "release_manifest.json", manifest) + return release, parent, artifact_root + + +def _validate(candidate, **kwargs): + release, parent, root = candidate + return enrichment.validate_source_enrichment_candidate( + release, parent_h5=parent, artifact_root=root, **kwargs + ) + + +def _refresh(release, filename): + manifest_path = release / "release_manifest.json" + manifest = json.loads(manifest_path.read_text()) + for entry in manifest["artifacts"].values(): + if entry["path"] == filename: + entry["sha256"] = enrichment.sha256_file(release / filename) + _write(manifest_path, manifest) + + +def test_pending_candidate_is_valid_but_actual_release_gate_refuses(candidate): + assert _validate(candidate)["compatibility"]["status"] == "pending" + release, parent, root = candidate + with pytest.raises(ReleaseContractError, match="compatibility is pending"): + validate_release_dir(release, parent_h5=parent, artifact_root=root) + + +def test_publisher_refuses_before_constructing_hub_client(candidate, monkeypatch): + import microcosm.data.release as release_module + + def no_hub(): + pytest.fail("publisher contacted Hub for a pending enrichment") + + monkeypatch.setattr(release_module, "_hf_api", no_hub) + release, parent, root = candidate + with pytest.raises(ReleaseContractError, match="compatibility is pending"): + publish_release( + release, "policyengine/populace-us", parent_h5=parent, artifact_root=root + ) + + +def test_preflight_runs_actual_gate_and_never_publishes(candidate, monkeypatch): + import microcosm.data.publish_cli as cli + + monkeypatch.setattr( + cli, "publish_release", lambda *a, **kw: pytest.fail("preflight published") + ) + release, parent, root = candidate + with pytest.raises(ReleaseContractError, match="compatibility is pending"): + publish_main( + [ + str(release), + "--parent-h5", + str(parent), + "--artifact-root", + str(root), + "--preflight-only", + ] + ) + + +def test_requires_actual_parent_and_h5(candidate): + release, _, root = candidate + with pytest.raises(ReleaseContractError, match="requires parent_h5"): + enrichment.validate_source_enrichment_candidate( + release, parent_h5=None, artifact_root=root + ) + + +def test_schema_five_cannot_be_relabelled_even_with_rehashed_manifests(candidate): + release, _, _ = candidate + diagnostics = release / "calibration_diagnostics.json" + _write(diagnostics, {"schema_version": 6, "targets": []}) + _refresh(release, diagnostics.name) + with pytest.raises(ReleaseContractError, match="reviewed parent bytes"): + _validate(candidate) + + +@pytest.mark.parametrize("field", ["person_weight", "person_spm_unit_id", "age"]) +def test_preservation_rejects_changed_old_values(candidate, field): + release, _, root = candidate + h5 = root / "populace_us_2024.h5" + with h5py.File(h5, "r+") as handle: + table = handle["person/table"] + row = table[0] + row[field] += 1 + table[0] = row + with pytest.raises(ReleaseContractError, match="exact H5 preservation failed"): + _validate(candidate) + + +def test_rehashed_role_evidence_attack_still_fails_reviewed_source_pin(candidate): + release, parent, root = candidate + path = release / enrichment.SOURCE_EVIDENCE_FILE + path.write_text(path.read_text().replace("1,1,False", "1,1,True")) + h5 = root / "populace_us_2024.h5" + with h5py.File(h5, "r+") as handle: + table = handle["person/table"] + row = table[0] + row[enrichment.ROLE_VARIABLE] = True + table[0] = row + report_path = release / enrichment.SOURCE_ENRICHMENT_FILE + report = json.loads(report_path.read_text()) + report["source"]["person_evidence_sha256"] = enrichment.sha256_file(path) + report["dataset"]["sha256"] = enrichment.sha256_file(h5) + report["preservation"] = compare_h5_enrichment(parent, h5) + _write(report_path, report) + _refresh(release, path.name) + _refresh(release, report_path.name) + with pytest.raises( + ReleaseContractError, match="independently reviewed Census-derived" + ): + _validate(candidate) + + +def test_pending_candidate_cannot_copy_old_model_claims(candidate): + release, _, _ = candidate + path = release / "release_manifest.json" + manifest = json.loads(path.read_text()) + manifest["build"]["built_with_model_package"] = { + "name": "policyengine-us", + "version": "1.764.6", + } + manifest["compatible_model_packages"] = [ + {"name": "policyengine-us", "specifier": "==1.764.6"} + ] + _write(path, manifest) + with pytest.raises( + ReleaseContractError, match="must not claim model/Core compatibility" + ): + _validate(candidate) + + +def test_fabricated_passed_receipt_is_replayed(candidate, monkeypatch): + release, _, _ = candidate + report_path = release / enrichment.SOURCE_ENRICHMENT_FILE + report = json.loads(report_path.read_text()) + receipt = release / enrichment.COMPATIBILITY_FILE + _write(receipt, {"status": "passed"}) + report["compatibility"] = { + "status": "passed", + "filename": receipt.name, + "sha256": enrichment.sha256_file(receipt), + } + _write(report_path, report) + _refresh(release, report_path.name) + calls = [] + + def actual_test(*args, **kwargs): + calls.append(kwargs) + raise ValueError("actual country role is unavailable") + + monkeypatch.setattr(enrichment, "run_native_loader_compatibility", actual_test) + with pytest.raises( + ReleaseContractError, match="actual country role is unavailable" + ): + _validate(candidate, require_compatibility=True) + assert calls == [{"require_wheels": True, "compatibility_wheels": ()}] + + +def test_unknown_release_type_cannot_fall_back_to_calibration(candidate): + release, _, _ = candidate + path = release / "release_manifest.json" + manifest = json.loads(path.read_text()) + manifest["release_type"] = "trust_parent" + _write(path, manifest) + with pytest.raises(ReleaseContractError, match="unknown release_type"): + validate_release_dir(release) + + +def test_certification_writes_new_bundle_and_preflight_replays( + candidate, tmp_path, monkeypatch +): + from importlib import metadata + + release, parent, root = candidate + receipt = { + "status": "passed", + "dataset_sha256": enrichment.sha256_file(root / "populace_us_2024.h5"), + "packages": { + "policyengine-us": {"version": "1.999.0"}, + "policyengine-core": {"version": "3.99.0"}, + "policyengine": {"version": "5.99.0"}, + "spm-calculator": {"version": "1.0.0"}, + }, + } + calls = [] + + def actual_test(*args, **kwargs): + calls.append(kwargs) + return receipt + + monkeypatch.setattr(enrichment, "run_native_loader_compatibility", actual_test) + monkeypatch.setattr(metadata, "version", lambda name: "0.1.0") + monkeypatch.setattr( + enrichment, "_check_producer_source_identity", lambda code: None + ) + output = tmp_path / "certified" / release.name + original_report = (release / enrichment.SOURCE_ENRICHMENT_FILE).read_bytes() + result = enrichment.certify_source_enrichment( + release, + output, + parent_h5=parent, + artifact_root=root, + compatibility_wheels=(tmp_path / "country.whl",), + ) + assert result == output + assert (release / enrichment.SOURCE_ENRICHMENT_FILE).read_bytes() == original_report + assert (output / enrichment.SOURCE_EVIDENCE_FILE).read_bytes() == ( + release / enrichment.SOURCE_EVIDENCE_FILE + ).read_bytes() + assert ( + publish_main( + [ + str(output), + "--parent-h5", + str(parent), + "--artifact-root", + str(root), + "--compatibility-wheel", + str(tmp_path / "country.whl"), + "--preflight-only", + ] + ) + == 0 + ) + assert len(calls) == 3 # Test, staged revalidation, then real publisher preflight. + assert all(call["require_wheels"] for call in calls) + + +def test_native_compatibility_requires_wheels_for_all_runtime_packages(): + with pytest.raises(ValueError, match="exact installed policyengine-us"): + enrichment._runtime_package_identities((), require_wheels=True) + + +@pytest.mark.parametrize("distribution_name", ["policyengine-us", "spm-calculator"]) +def test_wheel_identity_rejects_changed_installed_source( + tmp_path, monkeypatch, distribution_name +): + from importlib import metadata + from zipfile import ZipFile + + package = tmp_path / "pkg" + package.mkdir() + (package / "module.py").write_text("VALUE = 2\n") + normalized = distribution_name.replace("-", "_") + wheel = tmp_path / f"{normalized}-1.0-py3-none-any.whl" + with ZipFile(wheel, "w") as archive: + archive.writestr( + f"{normalized}-1.0.dist-info/METADATA", + f"Name: {distribution_name}\nVersion: 1.0\n", + ) + archive.writestr("pkg/module.py", "VALUE = 1\n") + + class Installed: + version = "1.0" + files = ["pkg/module.py"] + + def read_text(self, name): + return None + + def locate_file(self, relative): + return tmp_path / relative + + monkeypatch.setattr(metadata, "distribution", lambda name: Installed()) + with pytest.raises(ValueError, match="installed source differs from wheel"): + enrichment._runtime_package_identities((wheel,), require_wheels=False) + + +def test_three_wheels_cannot_omit_calculator(tmp_path, monkeypatch): + monkeypatch.setattr(enrichment, "_wheel_files", lambda path: (path.stem, "1.0", {})) + wheels = tuple( + tmp_path / f"{name}.whl" + for name in ("policyengine-us", "policyengine-core", "policyengine") + ) + with pytest.raises(ValueError, match="spm-calculator wheels"): + enrichment._runtime_package_identities(wheels, require_wheels=True) + + +def test_registered_native_role_is_owned_by_calculator_wheel(tmp_path, monkeypatch): + import importlib + from importlib import metadata + from types import SimpleNamespace + + monkeypatch.setattr( + metadata, + "distribution", + lambda name: SimpleNamespace(locate_file=lambda relative: tmp_path / relative), + ) + imported = [] + + def module(name): + imported.append(name) + return SimpleNamespace(__file__=str(tmp_path / name / "__init__.py")) + + monkeypatch.setattr(importlib, "import_module", module) + paths = { + "country_loader": tmp_path / "policyengine_us/data/dataset_schema.py", + "wrapper_loader": tmp_path / "policyengine/tax_benefit_models/us/datasets.py", + "native_role": tmp_path / "spm_calculator/policyengine_adapter.py", + } + enrichment._check_loaded_source_ownership(paths) + assert set(imported) == { + "policyengine_us", + "policyengine_core", + "policyengine", + "spm_calculator", + } + # A same-named country class or source outside the verified calculator is + # not the production calculator-owned primitive. + for replacement in ( + tmp_path / "policyengine_us/native_role.py", + tmp_path / "unverified_calculator/policyengine_adapter.py", + ): + with pytest.raises(ValueError, match="native_role.*spm-calculator wheel"): + enrichment._check_loaded_source_ownership( + paths | {"native_role": replacement} + ) + + +def test_evidence_tier_cannot_bypass_enrichment_gate(candidate): + from microcosm.data.contract import validate_evidence_release_dir + + release, _, _ = candidate + with pytest.raises(ReleaseContractError, match="evidence tier does not accept"): + validate_evidence_release_dir(release) + + +def test_core_probe_rejects_silently_discarded_native_input(candidate, monkeypatch): + import sys + from types import ModuleType, SimpleNamespace + + import pandas as pd + + _, _, root = candidate + with pd.HDFStore(root / "populace_us_2024.h5", "r") as store: + tables = { + entity: store[entity] + for entity in ( + "person", + "household", + "tax_unit", + "spm_unit", + "family", + "marital_unit", + ) + } + for entity in ("tax_unit", "family", "marital_unit"): + tables["person"][f"person_{entity}_id"] = tables["person"][ + "person_household_id" + ] + calls = [] + + class IgnoresInput: + def __init__(self, dataset): + calls.append(dataset) + self.dataset = dataset + + def calculate(self, variable, year): + return np.zeros(len(self.dataset.person), dtype=bool) + + fake_country = ModuleType("policyengine_us") + fake_country.Microsimulation = IgnoresInput + fake_data = ModuleType("policyengine_us.data") + fake_data.USSingleYearDataset = lambda **kwargs: SimpleNamespace(**kwargs) + monkeypatch.setitem(sys.modules, "policyengine_us", fake_country) + monkeypatch.setitem(sys.modules, "policyengine_us.data", fake_data) + with pytest.raises(ValueError, match="discarded the supplied native person role"): + enrichment._check_native_input_precedence(SimpleNamespace(**tables)) + assert len(calls) == 2 + assert all(len(dataset.person) == 2 for dataset in calls) + assert all( + dataset.person["person_spm_unit_id"].tolist() == [1, 1] for dataset in calls + ) + + +def test_producer_identity_rejects_fabricated_clean_commit(): + code = { + "git_commit": "0" * 40, + "git_dirty": False, + "source_files_sha256": { + name: "a" * 64 for name in enrichment.PRODUCER_SOURCE_FILES + }, + } + with pytest.raises(ValueError, match="cannot be authenticated"): + enrichment._check_producer_source_identity(code) + + +def test_producer_identity_rejects_uncommitted_source_mutation(tmp_path, monkeypatch): + import hashlib + import subprocess + from pathlib import Path + from types import SimpleNamespace + + root = Path(__file__).resolve().parents[3] + commit = "a" * 40 + sources = { + name: (root / name).read_bytes() for name in enrichment.PRODUCER_SOURCE_FILES + } + for name, content in sources.items(): + path = tmp_path / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(content) + changed = tmp_path / enrichment.PRODUCER_SOURCE_FILES[0] + changed.write_bytes(changed.read_bytes() + b"\n# uncommitted source edit\n") + code = { + "git_commit": commit, + "git_dirty": False, + "source_files_sha256": { + name: hashlib.sha256(content).hexdigest() + for name, content in sources.items() + }, + } + + def git_result(args, **kwargs): + if args[1:] == ["rev-parse", "--show-toplevel"]: + return SimpleNamespace(stdout=str(tmp_path).encode()) + if args[1] == "rev-parse": + return SimpleNamespace(stdout=commit.encode()) + return SimpleNamespace(stdout=sources[args[2].split(":", 1)[1]]) + + monkeypatch.setattr(subprocess, "run", git_result) + with pytest.raises(ValueError, match="checkout source differs"): + enrichment._check_producer_source_identity(code) diff --git a/tools/build_us_spm_role_enrichment.py b/tools/build_us_spm_role_enrichment.py new file mode 100644 index 000000000..f7167c64c --- /dev/null +++ b/tools/build_us_spm_role_enrichment.py @@ -0,0 +1,317 @@ +#!/usr/bin/env python3 +"""Create a local BuildP native-role candidate; never calibrate or publish.""" + +from __future__ import annotations + +import argparse +import inspect +import json +import os +import platform +import shutil +import subprocess +import tempfile +from collections.abc import Sequence +from importlib import metadata +from pathlib import Path + +from microcosm.build.us_runtime import education_assistance_source +from microcosm.build.us_runtime.spm_role_source import ( + ASEC_SPM_ROLE_SOURCES, + derive_spm_role_source, +) +from microcosm.data.contract import ReleaseContractError +from microcosm.data.h5_enrichment import append_native_spm_role, file_sha256 +from microcosm.data.source_enrichment import ( + PARENT_BUILD_ID, + PARENT_DATASET_SHA256, + PARENT_FILES, + ROLE_VARIABLE, + SOURCE_ENRICHMENT_FILE, + SOURCE_ENRICHMENT_RELEASE_TYPE, + SOURCE_EVIDENCE_FILE, + SOURCE_PROVENANCE_FILE, + validate_source_enrichment_candidate, +) + +REFERENCE_EVIDENCE_SHA256 = ( + "22b5968d90fecfeef7614583e493fe10cc16bda8b5be82e6f49a5bc2102d3ce5" +) +_PRODUCER_FILES = ( + "tools/build_us_spm_role_enrichment.py", + "packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py", + "packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py", + "packages/microcosm-data/src/microcosm/data/h5_enrichment.py", + "packages/microcosm-data/src/microcosm/data/source_enrichment.py", + "packages/microcosm-data/src/microcosm/data/contract.py", +) + + +def _json_write(path: Path, value: object) -> None: + with path.open("x") as stream: + json.dump(value, stream, indent=2, sort_keys=True, allow_nan=False) + stream.write("\n") + path.chmod(0o600) + + +def _producer_identity() -> dict: + import h5py + + root = Path(__file__).resolve().parents[1] + + loaded = { + _PRODUCER_FILES[0]: __file__, + _PRODUCER_FILES[1]: inspect.getsourcefile(derive_spm_role_source), + _PRODUCER_FILES[2]: education_assistance_source.__file__, + _PRODUCER_FILES[3]: inspect.getsourcefile(append_native_spm_role), + _PRODUCER_FILES[4]: inspect.getsourcefile(validate_source_enrichment_candidate), + _PRODUCER_FILES[5]: inspect.getsourcefile(ReleaseContractError), + } + for filename, actual in loaded.items(): + if actual is None or Path(actual).resolve() != (root / filename).resolve(): + raise ValueError( + f"Producer must execute this checkout's {filename}; " + "install this workspace's local shards first" + ) + + def git(*args): + return subprocess.check_output(["git", *args], cwd=root, text=True).strip() + + return { + "repository": "https://github.com/PolicyEngine/microcosm", + "git_commit": git("rev-parse", "HEAD"), + "git_dirty": bool(git("status", "--porcelain")), + "runtime": { + "python": platform.python_version(), + "hdf5": h5py.version.hdf5_version, + **{ + package: metadata.version(package) + for package in ("numpy", "pandas", "h5py", "tables", "microcosm-data") + }, + }, + "source_files_sha256": { + filename: file_sha256(root / filename) for filename in _PRODUCER_FILES + }, + } + + +def build_candidate( + *, + parent_h5: Path, + parent_release_dir: Path, + source_paths: dict[int, Path], + output_dir: Path, + release_id: str, + reference_evidence_csv: Path | None = None, +) -> dict: + """Derive, preserve, validate, then atomically expose a new private bundle.""" + if ( + not release_id.startswith("populace-us-2024-") + or Path(release_id).name != release_id + or release_id == PARENT_BUILD_ID + or "\\" in release_id + ): + raise ValueError("Choose a new bare US 2024 source-enrichment release ID") + if output_dir.exists() or output_dir.is_symlink(): + raise FileExistsError("Output directory already exists; choose a new candidate") + if file_sha256(parent_h5) != PARENT_DATASET_SHA256: + raise ValueError("Only the exact reviewed BuildP parent can be enriched") + for destination, expected in PARENT_FILES.items(): + source = parent_release_dir / destination.removeprefix("parent_") + if file_sha256(source) != expected: + raise ValueError(f"Parent evidence SHA-256 mismatch: {source.name}") + if reference_evidence_csv is not None and ( + file_sha256(reference_evidence_csv) != REFERENCE_EVIDENCE_SHA256 + ): + raise ValueError("Reference evidence CSV SHA-256 mismatch") + producer_identity = _producer_identity() + reconstructed = derive_spm_role_source( + parent_h5, + source_paths, + expected_parent_sha256=PARENT_DATASET_SHA256, + ) + evidence_bytes = reconstructed.evidence.to_csv( + index=False, lineterminator="\n" + ).encode() + # This is an independent oracle, never the producer's input data source. + import hashlib + + if hashlib.sha256(evidence_bytes).hexdigest() != REFERENCE_EVIDENCE_SHA256: + raise ValueError( + "Reconstructed source evidence differs from reviewed BuildP roles" + ) + if ( + reference_evidence_csv is not None + and reference_evidence_csv.read_bytes() != evidence_bytes + ): + raise ValueError( + "Reconstruction differs from the supplied independent evidence" + ) + output_dir.parent.mkdir(parents=True, exist_ok=True) + staging = Path(tempfile.mkdtemp(prefix=".spm-enrichment-", dir=output_dir.parent)) + try: + release_dir = staging / "releases" / release_id + release_dir.mkdir(parents=True, mode=0o700) + artifact_root = staging / "artifacts" + artifact_root.mkdir(mode=0o700) + candidate_h5 = artifact_root / "populace_us_2024.h5" + preservation = append_native_spm_role( + parent_h5, + candidate_h5, + reconstructed.role, + expected_parent_sha256=PARENT_DATASET_SHA256, + ) + evidence_path = release_dir / SOURCE_EVIDENCE_FILE + evidence_path.write_bytes(evidence_bytes) + evidence_path.chmod(0o400) + _json_write(release_dir / SOURCE_PROVENANCE_FILE, reconstructed.provenance) + (release_dir / SOURCE_PROVENANCE_FILE).chmod(0o400) + for destination in PARENT_FILES: + shutil.copyfile( + parent_release_dir / destination.removeprefix("parent_"), + release_dir / destination, + ) + (release_dir / destination).chmod(0o400) + dataset = { + "filename": candidate_h5.name, + "sha256": preservation["candidate_sha256"], + } + report = { + "schema_version": 1, + "release_type": SOURCE_ENRICHMENT_RELEASE_TYPE, + "operation": "add_native_spm_independent_minor_role", + "parent": { + "build_id": PARENT_BUILD_ID, + "repo_id": "policyengine/populace-us", + "revision": PARENT_BUILD_ID, + "dataset_sha256": PARENT_DATASET_SHA256, + "calibration_diagnostics_schema_version": 5, + "files": PARENT_FILES, + }, + "dataset": dataset, + "added_variable": { + "name": ROLE_VARIABLE, + "entity": "person", + "dtype": "bool", + "age_gate_applied": False, + }, + "preservation": preservation, + "source": { + "person_evidence_filename": SOURCE_EVIDENCE_FILE, + "person_evidence_sha256": file_sha256(evidence_path), + "provenance_filename": SOURCE_PROVENANCE_FILE, + "provenance_sha256": file_sha256(release_dir / SOURCE_PROVENANCE_FILE), + }, + "reconciliation": reconstructed.provenance, + "compatibility": {"status": "pending"}, + } + _json_write(release_dir / SOURCE_ENRICHMENT_FILE, report) + build = { + "build_id": release_id, + "release_type": SOURCE_ENRICHMENT_RELEASE_TYPE, + "code": producer_identity, + "dataset": dataset, + "calibration": { + "mode": "inherited", + "parent_build_id": PARENT_BUILD_ID, + "diagnostics_sha256": PARENT_FILES["calibration_diagnostics.json"], + "diagnostics_schema_version": 5, + }, + "staging": { + "enabled": False, + "reason": "local source-enrichment candidate; no calibration", + }, + } + _json_write(release_dir / "build_manifest.json", build) + + def artifact(path: Path, *, kind: str) -> dict: + return { + "repo_id": "policyengine/populace-us", + "revision": release_id, + "path": path.name, + "sha256": file_sha256(path), + "kind": kind, + } + + manifest = { + "schema_version": 1, + "release_type": SOURCE_ENRICHMENT_RELEASE_TYPE, + "data_package": { + "name": "microcosm-data", + "version": metadata.version("microcosm-data"), + }, + "default_datasets": {"national": "populace_us_2024"}, + "build": {"build_id": release_id}, + "compatible_core_packages": [], + "compatible_model_packages": [], + "artifacts": { + "populace_us_2024": artifact(candidate_h5, kind="microdata"), + **{ + path.name.removesuffix(".json").removesuffix(".csv"): artifact( + path, kind="evidence" + ) + for path in sorted(release_dir.iterdir()) + if path.is_file() + }, + }, + } + _json_write(release_dir / "release_manifest.json", manifest) + validate_source_enrichment_candidate( + release_dir, parent_h5=parent_h5, artifact_root=artifact_root + ) + if ( + _producer_identity()["source_files_sha256"] + != producer_identity["source_files_sha256"] + ): + raise ValueError( + "Producer source files changed while building the candidate" + ) + candidate_h5.chmod(0o400) + if output_dir.exists() or output_dir.is_symlink(): + raise FileExistsError("Output appeared during build; refusing to overwrite") + os.rename(staging, output_dir) + return report + except BaseException: + shutil.rmtree(staging) + raise + + +def main(argv: Sequence[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--parent-h5", type=Path, required=True) + parser.add_argument("--parent-release-dir", type=Path, required=True) + parser.add_argument("--output-dir", type=Path, required=True) + parser.add_argument("--release-id", required=True) + parser.add_argument("--reference-evidence-csv", type=Path) + parser.add_argument( + "--source-cache", + type=Path, + default=Path.home() / ".cache/microcosm/cps/asec_education", + help="Directory containing the pinned complete pppub23/24/25.csv files", + ) + args = parser.parse_args(argv) + try: + report = build_candidate( + parent_h5=args.parent_h5, + parent_release_dir=args.parent_release_dir, + source_paths={ + year: args.source_cache / pin.member + for year, pin in ASEC_SPM_ROLE_SOURCES.items() + }, + output_dir=args.output_dir, + release_id=args.release_id, + reference_evidence_csv=args.reference_evidence_csv, + ) + except (ValueError, OSError) as exc: + parser.exit(2, f"Source-enrichment candidate refused: {exc}\n") + print( + json.dumps( + {"dataset": report["dataset"], "compatibility": report["compatibility"]}, + indent=2, + ) + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From acfe2b17543b90810b663cb47116226dd19f0268 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Wed, 9 Sep 2026 12:22:15 -0400 Subject: [PATCH 02/51] Bind SPM enrichment publication to verified dataset and source identities --- PROGRESS.md | 36 +++++ docs/us-native-spm-role-source-enrichment.md | 7 + .../test_us_spm_role_enrichment_builder.py | 4 +- .../tests/test_us_spm_role_source.py | 29 ++++ .../src/microcosm/data/source_enrichment.py | 56 ++++++-- .../tests/test_source_enrichment.py | 129 ++++++++++++++++-- 6 files changed, 240 insertions(+), 21 deletions(-) diff --git a/PROGRESS.md b/PROGRESS.md index 97313a628..a6a8a4552 100644 --- a/PROGRESS.md +++ b/PROGRESS.md @@ -1,5 +1,41 @@ # Native SPM role source enrichment — 2026-09-09 +## State + +Completed the bounded independent-review fixes for draft PR #894 at +`63c8982459317c072ef748e4e642238939b32ede`. No commits, pushes or publication +are authorized in this pass; the existing native H5 remains immutable. + +## Done + +Rejected release-local duplicate H5 entries before compatibility or Hub activity, +and pinned each Census year's archive SHA256/URL/member/income year. Added 27 +duplicate/resealing regressions and acquisition pin parity; the first 24 cases +failed before enforcement. All 381 relevant tests pass. The actual candidate +passes the tightened contract and exhaustive H5 comparison; SHA256 remains +`6496cc4393d4d3c6574f76eca231de5898c803b9067645591fd5c4d3e65aee84`. +Ruff, CI inventory and whitespace checks pass; separate code review found no +concrete remaining P1/P2 defect. + +The corrected actual wrapper wheel passes the final-source four-wheel probe: +30 native-input checks plus all six entity weight checks. Neither external +publication nor canonical numerical acceptance is attested. Exact commands, +diagnostic identities and changed files are recorded in +`MICRO-CANONICAL-SPM-REVIEW-FIXES.md` and `MICRO-CANONICAL-SPM-HANDOFF.md`. + +## Next + +Root independently re-reviews and commits these changes, completes exact Fable +and PR CI gates, creates fresh receipts from clean source, and supplies external +package/numerical acceptance before any authorized HF publication or promotion. +Existing candidate receipts remain unchanged. Earlier test evidence is retained +below. + +--- + +> Earlier implementation status below predates the independent review fixes and +> corrected wrapper wheel; the State/Done/Next entries above are current. + Source implementation prepared for a draft PR. The candidate was built locally; no new H5, release tag, calibration or production deployment was published. This journal records development evidence; check GitHub for current PR state. diff --git a/docs/us-native-spm-role-source-enrichment.md b/docs/us-native-spm-role-source-enrichment.md index aad09b0f8..5f0ef6c4a 100644 --- a/docs/us-native-spm-role-source-enrichment.md +++ b/docs/us-native-spm-role-source-enrichment.md @@ -21,6 +21,9 @@ separately reviewed contract; there is no caller-supplied legacy-schema waiver. `spm_role_source.py` reuses Microcosm's existing Census ASEC archive/member pins from `education_assistance_source.py`. The complete source person CSVs are `pppub23.csv`, `pppub24.csv`, and `pppub25.csv` (income years 2022–2024). +The release gate independently checks each survey year's exact archive SHA256, +official URL, person member and income year, alongside its existing CSV pin. +Rehashing provenance and enclosing manifests cannot authorize another archive. The source role is: ```python @@ -64,6 +67,10 @@ The generated three-column evidence CSV must also match independently reviewed SHA256 `22b5968d90fecfeef7614583e493fe10cc16bda8b5be82e6f49a5bc2102d3ce5`. The CSV is never used to derive the role. Parent and source evidence are copied unchanged into the release bundle; H5 and evidence files are read-only on exit. +The H5 belongs only in `artifact_root`, for upload at its manifest-declared root +path. A same-named entry in the release directory is rejected before compatibility +probing or Hub client activity, including identical copies and symlinks: either +would otherwise change the publisher's upload destination. ## Build and validate a local candidate diff --git a/packages/microcosm-build/tests/test_us_spm_role_enrichment_builder.py b/packages/microcosm-build/tests/test_us_spm_role_enrichment_builder.py index 620dce135..2ec1cb581 100644 --- a/packages/microcosm-build/tests/test_us_spm_role_enrichment_builder.py +++ b/packages/microcosm-build/tests/test_us_spm_role_enrichment_builder.py @@ -93,14 +93,12 @@ def inputs(tmp_path, monkeypatch): "evidence_column": contract.EVIDENCE_COLUMN, "source_checks": [ { - "income_year": 2024, "survey_year": 2025, "csv_sha256": "a" * 64, "persons": 3, "units": 2, "adult_child_person_count_mismatch_units": 0, - "official_archive_url": "https://www2.census.gov/fixture.zip", - "archive_sha256": "b" * 64, + **contract.CENSUS_ARCHIVE_PINS[2025], } ], } diff --git a/packages/microcosm-build/tests/test_us_spm_role_source.py b/packages/microcosm-build/tests/test_us_spm_role_source.py index 76e9e644f..22fec7077 100644 --- a/packages/microcosm-build/tests/test_us_spm_role_source.py +++ b/packages/microcosm-build/tests/test_us_spm_role_source.py @@ -8,6 +8,7 @@ import pytest from microcosm.build.us_runtime.spm_role_source import ( + ASEC_SPM_ROLE_SOURCES, EVIDENCE_SPM_ROLE, AsecSpmRoleSource, derive_spm_role_source, @@ -17,6 +18,34 @@ pytest.importorskip("tables") +def test_release_contract_pins_match_actual_source_acquisition(): + from microcosm.build.us_runtime.education_assistance_source import ( + ASEC_EDUCATION_ASSISTANCE_ARCHIVES, + ) + from microcosm.data.source_enrichment import CENSUS_ARCHIVE_PINS, CENSUS_PERSON_PINS + + assert ( + set(CENSUS_ARCHIVE_PINS) + == set(CENSUS_PERSON_PINS) + == {pin.survey_year for pin in ASEC_EDUCATION_ASSISTANCE_ARCHIVES.values()} + ) + for income_year, pin in ASEC_EDUCATION_ASSISTANCE_ARCHIVES.items(): + assert CENSUS_PERSON_PINS[pin.survey_year] == pin.member_sha256 + assert CENSUS_ARCHIVE_PINS[pin.survey_year] == { + "income_year": income_year, + "official_archive_url": pin.zip_url, + "archive_sha256": pin.zip_sha256, + "member": pin.member, + } + role = ASEC_SPM_ROLE_SOURCES[income_year] + assert role.survey_year == pin.survey_year + assert role.csv_sha256 == pin.member_sha256 + assert role.income_year == income_year + assert role.official_archive_url == pin.zip_url + assert role.archive_sha256 == pin.zip_sha256 + assert role.member == pin.member + + def _digest(path): return hashlib.sha256(path.read_bytes()).hexdigest() diff --git a/packages/microcosm-data/src/microcosm/data/source_enrichment.py b/packages/microcosm-data/src/microcosm/data/source_enrichment.py index d327f42fd..5495d0aeb 100644 --- a/packages/microcosm-data/src/microcosm/data/source_enrichment.py +++ b/packages/microcosm-data/src/microcosm/data/source_enrichment.py @@ -63,6 +63,37 @@ 2024: "21a2b9e0e4b08534563578a45acad77868af4ae9a7d46f23776b707d4a559aa7", 2025: "06921fe83fc66c907e6c7b86b82255dc70458ee7d76258fc48297cb34f0c06b5", } +# Match the producer's education_assistance_source archive pins without making +# the data/consumer shard depend on the build shard. A build test checks parity. +CENSUS_ARCHIVE_PINS = { + 2023: { + "income_year": 2022, + "official_archive_url": ( + "https://www2.census.gov/programs-surveys/cps/datasets/2023/" + "march/asecpub23csv.zip" + ), + "archive_sha256": "d2e000250782adfbdd7f29c82b66d866591a30f0d330496698ec19f9c784ce11", + "member": "pppub23.csv", + }, + 2024: { + "income_year": 2023, + "official_archive_url": ( + "https://www2.census.gov/programs-surveys/cps/datasets/2024/" + "march/asecpub24csv.zip" + ), + "archive_sha256": "cdb39cdac34bef99dd0940ab28e306f692404c2eea44d85dfd634214872a0a09", + "member": "pppub24.csv", + }, + 2025: { + "income_year": 2024, + "official_archive_url": ( + "https://www2.census.gov/programs-surveys/cps/datasets/2025/" + "march/asecpub25csv.zip" + ), + "archive_sha256": "318845a2b5e0034eb2973898de1738f4df0025727de38499e7669cb9c0deef0b", + "member": "pppub25.csv", + }, +} EXPECTED_COUNTS = { "persons_joined": 166321, "native_spm_units": 59900, @@ -183,15 +214,11 @@ def _check_source_provenance(provenance: Mapping, failures: list[str]) -> None: failures.append( f"source provenance Census {year} count reconciliation failed" ) - if not str(row.get("official_archive_url", "")).startswith( - "https://www2.census.gov/" - ): - failures.append( - f"source provenance Census {year} requires official archive" - ) - archive_sha = row.get("archive_sha256") - if not isinstance(archive_sha, str) or not _SHA256_RE.fullmatch(archive_sha): - failures.append(f"source provenance Census {year} requires archive SHA256") + for field, expected in CENSUS_ARCHIVE_PINS[year].items(): + if row.get(field) != expected: + failures.append( + f"source provenance Census {year} pinned archive {field} differs" + ) def validate_source_enrichment_candidate( @@ -286,6 +313,17 @@ def validate_source_enrichment_candidate( ): failures.append("source enrichment dataset.filename must be a bare H5 filename") filename = None + if filename: + release_local_h5 = release_dir / filename + if release_local_h5.exists() or release_local_h5.is_symlink(): + # The publisher treats release-local files as release-dir uploads + # and suppresses their root uploads, even when the bytes match. + # Reject ambiguity before compatibility imports or Hub activity. + failures.append( + f"source enrichment rejects release-local H5 {filename}; " + "the native dataset must exist only in artifact_root" + ) + raise ReleaseContractError(release_dir, failures) candidate = Path(artifact_root) / filename if artifact_root and filename else None if parent_h5 is None or candidate is None: failures.append( diff --git a/packages/microcosm-data/tests/test_source_enrichment.py b/packages/microcosm-data/tests/test_source_enrichment.py index 4757e82d8..aba22eace 100644 --- a/packages/microcosm-data/tests/test_source_enrichment.py +++ b/packages/microcosm-data/tests/test_source_enrichment.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import shutil import h5py import numpy as np @@ -75,7 +76,6 @@ def candidate(tmp_path, monkeypatch): "classification_changed_units_vs_age_only": 1, } monkeypatch.setattr(enrichment, "EXPECTED_COUNTS", counts) - monkeypatch.setattr(enrichment, "CENSUS_PERSON_PINS", {2025: "a" * 64}) evidence = release / enrichment.SOURCE_EVIDENCE_FILE evidence.write_text( "person_id,person_spm_unit_id,is_spm_independence_role\n1,1,False\n2,1,True\n3,2,True\n" @@ -95,12 +95,12 @@ def candidate(tmp_path, monkeypatch): "evidence_column": enrichment.EVIDENCE_COLUMN, "source_checks": [ { - "survey_year": 2025, - "csv_sha256": "a" * 64, + "survey_year": year, + "csv_sha256": csv_sha, "adult_child_person_count_mismatch_units": 0, - "official_archive_url": "https://www2.census.gov/test.zip", - "archive_sha256": "b" * 64, + **enrichment.CENSUS_ARCHIVE_PINS[year], } + for year, csv_sha in enrichment.CENSUS_PERSON_PINS.items() ], } _write(release / enrichment.SOURCE_PROVENANCE_FILE, provenance) @@ -343,9 +343,7 @@ def test_unknown_release_type_cannot_fall_back_to_calibration(candidate): validate_release_dir(release) -def test_certification_writes_new_bundle_and_preflight_replays( - candidate, tmp_path, monkeypatch -): +def _qualify_candidate(candidate, tmp_path, monkeypatch): from importlib import metadata release, parent, root = candidate @@ -371,7 +369,6 @@ def actual_test(*args, **kwargs): enrichment, "_check_producer_source_identity", lambda code: None ) output = tmp_path / "certified" / release.name - original_report = (release / enrichment.SOURCE_ENRICHMENT_FILE).read_bytes() result = enrichment.certify_source_enrichment( release, output, @@ -380,6 +377,15 @@ def actual_test(*args, **kwargs): compatibility_wheels=(tmp_path / "country.whl",), ) assert result == output + return output, calls + + +def test_certification_writes_new_bundle_and_preflight_replays( + candidate, tmp_path, monkeypatch +): + release, parent, root = candidate + original_report = (release / enrichment.SOURCE_ENRICHMENT_FILE).read_bytes() + output, calls = _qualify_candidate(candidate, tmp_path, monkeypatch) assert (release / enrichment.SOURCE_ENRICHMENT_FILE).read_bytes() == original_report assert (output / enrichment.SOURCE_EVIDENCE_FILE).read_bytes() == ( release / enrichment.SOURCE_EVIDENCE_FILE @@ -403,6 +409,111 @@ def actual_test(*args, **kwargs): assert all(call["require_wheels"] for call in calls) +@pytest.mark.parametrize("duplicate", ["identical", "different", "symlink"]) +@pytest.mark.parametrize( + "entrypoint", ["candidate", "preflight", "publisher", "publisher_existing_client"] +) +def test_release_local_h5_duplicate_is_rejected_before_hub_activity( + candidate, tmp_path, monkeypatch, duplicate, entrypoint +): + import microcosm.data.release as release_module + + _, parent, root = candidate + release, _ = _qualify_candidate(candidate, tmp_path, monkeypatch) + # The unduplicated fixture passes the full gate: pending compatibility or + # synthetic source identity must not accidentally account for rejection. + validate_release_dir(release, parent_h5=parent, artifact_root=root) + h5 = root / "populace_us_2024.h5" + local = release / h5.name + if duplicate == "symlink": + local.symlink_to(h5) + else: + shutil.copyfile(h5, local) + if duplicate == "different": + with h5py.File(local, "r+") as handle: + row = handle["person/table"][0] + row["person_weight"] += 1 + handle["person/table"][0] = row + monkeypatch.setattr( + release_module, + "_hf_api", + lambda: pytest.fail("duplicate reached Hub client construction"), + ) + + class NoHubActivity: + def __getattr__(self, name): + pytest.fail(f"duplicate accessed supplied Hub client: {name}") + + monkeypatch.setattr( + enrichment, + "run_native_loader_compatibility", + lambda *a, **kw: pytest.fail("duplicate reached compatibility probing"), + ) + with pytest.raises(ReleaseContractError, match="release-local H5"): + if entrypoint == "candidate": + _validate((release, parent, root)) + elif entrypoint == "preflight": + publish_main( + [ + str(release), + "--parent-h5", + str(parent), + "--artifact-root", + str(root), + "--preflight-only", + ] + ) + else: + publish_release( + release, + "policyengine/populace-us", + parent_h5=parent, + artifact_root=root, + notify=False, + api=NoHubActivity() + if entrypoint == "publisher_existing_client" + else None, + ) + + +@pytest.mark.parametrize("year", [2023, 2024, 2025]) +@pytest.mark.parametrize( + "field", ["archive_sha256", "official_archive_url", "member", "income_year"] +) +def test_resealed_provenance_rejects_archive_identity_from_another_year( + candidate, year, field +): + other_year = 2023 if year != 2023 else 2024 + replacement = enrichment.CENSUS_ARCHIVE_PINS[other_year][field] + _assert_resealed_archive_rejected(candidate, year, field, replacement) + + +@pytest.mark.parametrize("year", [2023, 2024, 2025]) +def test_resealed_provenance_rejects_unpinned_archive_sha(candidate, year): + _assert_resealed_archive_rejected(candidate, year, "archive_sha256", "0" * 64) + + +def _assert_resealed_archive_rejected(candidate, year, field, replacement): + assert _validate(candidate)["compatibility"]["status"] == "pending" + release, _, _ = candidate + provenance_path = release / enrichment.SOURCE_PROVENANCE_FILE + provenance = json.loads(provenance_path.read_text()) + row = next(row for row in provenance["source_checks"] if row["survey_year"] == year) + row[field] = replacement + _write(provenance_path, provenance) + report_path = release / enrichment.SOURCE_ENRICHMENT_FILE + report = json.loads(report_path.read_text()) + report["source"]["provenance_sha256"] = enrichment.sha256_file(provenance_path) + report["reconciliation"] = provenance + _write(report_path, report) + _refresh(release, provenance_path.name) + _refresh(release, report_path.name) + with pytest.raises( + ReleaseContractError, match=f"Census {year} pinned archive {field} differs" + ): + _validate(candidate) + + def test_native_compatibility_requires_wheels_for_all_runtime_packages(): with pytest.raises(ValueError, match="exact installed policyengine-us"): enrichment._runtime_package_identities((), require_wheels=True) From 18ca76eab97761250797dd5f3e40a9f69e28475e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:11:30 +0200 Subject: [PATCH 03/51] Add exact-household-count UK dataset candidates (#355) --- CLAUDE.md | 6 + changelog.d/uk-dataset-sizes-355.added.md | 1 + docs/uk-dataset-size-plan-355.md | 105 +++++++++ .../build/uk_runtime/dataset_size.py | 198 +++++++++++++++++ .../build/uk_runtime/local_rowwise.py | 62 +++++- .../tests/test_uk_local_rowwise.py | 199 +++++++++++++++++- .../tests/test_uk_rowwise_candidate.py | 80 +++++++ .../src/microcosm/calibrate/gates.py | 39 +++- .../src/microcosm/calibrate/initialization.py | 64 ++++++ .../src/microcosm/calibrate/solve.py | 41 +++- .../tests/test_informed_gates.py | 42 ++++ tools/build_uk_rowwise_candidate.py | 56 ++++- 12 files changed, 879 insertions(+), 14 deletions(-) create mode 100644 changelog.d/uk-dataset-sizes-355.added.md create mode 100644 docs/uk-dataset-size-plan-355.md create mode 100644 packages/microcosm-build/src/microcosm/build/uk_runtime/dataset_size.py create mode 100644 packages/microcosm-calibrate/src/microcosm/calibrate/initialization.py create mode 100644 packages/microcosm-calibrate/tests/test_informed_gates.py diff --git a/CLAUDE.md b/CLAUDE.md index ce1b71c3e..73a10fb72 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -106,3 +106,9 @@ with the journal pointing to them. Update this guide in the same PR whenever the workspace layout, test commands, or release flow change. If you find it contradicting the repo, trust the repo and fix this file. + +UK size experiments use `tools/build_uk_rowwise_candidate.py --dataset-households` +with the same pool inputs as the dense candidate. The flag changes exported +support, not clone K. Sizes remain candidate-only until their matched comparison +and promotion scorecard are adjudicated; see +[the size plan](docs/uk-dataset-size-plan-355.md). diff --git a/changelog.d/uk-dataset-sizes-355.added.md b/changelog.d/uk-dataset-sizes-355.added.md new file mode 100644 index 000000000..4e9bea82a --- /dev/null +++ b/changelog.d/uk-dataset-sizes-355.added.md @@ -0,0 +1 @@ +Add exact-household-count UK rowwise candidates using contribution-informed L0 initialization, protected target carriers, fixed-size sampling and refitting. Preserve the dense pool doctrine and local gates, export compact linked entities, and record selection provenance without promoting candidates to the certified dense release or changing dataset defaults. diff --git a/docs/uk-dataset-size-plan-355.md b/docs/uk-dataset-size-plan-355.md new file mode 100644 index 000000000..a098ee0ae --- /dev/null +++ b/docs/uk-dataset-size-plan-355.md @@ -0,0 +1,105 @@ +# UK dataset sizes: implementation plan and operating boundary + +This implements the candidate-building part of [#355](https://github.com/PolicyEngine/microcosm/issues/355) +on [#870](https://github.com/PolicyEngine/microcosm/pull/870)'s branch. The PR is still open as of +2026-09-05 and itself stacks on #852. Do not base the work on the old issue's +535,080-household 2023 dataset. The authoritative inputs are the current raw-FRS +2024-25 spine, OA ladder, and pinned Chronicle facts used by the joint candidate. + +## Decisions retained + +- Pool generation keeps clone count **K=15**. Requested output households is a + different parameter, applied after cloning and materialization. +- The joint surface retains national, constituency and local-authority rows, + the declared stretch bound **10**, loss cap **10**, **grain_equal** weighting, + and **1,500 epochs** per solve in the normal driver defaults. +- Existing binding adjudications, signed deferrals, measure exclusions, + census-vintage uprating, and all local gates remain in force. Size selection + does not silently drop target rows, loosen ESS floors or change registers. +- Population-normalized engine measures are frozen from the full pool. The + refit uses their selected household contributions rather than re-running + those formulas on a smaller population. +- The Frame carries complete households, benefit units and people; every + non-dry attempt retains the existing Logbook recording envelope. + +## Implemented sequence + +1. Build the usual joint dense solve from the pinned inputs. This also supplies + the same-target reference loss for the size comparison. +2. Compute each household's maximum absolute target-contribution share from + original pool weights and the compiled sparse matrix, following + [#346's correction](https://github.com/PolicyEngine/microcosm/issues/346#issuecomment-4902880142). + Protect the largest absolute weighted carrier of each nonzero target, with + first-column tie breaking. Initialize open probabilities with + `0.1 + 0.8 * score / (score + median_positive_score)`. This bounded smooth + prior is an implementation choice for candidate evaluation, not a measured + UK release ruling. Run L0 budget search; never select the largest prior scores. +3. Use the existing exact-count Sampford sampler on learned probabilities. + Probability-one gates are certainties (`pi_hi=1`), including every protected + carrier. Refuse impossible budgets or sampling designs rather than clamp. +4. Refit on the selected support through `microcosm.calibrate`, with no L0 + penalty. Reuse the existing normalized Horvitz–Thompson `w/q` baseline. + The stretch multiplier remains 10 **relative to that inclusion-adjusted + baseline**; this is the shared exact-k refit's contract, not a claim that + sparse weights remain within 10 times the unexpanded pool-row weights. + The manifest names the reference explicitly. Its empirical suitability for + UK release remains to be assessed alongside the size scorecard. +5. Restore the prepared carrier and export only selected linked entities. + Re-run the existing local gate battery on the compact frame. Each holdout + fold independently reruns selection on training targets; held targets do + not inform the prior or protected set. +6. Record requested/realized counts, pool positions, inclusion probabilities, + protected count, seed, learned penalty, dense/refit loss and target change, + the gate reports, and ordinary byte-pinned output metadata. + +## Running a candidate + +Use the inputs and environment from the existing +[UK dense assembly runbook](uk-dense-release-assembly-runbook-762.md). +Pass the same pinned source arguments to the existing driver and add: + +```bash +uv run python tools/build_uk_rowwise_candidate.py \ + --input-h5 "$UK_SPINE_H5" --input-sha256 "$UK_SPINE_SHA256" \ + --ladder "$UK_LADDER_NPZ" --ladder-sha256 "$UK_LADDER_SHA256" \ + --ledger-facts "$UK_LEDGER_FACTS" \ + --ledger-facts-sha256 "$UK_LEDGER_FACTS_SHA256" \ + --ledger-manifest-sha256 "$UK_LEDGER_MANIFEST_SHA256" \ + --dataset-households 50000 --seed 42 --out out/uk-k50000 +``` + +Repeat with another positive household count and a fresh output directory to +compare sizes. These are requested counts, not certified presets. Omit +`--dataset-households` to retain the existing dense path. Add `--dry-run` to +inspect input binding and parameters without solving or writing a candidate. +Never reduce `--n-clones` to request a smaller output. Full builds still need +the dense build's peak memory and add L0/refit work; the reduction is in the +exported dataset's storage and downstream loading/simulation footprint. + +## Certification and publication still required + +The implementation produces **candidates**, not a new certified UK default. +A size request refuses `--release-candidate`, and its manifest records +`releasable=false` even if the diagnostic gate run passes. The dense assembler +must not interpret a compact candidate as the already reviewed dense line. +The existing registry, production pointers and pe.py default are unchanged. + +Before any size can be promoted: + +1. Run the licensed full-input build, retain all gate failures and measure + local ESS and fit. A nominal 50k size is not guaranteed to clear the floors + that motivated K=15. +2. Run #355's matched sound-comparison protocol and the referenced promotion + scorecard, including reform/distributional validation and untargeted bases. + Calibration loss alone is not a certificate. The included same-target loss + comparison is a diagnostic, not a substitute for those protocols. +3. Adjudicate any new size-specific acceptance decisions, including the + inclusion-adjusted stretch reference. A national-only product would need + its own explicit scope; this implementation does not downgrade local claims. +4. Add size-specific certified release identities, assembly contracts and + downstream bundle entries against that evidence; publication remains the + repository's deliberate human step. + +Accordingly, this increment does not close #355's default-flip requirement. +Synthetic CI checks prove code behavior and artifact structure; they do not +establish licensed-data fit, storage measurements, or release eligibility. diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/dataset_size.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/dataset_size.py new file mode 100644 index 000000000..1a61ff4f8 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/dataset_size.py @@ -0,0 +1,198 @@ +"""Exact household-count UK candidates on a fixed, materialized target surface.""" + +from dataclasses import dataclass, replace + +import numpy as np +import pandas as pd + +from microcosm.calibrate import ( + CalibrationResult, + TargetSet, + assert_exact_k_support, + calibrate, + refit_l0_selection, + select_exact_k, +) +from microcosm.calibrate.initialization import contribution_initialization +from microcosm.frame import Frame + + +@dataclass(frozen=True) +class UKDatasetSize: + """A compact refit with its full-pool row identities and selection evidence.""" + + result: CalibrationResult + support: np.ndarray + receipt: dict[str, object] + + +def refit_uk_dataset_size( + frame: Frame, + dense: CalibrationResult, + *, + households: int, + epochs: int, + learning_rate: float, + seed: int, +) -> UKDatasetSize: + """Run informed L0, a fixed-size draw, and refit under the dense doctrine. + + Freeze the already compiled household contributions, including engine + measures that depend on the full population. Re-evaluating those formulas + on a subset would change the target system. Target values, order, loss + weights, cap and stretch bound remain those of the dense solve. + """ + n = frame.n("household") + if ( + isinstance(households, bool) + or not isinstance(households, int) + or not 0 < households <= n + ): + raise ValueError(f"households must be an integer in [1, {n}]; never clamped.") + if ( + dense.skipped + or len(dense.weights) != n + or dense.l0_lambda != 0 + or dense.weight_entity != "household" + or not np.array_equal( + frame.table("household")["household_id"].to_numpy(), + dense.frame.table("household")["household_id"].to_numpy(), + ) + or not np.array_equal( + frame.weights_for("household").values, dense.initial_weights + ) + ): + raise ValueError( + "size selection requires an aligned, fully compiled dense solve." + ) + if households == n: + return UKDatasetSize( + dense, + np.arange(n), + { + "method": "full_pool", + "requested_households": n, + "realized_households": n, + "pool_households": n, + "seed": seed, + }, + ) + problem = dense.problem + # Input weights are the pool design, not the concentrated dense weights. + init = contribution_initialization( + problem.matrix, dense.initial_weights, problem.target_vector + ) + if int(init.protected.sum()) > households: + raise ValueError( + f"requested {households} households cannot retain {int(init.protected.sum())} protected target carriers." + ) + common = dict( + weight_entity="household", + epochs=epochs, + learning_rate=learning_rate, + mass=dense.options["mass"], + max_weight_ratio=dense.options["max_weight_ratio"], + seed=seed, + target_loss_weights=dense.target_loss_weights, + target_loss_scales=dense.target_loss_scales, + target_loss_cap=dense.target_loss_cap, + ) + selection = calibrate( + frame, + TargetSet(problem.targets), + target_records=households, + gate_initialization=init, + mass_reason=dense.options["mass_reason"], + **common, + ) + probabilities = selection.gate_open_probabilities + if probabilities is None: + raise RuntimeError("informed L0 returned no selection probabilities.") + # Exact-one certainties are protected by the gates themselves; no top-k + # ranking or post-hoc promotion of learned boundary scores is performed. + support, sampling, q = select_exact_k( + probabilities, households, pi_hi=1.0, seed=seed + ) + support = assert_exact_k_support(support, households, pool_size=n) + if not np.isin(np.flatnonzero(init.protected), support).all(): + raise RuntimeError("exact-count selection lost a protected carrier.") + frozen = _frozen_targets(frame, dense, support) + refit = refit_l0_selection( + frame, + frozen, + selection, + support=support, + k=households, + support_inclusion_probabilities=q, + mass_reason=dense.options["mass_reason"], + **common, + ).refit + if ( + refit.skipped + or refit.frame.n("household") != households + or (refit.weights <= 0).any() + ): + raise RuntimeError("compact refit lost targets or positive household support.") + if not np.array_equal(refit.problem.target_vector, problem.target_vector): + raise RuntimeError("compact refit changed target values.") + errors_dense = np.asarray([d.final_estimate for d in dense.diagnostics]) + errors_small = np.asarray([d.final_estimate for d in refit.diagnostics]) + return UKDatasetSize( + refit, + support, + { + "method": "contribution_informed_l0_exact_count_refit", + "requested_households": households, + "realized_households": households, + "pool_households": n, + "seed": seed, + "protected_carriers": int(init.protected.sum()), + "selection_receipt": sampling, + "selection_l0_lambda": selection.l0_lambda, + "selection_epochs": epochs, + "refit_epochs": epochs, + "pool_row_indices": support.tolist(), + "inclusion_probabilities": q.tolist(), + "refit_baseline": "normalized_horvitz_thompson_w_over_q", + "stretch_reference": "normalized_horvitz_thompson_w_over_q", + "dense_loss": dense.final_loss, + "compact_loss": refit.final_loss, + "max_target_scaled_change": float( + np.max(np.abs(errors_small - errors_dense) / dense.target_loss_scales) + ), + "certification": "candidate_only_pending_matched_comparison_and_promotion_scorecard", + }, + ) + + +def _frozen_targets( + frame: Frame, dense: CalibrationResult, support: np.ndarray +) -> TargetSet: + """Bind sparse contribution rows to selected ids, with one shared id join.""" + ids = pd.Index(frame.table("household")["household_id"].iloc[support]) + matrix = dense.problem.matrix[:, support].tocsr() + cached_ids = None + cached_positions = None + + def positions(subset): + nonlocal cached_ids, cached_positions + current = subset.table("household")["household_id"] + if cached_ids is None or not current.equals(cached_ids): + cached_positions = ids.get_indexer(current) + if (cached_positions < 0).any(): + raise ValueError("frozen target requested unknown household ids.") + cached_ids = current.copy() + return cached_positions + + def measure(row): + def values(subset): + return np.asarray(matrix[[row], :].toarray()).reshape(-1)[positions(subset)] + + return values + + return TargetSet( + [ + replace(target, entity="household", measure=measure(row), filter=None) + for row, target in enumerate(dense.problem.targets) + ] + ) diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/local_rowwise.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/local_rowwise.py index a38650ef8..3a1c4cc01 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/local_rowwise.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/local_rowwise.py @@ -151,6 +151,8 @@ class UKRowwiseDoctrineSolve: national_past_cap_census: Mapping[str, Any] all_past_cap_census: Mapping[str, Any] binding_adjudications: Mapping[str, Any] + selected_support: np.ndarray | None = None + size_receipt: Mapping[str, Any] | None = None def past_cap_census( @@ -1008,6 +1010,7 @@ def solve_uk_rowwise_weights_under_doctrine( learning_rate: float = 0.15, conserve_mass: bool = False, target_records: int | None = None, + dataset_households: int | None = None, l0_lambda: float = 0.0, budget_iters: int = 10, seed: int = 0, @@ -1142,6 +1145,26 @@ def solve_uk_rowwise_weights_under_doctrine( target_loss_weights=target_loss_weights, target_loss_cap=doctrine.target_loss_cap, ) + selected_support = None + size_receipt = None + if dataset_households is not None: + if target_records is not None or l0_lambda != 0: + raise ValueError( + "dataset_households requires an unpruned dense reference solve." + ) + from microcosm.build.uk_runtime.dataset_size import refit_uk_dataset_size + + sized = refit_uk_dataset_size( + frame, + result, + households=dataset_households, + epochs=epochs, + learning_rate=learning_rate, + seed=seed, + ) + result = sized.result + selected_support = sized.support + size_receipt = sized.receipt if result.skipped: reasons = [ f"{skipped.target.name}: {skipped.reason}" for skipped in result.skipped[:5] @@ -1280,9 +1303,32 @@ def solve_uk_rowwise_weights_under_doctrine( # ship if the kernel product held strata or metadata the rebuilt frame # lost (both trivially equal today; the guard is the boundary marker # for the day they are not). - clean_result = result.frame if restore is None else restore(result.frame) + if selected_support is None: + clean_result = result.frame if restore is None else restore(result.frame) + expected_ids = problem.household_ids + else: + # Restore the full prepared carrier before selecting: legacy restorers + # retain full original tables and cannot accept compact weight arrays. + clean_pool = frame if restore is None else restore(frame) + expected_ids = tuple(problem.household_ids[i] for i in selected_support) + clean_subset = clean_pool.select( + clean_pool.table("person")["person_household_id"] + .isin(expected_ids) + .to_numpy() + ) + clean_result = Frame( + {entity: clean_subset.table(entity) for entity in clean_subset.entities}, + clean_subset.schema, + { + entity: result.frame.weights_for(entity) + for entity in result.frame.weighted_entities + }, + clean_subset.strata, + mass_log=result.frame.mass_log, + metadata=result.frame.metadata, + ) restored_ids = tuple(clean_result.table("household")["household_id"].tolist()) - if restored_ids != problem.household_ids: + if restored_ids != expected_ids: raise ValueError( "restore returned households that do not match the problem's rows " "(same ids, same order); weights are written back by position, so a " @@ -1344,6 +1390,8 @@ def solve_uk_rowwise_weights_under_doctrine( national_past_cap_census=national_census, all_past_cap_census=all_census, binding_adjudications=binding_adjudications, + selected_support=selected_support, + size_receipt=size_receipt, ) @@ -1380,6 +1428,7 @@ def rotated_uk_local_holdout( learning_rate: float = 0.15, conserve_mass: bool = False, target_records: int | None = None, + dataset_households: int | None = None, l0_lambda: float = 0.0, budget_iters: int = 10, solve_seed: int = 0, @@ -1424,13 +1473,20 @@ def rotated_uk_local_holdout( learning_rate=learning_rate, conserve_mass=conserve_mass, target_records=target_records, + dataset_households=dataset_households, l0_lambda=l0_lambda, budget_iters=budget_iters, seed=solve_seed, ) held_targets = problem.targets[holdout_indices] held_estimates = np.asarray( - problem.matrix[holdout_indices] @ train_solve.weights, + problem.matrix[holdout_indices][ + :, + np.arange(problem.n_households) + if train_solve.selected_support is None + else train_solve.selected_support, + ] + @ train_solve.weights, dtype=np.float64, ).reshape(-1) held_weights = uk_local_target_loss_weights( diff --git a/packages/microcosm-build/tests/test_uk_local_rowwise.py b/packages/microcosm-build/tests/test_uk_local_rowwise.py index faecd4975..bf7a838bc 100644 --- a/packages/microcosm-build/tests/test_uk_local_rowwise.py +++ b/packages/microcosm-build/tests/test_uk_local_rowwise.py @@ -651,7 +651,7 @@ def test_rotated_holdout_keeps_national_rows_in_every_training_fold( def fake_solve(frame, training_problem, **kwargs): calls.append(kwargs["national_rows"]) - return type("Solve", (), {"weights": np.ones(3)})() + return type("Solve", (), {"weights": np.ones(3), "selected_support": None})() monkeypatch.setattr( local_rowwise, "solve_uk_rowwise_weights_under_doctrine", fake_solve @@ -1220,3 +1220,200 @@ def reordering_restore(frame): restore=reordering_restore, epochs=2, ) + + +def test_exact_size_preserves_targets_and_household_links(): + from microcosm.build.uk_runtime.dataset_size import refit_uk_dataset_size + from microcosm.calibrate import Target, TargetSet, calibrate + + frame = _clone_frame() + dense = calibrate( + frame, + TargetSet( + [Target("count", "household", lambda f: np.ones(f.n("household")), 3)] + ), + epochs=2, + ) + sized = refit_uk_dataset_size( + frame, dense, households=2, epochs=2, learning_rate=0.02, seed=7 + ) + assert sized.result.frame.n("household") == 2 + assert sized.result.frame.n("person") == 2 + assert sized.result.frame.n("benunit") == 2 + assert sized.result.problem.names == dense.problem.names + assert sized.result.initial_weights.sum() == pytest.approx( + dense.initial_weights.sum() + ) + assert (sized.result.weights <= 10 * sized.result.initial_weights).all() + again = refit_uk_dataset_size( + frame, dense, households=2, epochs=2, learning_rate=0.02, seed=7 + ) + np.testing.assert_array_equal(sized.support, again.support) + + +@pytest.mark.parametrize("size", [0, -1, True, 4, 1.5]) +def test_exact_size_refuses_invalid_sizes(size): + from microcosm.build.uk_runtime.dataset_size import refit_uk_dataset_size + from microcosm.calibrate import Target, TargetSet, calibrate + + frame = _clone_frame() + dense = calibrate( + frame, + TargetSet( + [Target("count", "household", lambda f: np.ones(f.n("household")), 3)] + ), + epochs=1, + ) + with pytest.raises(ValueError, match="integer"): + refit_uk_dataset_size( + frame, dense, households=size, epochs=1, learning_rate=0.02, seed=7 + ) + + +def test_size_solve_restores_full_prepared_tables_before_subsetting(): + frame = _clone_frame() + metrics = pd.DataFrame({"households": [1.0, 1.0, 1.0]}, index=[101, 102, 103]) + problem = build_uk_rowwise_local_matrix( + metrics, + _assigned(), + pd.DataFrame({"code": ["E001", "S001"], "households": [2.0, 1.0]}), + ) + restored_counts = [] + + def restore(full): + restored_counts.append(full.n("household")) + return full + + result = solve_uk_rowwise_weights_under_doctrine( + frame, + problem, + bound_families=["census_households/constituency"], + dataset_households=2, + epochs=2, + seed=7, + restore=restore, + ) + assert restored_counts == [3] + assert result.frame.n("household") == 2 + assert result.frame.n("person") == 2 + assert len(result.diagnostics) == 2 + assert "census_households/constituency" in result.frame.mass_log[-1].reason + assert result.selected_support.tolist() == [0, 2] + + +def test_size_holdout_uses_compact_support_and_reselects_each_fold(monkeypatch): + import microcosm.build.uk_runtime.local_rowwise as runtime + + metrics = pd.DataFrame({"households": [1.0, 2.0, 3.0]}, index=[101, 102, 103]) + calls = [] + + def solve(frame, training, **kwargs): + calls.append((len(training.targets), kwargs["dataset_households"])) + return type( + "Solve", + (), + {"weights": np.array([1.0, 1.0]), "selected_support": np.array([0, 2])}, + )() + + monkeypatch.setattr(runtime, "solve_uk_rowwise_weights_under_doctrine", solve) + # The holdout helper needs at least one target in each of five folds. + richer = build_uk_rowwise_local_matrix( + pd.DataFrame( + { + name: [1.0, 2.0, 3.0] + for name in ("households", "tenure/social_rent", "tenure/private_rent") + }, + index=metrics.index, + ), + _assigned(), + pd.DataFrame( + { + "code": ["E001", "S001"], + **{ + name: [3.0, 3.0] + for name in ( + "households", + "tenure/social_rent", + "tenure/private_rent", + ) + }, + } + ), + ) + monkeypatch.setattr( + runtime, "_derive_uk_local_bound_families_from_target_frame", lambda *a, **k: () + ) + receipt = rotated_uk_local_holdout( + _clone_frame(), richer, dataset_households=2, epochs=1 + ) + assert len(calls) == 5 + assert all(n < len(richer.targets) and k == 2 for n, k in calls) + assert np.isfinite(receipt["mean_holdout_loss"]) + + +def test_size_refit_freezes_population_dependent_measures(): + from microcosm.build.uk_runtime.dataset_size import refit_uk_dataset_size + from microcosm.calibrate import Target, TargetSet, calibrate + + frame = _clone_frame() + # A population-normalized measure changes if evaluated on two instead of + # three households. The refit must retain the full-pool value of 1/3. + targets = TargetSet( + [ + Target( + "normalized", + "household", + lambda f: np.full(f.n("household"), 1 / f.n("household")), + 1, + ) + ] + ) + dense = calibrate(frame, targets, epochs=2) + small = refit_uk_dataset_size( + frame, dense, households=2, epochs=2, learning_rate=0.02, seed=7 + ) + np.testing.assert_allclose(small.result.problem.matrix.toarray(), [[1 / 3, 1 / 3]]) + + +def test_full_size_returns_dense_reference_without_search(): + from microcosm.build.uk_runtime.dataset_size import refit_uk_dataset_size + from microcosm.calibrate import Target, TargetSet, calibrate + + frame = _clone_frame() + dense = calibrate( + frame, + TargetSet( + [Target("count", "household", lambda f: np.ones(f.n("household")), 3)] + ), + epochs=1, + ) + full = refit_uk_dataset_size( + frame, dense, households=3, epochs=1, learning_rate=0.02, seed=7 + ) + assert full.result is dense + assert full.support.tolist() == [0, 1, 2] + + +def test_size_refuses_budget_smaller_than_protected_carriers(): + from microcosm.build.uk_runtime.dataset_size import refit_uk_dataset_size + from microcosm.calibrate import Target, TargetSet, calibrate + + frame = _clone_frame() + targets = TargetSet( + [ + Target( + str(i), + "household", + lambda f, i=i: ( + f.table("household").household_id.to_numpy() == i + ).astype(float), + 1, + ) + for i in [101, 102, 103] + ] + ) + dense = calibrate(frame, targets, epochs=1) + with pytest.raises(ValueError, match="3 protected target carriers"): + refit_uk_dataset_size( + frame, dense, households=2, epochs=1, learning_rate=0.02, seed=7 + ) diff --git a/packages/microcosm-build/tests/test_uk_rowwise_candidate.py b/packages/microcosm-build/tests/test_uk_rowwise_candidate.py index 8bd053f90..b4623de8c 100644 --- a/packages/microcosm-build/tests/test_uk_rowwise_candidate.py +++ b/packages/microcosm-build/tests/test_uk_rowwise_candidate.py @@ -1822,3 +1822,83 @@ def test_candidate_multi_block_engine_run_is_never_releasable( "single_block_engine": False, "release_blocking_gates_passed": True, } + + +def test_size_candidate_exports_compact_links_and_cannot_claim_dense_release( + monkeypatch, tmp_path +): + builder = _load_builder_module() + input_h5 = tmp_path / "spine.h5" + ladder_path = tmp_path / "ladder.npz" + out = tmp_path / "k300" + _write_staging_h5(input_h5, households_per_region=52) + ladder = _write_ladder(ladder_path) + import microcosm.build.uk_runtime.battery_bindings as bindings + + monkeypatch.setattr( + bindings, + "_local_area_roster", + lambda _resource, levels: { + "constituency": tuple(sorted(set(ladder.constituency_code))), + "local_authority": tuple(sorted(set(ladder.local_authority_code))), + }, + ) + status = builder.main( + [ + "--input-h5", + str(input_h5), + "--ladder", + str(ladder_path), + "--out", + str(out), + "--n-clones", + "2", + "--dataset-households", + "300", + "--epochs", + "2", + "--skip-holdout", + "--seed", + "7", + ] + ) + assert status in (0, 1) # Gate failures remain reportable candidates. + manifest = json.loads((out / builder.MANIFEST_FILENAME).read_text()) + assert manifest["releasable"] is False + assert manifest["release_posture"]["size_certification_present"] is False + size = manifest["solve"]["dataset_size"] + assert size["requested_households"] == size["realized_households"] == 300 + assert size["pool_households"] == 416 + assert manifest["parameters"]["n_clones"] == 2 + assert ( + manifest["weights"]["stretch_reference"] + == "normalized_horvitz_thompson_w_over_q" + ) + path = out / builder.CANDIDATE_FILENAME_TEMPLATE.format(calibration_year=2025) + with pd.HDFStore(path, "r") as store: + households = store["household"] + persons = store["person"] + benunits = store["benunit"] + assert len(households) == 300 + assert set(persons.person_household_id) == set(households.household_id) + assert set(persons.person_benunit_id) == set(benunits.benunit_id) + assert len(_spool_rows(out)) == 1 + + +def test_size_cli_refuses_promotion_without_separate_certification(tmp_path): + builder = _load_builder_module() + args = builder._parse_args( + [ + "--input-h5", + str(tmp_path / "spine.h5"), + "--ladder", + str(tmp_path / "ladder.npz"), + "--out", + str(tmp_path / "out"), + "--dataset-households", + "50000", + "--release-candidate", + ] + ) + with pytest.raises(ValueError, match="candidate-only"): + builder._validate_cli_args(args) diff --git a/packages/microcosm-calibrate/src/microcosm/calibrate/gates.py b/packages/microcosm-calibrate/src/microcosm/calibrate/gates.py index 306d42615..5874d3803 100644 --- a/packages/microcosm-calibrate/src/microcosm/calibrate/gates.py +++ b/packages/microcosm-calibrate/src/microcosm/calibrate/gates.py @@ -19,6 +19,7 @@ import math +import numpy as np import torch from torch import nn @@ -66,6 +67,10 @@ class HardConcrete(nn.Module): penalty close them). Must lie in ``(0, 1)``. temperature: Concrete-distribution temperature; lower is closer to a hard Bernoulli. + initial_probabilities: Optional per-record actual open probabilities + in (0, 1); overrides the historical scalar initialization. + protected_mask: Optional boolean vector. These records have gates + fixed at one during training and evaluation. Raises: ValueError: If ``init_mean`` is not strictly between 0 and 1, or @@ -78,6 +83,8 @@ def __init__( *, init_mean: float = 0.999, temperature: float = 0.25, + initial_probabilities: np.ndarray | None = None, + protected_mask: np.ndarray | None = None, ) -> None: super().__init__() if not (0.0 < init_mean < 1.0): @@ -88,6 +95,22 @@ def __init__( raise ValueError( f"HardConcrete.temperature must be positive, got {temperature!r}." ) + if initial_probabilities is not None: + probabilities = np.asarray(initial_probabilities, dtype=np.float64) + if ( + probabilities.shape != (n,) + or not np.isfinite(probabilities).all() + or ((probabilities <= 0) | (probabilities >= 1)).any() + ): + raise ValueError( + "initial_probabilities must align with gates and lie in (0, 1)." + ) + if protected_mask is None: + protected_mask = np.zeros(n, dtype=bool) + protected_mask = np.asarray(protected_mask) + if protected_mask.shape != (n,) or protected_mask.dtype != np.bool_: + raise ValueError("protected_mask must be an aligned boolean vector.") + self.register_buffer("protected_mask", torch.tensor(protected_mask)) self.temperature = float(temperature) self.gamma = _GAMMA self.zeta = _ZETA @@ -95,6 +118,12 @@ def __init__( init_val = math.log(init_mean / (1.0 - init_mean)) with torch.no_grad(): self.qz_logits.fill_(init_val) + if initial_probabilities is not None: + # The new vector is an actual open probability, unlike the + # historical scalar init_mean's sigmoid-logit convention. + shift = self.temperature * math.log(-self.gamma / self.zeta) + logits = np.log(probabilities / (1 - probabilities)) + shift + self.qz_logits.copy_(torch.tensor(logits, dtype=self.qz_logits.dtype)) def forward(self) -> torch.Tensor: """Return the current gates: sampled in training, deterministic in eval.""" @@ -105,7 +134,7 @@ def forward(self) -> torch.Tensor: else: s = torch.sigmoid(self.qz_logits) stretched = s * (self.zeta - self.gamma) + self.gamma - return torch.clamp(stretched, 0.0, 1.0) + return torch.where(self.protected_mask, 1.0, torch.clamp(stretched, 0.0, 1.0)) def get_penalty(self) -> torch.Tensor: """Expected number of open gates — the differentiable L0 surrogate. @@ -116,9 +145,13 @@ def get_penalty(self) -> torch.Tensor: ``l0_lambda`` and adds to the loss. """ shift = self.temperature * math.log(-self.gamma / self.zeta) - return torch.sigmoid(self.qz_logits - shift).sum() + return torch.where( + self.protected_mask, 1.0, torch.sigmoid(self.qz_logits - shift) + ).sum() def get_active_prob(self) -> torch.Tensor: """Per-gate open-probability (the per-record version of the penalty).""" shift = self.temperature * math.log(-self.gamma / self.zeta) - return torch.sigmoid(self.qz_logits - shift) + return torch.where( + self.protected_mask, 1.0, torch.sigmoid(self.qz_logits - shift) + ) diff --git a/packages/microcosm-calibrate/src/microcosm/calibrate/initialization.py b/packages/microcosm-calibrate/src/microcosm/calibrate/initialization.py new file mode 100644 index 000000000..f8d15d2ab --- /dev/null +++ b/packages/microcosm-calibrate/src/microcosm/calibrate/initialization.py @@ -0,0 +1,64 @@ +"""Contribution-informed L0 initialization (microcosm#346, #355).""" + +from dataclasses import dataclass + +import numpy as np +from scipy import sparse + + +@dataclass(frozen=True) +class GateInitialization: + """Aligned open probabilities and records excluded from gate pruning.""" + + probabilities: np.ndarray + protected: np.ndarray + + +def contribution_initialization( + matrix: sparse.sparray, + weights: np.ndarray, + targets: np.ndarray, +) -> GateInitialization: + """Initialize search from maximum absolute target-contribution share. + + Protect the largest absolute weighted carrier of each nonzero target + (first column breaks ties deterministically). The smooth bounded prior + uses the median positive share as its scale. It only initializes L0; + neither these scores nor their ranks select the final support. Zero + targets use a unit denominator, matching the default calibration scale. + Work and storage are sparse in the target matrix, never targets x pool. + """ + matrix = sparse.csr_array(matrix, dtype=np.float64, copy=True) + matrix.sum_duplicates() + matrix.eliminate_zeros() + weights = np.asarray(weights, dtype=np.float64) + targets = np.asarray(targets, dtype=np.float64) + if weights.shape != (matrix.shape[1],) or targets.shape != (matrix.shape[0],): + raise ValueError("matrix, weights and targets must align.") + if ( + not np.isfinite(matrix.data).all() + or not np.isfinite(weights).all() + or not np.isfinite(targets).all() + or (weights <= 0).any() + ): + raise ValueError( + "contribution initialization requires finite inputs and positive weights." + ) + scores = np.zeros(len(weights)) + protected = np.zeros(len(weights), dtype=bool) + for row, target in enumerate(targets): + start, stop = matrix.indptr[row : row + 2] + indices = matrix.indices[start:stop] + contributions = np.abs(matrix.data[start:stop]) * weights[indices] + if not len(indices): + if target != 0: + raise ValueError(f"nonzero target at row {row} has no support.") + continue + shares = contributions / (abs(target) if target != 0 else 1.0) + np.maximum.at(scores, indices, shares) + if target != 0: + protected[indices[np.argmax(contributions)]] = True + positive = scores[scores > 0] + scale = float(np.median(positive)) if len(positive) else 1.0 + probabilities = 0.1 + 0.8 * scores / (scores + scale) + return GateInitialization(probabilities, protected) diff --git a/packages/microcosm-calibrate/src/microcosm/calibrate/solve.py b/packages/microcosm-calibrate/src/microcosm/calibrate/solve.py index c37ed85c9..40d0619c8 100644 --- a/packages/microcosm-calibrate/src/microcosm/calibrate/solve.py +++ b/packages/microcosm-calibrate/src/microcosm/calibrate/solve.py @@ -77,6 +77,7 @@ from microcosm.calibrate.exact_k import assert_exact_k_support from microcosm.calibrate.gates import HardConcrete +from microcosm.calibrate.initialization import GateInitialization from microcosm.calibrate.matrix import ( CalibrationProblem, SkippedTarget, @@ -757,6 +758,7 @@ def _optimize( target_loss_cap: float, initial_weights: np.ndarray, *, + gate_initialization: GateInitialization | None = None, warm_start_weights: np.ndarray | None = None, epochs: int, learning_rate: float, @@ -810,7 +812,19 @@ def _optimize( gates: HardConcrete | None = None params: list[torch.Tensor] = [log_w] if l0_lambda > 0.0 or target_records is not None: - gates = HardConcrete(len(w0), init_mean=init_mean, temperature=temperature) + gates = HardConcrete( + len(w0), + init_mean=init_mean, + temperature=temperature, + **( + {} + if gate_initialization is None + else { + "initial_probabilities": gate_initialization.probabilities, + "protected_mask": gate_initialization.protected, + } + ), + ) params = [log_w, *gates.parameters()] optimizer = torch.optim.Adam(params, lr=learning_rate) @@ -1098,6 +1112,7 @@ def _search_l0_lambda_for_budget( target_loss_cap: float, initial_weights: np.ndarray, *, + gate_initialization: GateInitialization | None = None, target_records: int, epochs: int, learning_rate: float, @@ -1189,6 +1204,8 @@ def evaluate( "l0_lambda": lam, }, } + if gate_initialization is not None: + optimize_kwargs["gate_initialization"] = gate_initialization if return_gate_open_probabilities: weights, trajectory, gate_open_probabilities = _optimize( matrix, @@ -1377,6 +1394,7 @@ def calibrate( frame: Frame, targets: TargetSet, *, + gate_initialization: GateInitialization | None = None, weight_entity: str = "household", method: str = "adam", epochs: int = 256, @@ -1478,6 +1496,9 @@ def calibrate( cannot express, e.g. the pre-selection *design* weights during a post-L0 refit. When supplied, ``l2_anchor`` becomes a free-form provenance label recorded in options (e.g. ``"design"``). + gate_initialization: Optional informed L0 prior and protected-record + mask, aligned to the weight entity. Requires an Adam L0 solve; + protected gates remain open throughout training and selection. init_mean: Initial expected open-probability of the L0 gates (only used when pruning). temperature: Hard-concrete temperature (only used when pruning). @@ -1565,6 +1586,9 @@ def calibrate( "total cannot be conserved (sum of caps < input total). Use " "max_weight_ratio >= 1, or mass='free'." ) + if gate_initialization is not None: + if method != "adam" or (target_records is None and l0_lambda <= 0): + raise ValueError("gate_initialization requires an Adam L0 solve.") if target_records is not None and ( not isinstance(target_records, int) or target_records <= 0 ): @@ -1755,6 +1779,11 @@ def calibrate( progress_callback=progress_callback, budget_iters=budget_iters, return_gate_open_probabilities=True, + **( + {} + if gate_initialization is None + else {"gate_initialization": gate_initialization} + ), ) else: effective_l0 = l0_lambda @@ -1780,6 +1809,11 @@ def calibrate( progress_callback=progress_callback, return_gate_open_probabilities=True, selection_receipt=iterate_selection_receipt, + **( + {} + if gate_initialization is None + else {"gate_initialization": gate_initialization} + ), ) n_nonzero = int((final_weights > prune_atol).sum()) if effective_l0 > 0.0 and n_nonzero == 0: @@ -1840,6 +1874,7 @@ def calibrate( target_loss_scales=target_loss_scales_np.copy(), target_loss_cap=target_loss_cap, options={ + "gate_initialization_supplied": gate_initialization is not None, "method": method, "epochs": epochs, "learning_rate": learning_rate, @@ -2012,6 +2047,7 @@ def refit_l0_selection( epochs: int = 256, learning_rate: float = 0.02, mass: str = FREE_MASS, + mass_reason: str | None = None, max_weight_ratio: float | None = None, l2_lambda: float = 0.0, l2_anchor: str = "initial", @@ -2035,6 +2071,8 @@ def refit_l0_selection( gates and L0 penalty are removed and :func:`calibrate` runs again with ``l0_lambda=0`` on only the selected records. + ``mass_reason`` carries a caller's calibration provenance into the refit. + The explicit exact-k path subsets the original ``frame`` rather than ``selection.frame``. A record may have positive open probability while its deterministic L0 gate is closed; starting from the original frame keeps @@ -2184,6 +2222,7 @@ def refit_l0_selection( epochs=epochs, learning_rate=learning_rate, mass=mass, + mass_reason=mass_reason, max_weight_ratio=max_weight_ratio, target_records=None, l0_lambda=0.0, diff --git a/packages/microcosm-calibrate/tests/test_informed_gates.py b/packages/microcosm-calibrate/tests/test_informed_gates.py new file mode 100644 index 000000000..b4a779f0c --- /dev/null +++ b/packages/microcosm-calibrate/tests/test_informed_gates.py @@ -0,0 +1,42 @@ +"""Informed L0 must preserve rare carriers without replacing the search.""" + +import numpy as np +import pytest +import torch +from scipy import sparse + +from microcosm.calibrate.gates import HardConcrete +from microcosm.calibrate.initialization import contribution_initialization + + +def test_small_weight_large_measure_is_protected(): + init = contribution_initialization( + sparse.csr_array([[1000.0, 0, 0], [0, 1, 1]]), + np.array([0.01, 100.0, 100.0]), + np.array([10.0, 200.0]), + ) + assert init.protected[0] + assert init.probabilities[0] > init.probabilities[1] + + +def test_protected_gates_stay_open_during_training_and_evaluation(): + gates = HardConcrete( + 3, + initial_probabilities=np.array([0.2, 0.4, 0.8]), + protected_mask=np.array([True, False, False]), + ) + assert np.allclose(gates.get_active_prob().detach().numpy(), [1, 0.4, 0.8]) + with torch.no_grad(): + gates.qz_logits.fill_(-100) + for training in (True, False): + gates.train(training) + assert gates()[0].item() == 1 + assert gates.get_active_prob()[0].item() == 1 + + +@pytest.mark.parametrize( + "probabilities", [[0, 0.5], [0.5, 1], [float("nan"), 0.5], [0.5]] +) +def test_invalid_initial_probabilities_refuse(probabilities): + with pytest.raises(ValueError): + HardConcrete(2, initial_probabilities=np.array(probabilities)) diff --git a/tools/build_uk_rowwise_candidate.py b/tools/build_uk_rowwise_candidate.py index 101ca0bb2..a7a0a4dd5 100644 --- a/tools/build_uk_rowwise_candidate.py +++ b/tools/build_uk_rowwise_candidate.py @@ -591,6 +591,11 @@ def _parse_args(argv: list[str] | None = None) -> argparse.Namespace: required=True, help="Output directory for the candidate H5 and evidence sidecars.", ) + parser.add_argument( + "--dataset-households", + type=int, + help="Exact output household count after informed L0 and refit; pool clone K is unchanged. Candidate-only until size certification.", + ) parser.add_argument("--n-clones", type=int, default=UK_LOCAL_CLONE_COUNT) parser.add_argument( "--candidate-clone-counts", @@ -841,6 +846,13 @@ def _run_candidate( source_lineage_modulus=args.source_lineage_modulus, ) clone = assignment.result + if ( + args.dataset_households is not None + and args.dataset_households > clone.frame.n("household") + ): + raise ValueError( + "--dataset-households exceeds the cloned pool; selection never clamps the request." + ) if state is not None: append_phase(state, "cloned") @@ -977,6 +989,7 @@ def _run_candidate( learning_rate=args.learning_rate, conserve_mass=_CONSERVE_MASS, target_records=_TARGET_RECORDS, + dataset_households=args.dataset_households, l0_lambda=_L0_LAMBDA, budget_iters=_BUDGET_ITERS, seed=args.seed, @@ -1083,6 +1096,7 @@ def _run_candidate( learning_rate=args.learning_rate, conserve_mass=_CONSERVE_MASS, target_records=_TARGET_RECORDS, + dataset_households=args.dataset_households, l0_lambda=_L0_LAMBDA, budget_iters=_BUDGET_ITERS, solve_seed=args.seed, @@ -1587,6 +1601,7 @@ def _joint_dry_run_plan( }, "candidate_clone_counts": list(args.candidate_clone_counts or (args.n_clones,)), "candidate_clone_support": clone_support, + "parameters": _parameters(args, source_year=source_year), "releasable": False, "engine": "not_run", "ladder_target_provenance": dict(target_provenance), @@ -1995,7 +2010,9 @@ def _dry_run_plan( "fraction": args.sample_fraction, "unreachable_check": "completed", }, - "releasable": args.sample_fraction == 1.0 and args.engine_blocks == 1, + "releasable": args.sample_fraction == 1.0 + and args.engine_blocks == 1 + and args.dataset_households is None, "parameters": _parameters(args, source_year=source_year), "shapes": { "person": list(clone.frame.table("person").shape), @@ -2289,11 +2306,14 @@ def _manifest( }, "abs_delta": abs(new_total - old_total), "declared_stretch_bound": float(UK_LOCAL_MAX_WEIGHT_RATIO), + "stretch_reference": "pool_design" + if solve.size_receipt is None + else "normalized_horvitz_thompson_w_over_q", "realized_max_weight_ratio_vs_design": float( np.max( np.divide( np.asarray(solve.weights, dtype=np.float64), - np.asarray(clone.frame.weights_for("household").values), + np.asarray(solve.initial_weights), ) ) ), @@ -2305,7 +2325,11 @@ def _manifest( "ladder": ladder_rows, "national": int(len(solve.national_diagnostics)), }, - "n_households": int(problem.n_households), + "n_households": int(solve.frame.n("household")), + "pool_households": int(problem.n_households), + "dataset_size": None + if solve.size_receipt is None + else dict(solve.size_receipt), "initial_loss": float(solve.initial_loss), "final_loss": float(solve.final_loss), "max_abs_relative_error": float(abs_errors.max()), @@ -2390,8 +2414,15 @@ def _manifest( for gate_id, payload in gate_rows.items() if not isinstance(payload, Mapping) or payload.get("status") != "passed" ), - "releasable": releasable, - "release_posture": release_posture, + "releasable": releasable and args.dataset_households is None, + "release_posture": { + **release_posture, + **( + {} + if args.dataset_households is None + else {"size_certification_present": False} + ), + }, "ladder_household_uprating": dict( cross_grain.get("ladder_household_uprating") or {"applied": False, "reason": "no cross-grain receipt"} @@ -2421,6 +2452,7 @@ def _manifest( def _parameters(args: argparse.Namespace, *, source_year: int) -> dict[str, Any]: return { "n_clones": int(args.n_clones), + "dataset_households": args.dataset_households, "seed": int(args.seed), "source_year": source_year, "source_lineage_modulus": args.source_lineage_modulus, @@ -2523,7 +2555,12 @@ def _validate_solve_result( raise RuntimeError( "doctrine solve returned no past-cap census; refusing candidate." ) - if len(solve.weights) != problem.n_households: + expected_count = ( + problem.n_households + if solve.selected_support is None + else len(solve.selected_support) + ) + if len(solve.weights) != expected_count: raise RuntimeError( "doctrine solve returned a weight vector with the wrong length." ) @@ -2585,6 +2622,13 @@ def _validate_cli_args(args: argparse.Namespace) -> None: raise ValueError( "the joint registry path requires --input-sha256 and --ladder-sha256." ) + if args.dataset_households is not None: + if args.dataset_households <= 0: + raise ValueError("--dataset-households must be positive.") + if args.release_candidate: + raise ValueError( + "--dataset-households is candidate-only: size-specific matched comparison and promotion scorecard are required before release." + ) if args.release_candidate: required_release = { "--input-sha256": args.input_sha256, From a53236f892e4e9b298dc06ff4ba0b7f350399beb Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Tue, 8 Sep 2026 13:52:59 +0200 Subject: [PATCH 04/51] Size runs keep their dense reference and selection as evidence; --selection-seed moves only the draw (#355) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A --dataset-households run replaces its calibration product with the compact refit, but the dense joint solve it was cut from is byte-identical to a standalone dense run on the same inputs, seed and epochs. Keeping it lets a size candidate be compared with its own dense reference without a second full-pool run: - local_rowwise: the evidence labelling (alignment by name, exact target values, local/national diagnostics, past-cap censuses, initial loss) moves into _doctrine_solve_evidence and runs for both results; the solve product carries `dense_reference` (weights, initial weights, diagnostics, losses, censuses). - driver: a size run writes dense_reference_diagnostics.csv (every target's dense estimate, local rows then national, with a `grain` column) and dataset_size_selection.csv (pool row index, household id, clone index, design weight, inclusion probability, certainty, Horvitz–Thompson baseline, refit weight), lists both under outputs with digests, and summarises the reference under solve.dataset_size (losses, fit by family, uk_weight_summary). Dense runs are unchanged; the two files are size-run-only in the publish order. - --selection-seed (requires --dataset-households; defaults to --seed) seeds only the informed L0 search, the exact-count draw and the refit, threaded through the holdout as well, so two selections compare on one pool and one dense reference; recorded as parameters.selection_seed. Tests: the doctrine solve keeps a reference equal to the standalone dense solve and a selection seed leaves it untouched; the CLI size run writes both sidecars, lists them, and records both seeds; --selection-seed without a size is refused. Co-Authored-By: Claude Fable 5.1 --- changelog.d/uk-dataset-sizes-355.added.md | 1 + .../build/uk_runtime/local_rowwise.py | 299 ++++++++++++------ .../tests/test_uk_local_rowwise.py | 46 +++ .../tests/test_uk_rowwise_candidate.py | 78 +++++ tools/build_uk_rowwise_candidate.py | 135 +++++++- 5 files changed, 464 insertions(+), 95 deletions(-) diff --git a/changelog.d/uk-dataset-sizes-355.added.md b/changelog.d/uk-dataset-sizes-355.added.md index 4e9bea82a..aa8e3a7c9 100644 --- a/changelog.d/uk-dataset-sizes-355.added.md +++ b/changelog.d/uk-dataset-sizes-355.added.md @@ -1 +1,2 @@ Add exact-household-count UK rowwise candidates using contribution-informed L0 initialization, protected target carriers, fixed-size sampling and refitting. Preserve the dense pool doctrine and local gates, export compact linked entities, and record selection provenance without promoting candidates to the certified dense release or changing dataset defaults. +A size run also ships the dense joint solve it was cut from (`dense_reference_diagnostics.csv`, a `dense_reference` summary under `solve.dataset_size`) and the selection itself (`dataset_size_selection.csv`: pool row, design weight, inclusion probability, Horvitz–Thompson baseline, refit weight), and `--selection-seed` re-draws the selection on one pool and one dense reference. diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/local_rowwise.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/local_rowwise.py index 3a1c4cc01..770fc304a 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/local_rowwise.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/local_rowwise.py @@ -153,6 +153,42 @@ class UKRowwiseDoctrineSolve: binding_adjudications: Mapping[str, Any] selected_support: np.ndarray | None = None size_receipt: Mapping[str, Any] | None = None + dense_reference: UKRowwiseDenseReference | None = None + + +@dataclass(frozen=True) +class UKRowwiseDenseReference: + """The dense joint solve a size selection started from, kept as evidence. + + A ``dataset_households`` solve replaces its calibration product with the + compact refit; the full-pool solve it was cut from is byte-identical to + a standalone dense run on the same inputs, seed and epochs, so keeping + its evidence lets a size candidate be compared with its own dense + reference without a second run. Same vocabulary as the solve product: + labelled local and national diagnostics on the doctrine's scale rule, + the microcosm#492 past-cap censuses, the closing loss. + """ + + weights: np.ndarray + initial_weights: np.ndarray + diagnostics: pd.DataFrame + national_diagnostics: pd.DataFrame + initial_loss: float + final_loss: float + n_nonzero: int + past_cap_census: Mapping[str, Any] + national_past_cap_census: Mapping[str, Any] + all_past_cap_census: Mapping[str, Any] + + +@dataclass(frozen=True) +class _DoctrineSolveEvidence: + diagnostics: pd.DataFrame + national_diagnostics: pd.DataFrame + past_cap_census: Mapping[str, Any] + national_past_cap_census: Mapping[str, Any] + all_past_cap_census: Mapping[str, Any] + initial_loss: float def past_cap_census( @@ -1014,9 +1050,14 @@ def solve_uk_rowwise_weights_under_doctrine( l0_lambda: float = 0.0, budget_iters: int = 10, seed: int = 0, + selection_seed: int | None = None, ) -> UKRowwiseDoctrineSolve: """Solve rowwise household weights under the reviewed doctrine. + ``selection_seed`` (default ``seed``) seeds only the size selection — + the informed L0 search, the exact-count draw and the refit — so two + selections can be compared on one pool and one dense reference. + Structurally knob-free like before the ``calibrate()`` migration: no per-target parameters and no doctrine parameter — the bounds always come from :data:`UK_LOCAL_SOLVE_DOCTRINE` and ride into the public front door @@ -1147,6 +1188,7 @@ def solve_uk_rowwise_weights_under_doctrine( ) selected_support = None size_receipt = None + dense_result = None if dataset_households is not None: if target_records is not None or l0_lambda != 0: raise ValueError( @@ -1160,11 +1202,169 @@ def solve_uk_rowwise_weights_under_doctrine( households=dataset_households, epochs=epochs, learning_rate=learning_rate, - seed=seed, + seed=seed if selection_seed is None else selection_seed, ) + dense_result = result result = sized.result selected_support = sized.support size_receipt = sized.receipt + evidence = _doctrine_solve_evidence( + result, + target_set=target_set, + problem=problem, + national_rows=national_rows, + local_count=local_count, + target_loss_weights=target_loss_weights, + doctrine=doctrine, + ) + diagnostics = evidence.diagnostics + national_diagnostics = evidence.national_diagnostics + census = evidence.past_cap_census + national_census = evidence.national_past_cap_census + all_census = evidence.all_past_cap_census + initial_loss = evidence.initial_loss + dense_reference = None + if dense_result is not None: + dense_evidence = _doctrine_solve_evidence( + dense_result, + target_set=target_set, + problem=problem, + national_rows=national_rows, + local_count=local_count, + target_loss_weights=target_loss_weights, + doctrine=doctrine, + ) + dense_reference = UKRowwiseDenseReference( + weights=np.asarray(dense_result.weights, dtype=np.float64), + initial_weights=np.asarray(dense_result.initial_weights, dtype=np.float64), + diagnostics=dense_evidence.diagnostics, + national_diagnostics=dense_evidence.national_diagnostics, + initial_loss=dense_evidence.initial_loss, + final_loss=float(dense_result.final_loss), + n_nonzero=int(dense_result.n_nonzero), + past_cap_census=dense_evidence.past_cap_census, + national_past_cap_census=dense_evidence.national_past_cap_census, + all_past_cap_census=dense_evidence.all_past_cap_census, + ) + + # The UK carrier persists the weight column, and the kernel product is + # immutable with no table-refresh operation, so the finished frame is + # *rebuilt* through the canonical assembler from the kernel product's + # tables, mass log, and period. A rebuild can silently drop kernel + # surfaces the assembler does not carry, so the guards below refuse to + # ship if the kernel product held strata or metadata the rebuilt frame + # lost (both trivially equal today; the guard is the boundary marker + # for the day they are not). + if selected_support is None: + clean_result = result.frame if restore is None else restore(result.frame) + expected_ids = problem.household_ids + else: + # Restore the full prepared carrier before selecting: legacy restorers + # retain full original tables and cannot accept compact weight arrays. + clean_pool = frame if restore is None else restore(frame) + expected_ids = tuple(problem.household_ids[i] for i in selected_support) + clean_subset = clean_pool.select( + clean_pool.table("person")["person_household_id"] + .isin(expected_ids) + .to_numpy() + ) + clean_result = Frame( + {entity: clean_subset.table(entity) for entity in clean_subset.entities}, + clean_subset.schema, + { + entity: result.frame.weights_for(entity) + for entity in result.frame.weighted_entities + }, + clean_subset.strata, + mass_log=result.frame.mass_log, + metadata=result.frame.metadata, + ) + restored_ids = tuple(clean_result.table("household")["household_id"].tolist()) + if restored_ids != expected_ids: + raise ValueError( + "restore returned households that do not match the problem's rows " + "(same ids, same order); weights are written back by position, so a " + "reordering restore would misattribute them silently." + ) + slash_columns = { + entity: [ + str(column) + for column in clean_result.table(entity).columns + if "/" in str(column) + ] + for entity in clean_result.entities + } + slash_columns = { + entity: columns for entity, columns in slash_columns.items() if columns + } + if slash_columns: + raise ValueError( + "prepared slash-named measure columns survived the rowwise solve " + f"restore and cannot reach an H5 writer: {slash_columns}." + ) + calibrated_household = clean_result.table("household").copy() + calibrated_household["household_weight"] = np.asarray( + result.weights, dtype=np.float64 + ) + finished = uk_national_frame( + person=clean_result.table("person"), + benunit=clean_result.table("benunit"), + household=calibrated_household, + time_period=uk_time_period(clean_result), + weight_kind=WeightKind.CALIBRATED, + mass_log=clean_result.mass_log, + ) + validate_uk_national_frame(finished) + if not clean_result.strata.equals(finished.strata): + raise ValueError( + "the calibrated frame carries strata the rebuilt UK national " + "frame would drop; extend uk_national_frame to carry them " + "before shipping a strata-bearing local solve." + ) + if dict(clean_result.metadata) != dict(finished.metadata): + raise ValueError( + "the calibrated frame carries metadata beyond the UK time " + "period; the rebuild would drop it, so the solve refuses " + "instead." + ) + return UKRowwiseDoctrineSolve( + frame=finished, + calibration_result=result, + weights=np.asarray(result.weights, dtype=np.float64), + initial_weights=np.asarray(result.initial_weights, dtype=np.float64), + diagnostics=diagnostics, + national_diagnostics=national_diagnostics, + loss_trajectory=np.asarray(result.loss_trajectory, dtype=np.float64), + initial_loss=initial_loss, + final_loss=float(result.final_loss), + n_nonzero=int(result.n_nonzero), + past_cap_census=census, + national_past_cap_census=national_census, + all_past_cap_census=all_census, + binding_adjudications=binding_adjudications, + selected_support=selected_support, + size_receipt=size_receipt, + dense_reference=dense_reference, + ) + + +def _doctrine_solve_evidence( + result: CalibrationResult, + *, + target_set: TargetSet, + problem: UKRowwiseLocalMatrix, + national_rows: UKRowwiseNationalRows | None, + local_count: int, + target_loss_weights: np.ndarray, + doctrine: Any, +) -> _DoctrineSolveEvidence: + """Label one front-door result against the declared joint surface. + + Shared by the shipped solve and, on a size run, the dense reference it + was cut from: both must compile the whole surface, align by name, and + reproduce the declared target values exactly. + """ + if result.skipped: reasons = [ f"{skipped.target.name}: {skipped.reason}" for skipped in result.skipped[:5] @@ -1294,104 +1494,13 @@ def solve_uk_rowwise_weights_under_doctrine( target_loss_cap=doctrine.target_loss_cap, ) ) - - # The UK carrier persists the weight column, and the kernel product is - # immutable with no table-refresh operation, so the finished frame is - # *rebuilt* through the canonical assembler from the kernel product's - # tables, mass log, and period. A rebuild can silently drop kernel - # surfaces the assembler does not carry, so the guards below refuse to - # ship if the kernel product held strata or metadata the rebuilt frame - # lost (both trivially equal today; the guard is the boundary marker - # for the day they are not). - if selected_support is None: - clean_result = result.frame if restore is None else restore(result.frame) - expected_ids = problem.household_ids - else: - # Restore the full prepared carrier before selecting: legacy restorers - # retain full original tables and cannot accept compact weight arrays. - clean_pool = frame if restore is None else restore(frame) - expected_ids = tuple(problem.household_ids[i] for i in selected_support) - clean_subset = clean_pool.select( - clean_pool.table("person")["person_household_id"] - .isin(expected_ids) - .to_numpy() - ) - clean_result = Frame( - {entity: clean_subset.table(entity) for entity in clean_subset.entities}, - clean_subset.schema, - { - entity: result.frame.weights_for(entity) - for entity in result.frame.weighted_entities - }, - clean_subset.strata, - mass_log=result.frame.mass_log, - metadata=result.frame.metadata, - ) - restored_ids = tuple(clean_result.table("household")["household_id"].tolist()) - if restored_ids != expected_ids: - raise ValueError( - "restore returned households that do not match the problem's rows " - "(same ids, same order); weights are written back by position, so a " - "reordering restore would misattribute them silently." - ) - slash_columns = { - entity: [ - str(column) - for column in clean_result.table(entity).columns - if "/" in str(column) - ] - for entity in clean_result.entities - } - slash_columns = { - entity: columns for entity, columns in slash_columns.items() if columns - } - if slash_columns: - raise ValueError( - "prepared slash-named measure columns survived the rowwise solve " - f"restore and cannot reach an H5 writer: {slash_columns}." - ) - calibrated_household = clean_result.table("household").copy() - calibrated_household["household_weight"] = np.asarray( - result.weights, dtype=np.float64 - ) - finished = uk_national_frame( - person=clean_result.table("person"), - benunit=clean_result.table("benunit"), - household=calibrated_household, - time_period=uk_time_period(clean_result), - weight_kind=WeightKind.CALIBRATED, - mass_log=clean_result.mass_log, - ) - validate_uk_national_frame(finished) - if not clean_result.strata.equals(finished.strata): - raise ValueError( - "the calibrated frame carries strata the rebuilt UK national " - "frame would drop; extend uk_national_frame to carry them " - "before shipping a strata-bearing local solve." - ) - if dict(clean_result.metadata) != dict(finished.metadata): - raise ValueError( - "the calibrated frame carries metadata beyond the UK time " - "period; the rebuild would drop it, so the solve refuses " - "instead." - ) - return UKRowwiseDoctrineSolve( - frame=finished, - calibration_result=result, - weights=np.asarray(result.weights, dtype=np.float64), - initial_weights=np.asarray(result.initial_weights, dtype=np.float64), + return _DoctrineSolveEvidence( diagnostics=diagnostics, national_diagnostics=national_diagnostics, - loss_trajectory=np.asarray(result.loss_trajectory, dtype=np.float64), - initial_loss=initial_loss, - final_loss=float(result.final_loss), - n_nonzero=int(result.n_nonzero), past_cap_census=census, national_past_cap_census=national_census, all_past_cap_census=all_census, - binding_adjudications=binding_adjudications, - selected_support=selected_support, - size_receipt=size_receipt, + initial_loss=initial_loss, ) @@ -1432,6 +1541,7 @@ def rotated_uk_local_holdout( l0_lambda: float = 0.0, budget_iters: int = 10, solve_seed: int = 0, + selection_seed: int | None = None, ) -> dict[str, object]: """Run five local-row rotations with national rows fixed in training.""" @@ -1477,6 +1587,7 @@ def rotated_uk_local_holdout( l0_lambda=l0_lambda, budget_iters=budget_iters, seed=solve_seed, + selection_seed=selection_seed, ) held_targets = problem.targets[holdout_indices] held_estimates = np.asarray( diff --git a/packages/microcosm-build/tests/test_uk_local_rowwise.py b/packages/microcosm-build/tests/test_uk_local_rowwise.py index bf7a838bc..896b9b052 100644 --- a/packages/microcosm-build/tests/test_uk_local_rowwise.py +++ b/packages/microcosm-build/tests/test_uk_local_rowwise.py @@ -1301,6 +1301,52 @@ def restore(full): assert result.selected_support.tolist() == [0, 2] +def test_size_solve_keeps_its_dense_reference_and_selection_seed_moves_only_the_draw(): + frame = _clone_frame() + metrics = pd.DataFrame({"households": [1.0, 1.0, 1.0]}, index=[101, 102, 103]) + problem = build_uk_rowwise_local_matrix( + metrics, + _assigned(), + pd.DataFrame({"code": ["E001", "S001"], "households": [2.0, 1.0]}), + ) + common = dict( + bound_families=["census_households/constituency"], + epochs=2, + seed=7, + ) + dense = solve_uk_rowwise_weights_under_doctrine(frame, problem, **common) + sized = solve_uk_rowwise_weights_under_doctrine( + frame, problem, dataset_households=2, **common + ) + assert dense.dense_reference is None + reference = sized.dense_reference + assert reference is not None + # The reference is the standalone dense solve, byte for byte. + np.testing.assert_array_equal(reference.weights, dense.weights) + np.testing.assert_array_equal(reference.initial_weights, dense.initial_weights) + assert reference.final_loss == dense.final_loss + assert reference.final_loss == sized.size_receipt["dense_loss"] + assert reference.initial_loss == dense.initial_loss + assert reference.n_nonzero == dense.n_nonzero + pd.testing.assert_frame_equal(reference.diagnostics, dense.diagnostics) + pd.testing.assert_frame_equal( + reference.national_diagnostics, dense.national_diagnostics + ) + assert dict(reference.past_cap_census) == dict(dense.past_cap_census) + assert dict(reference.all_past_cap_census) == dict(dense.all_past_cap_census) + # The shipped product is the compact refit, not the reference. + assert sized.weights.size == 2 + assert reference.weights.size == 3 + # A selection seed moves the draw only: same pool, same dense reference. + other = solve_uk_rowwise_weights_under_doctrine( + frame, problem, dataset_households=2, selection_seed=11, **common + ) + np.testing.assert_array_equal(other.dense_reference.weights, reference.weights) + assert other.size_receipt["seed"] == 11 + assert sized.size_receipt["seed"] == 7 + assert other.size_receipt["dense_loss"] == sized.size_receipt["dense_loss"] + + def test_size_holdout_uses_compact_support_and_reselects_each_fold(monkeypatch): import microcosm.build.uk_runtime.local_rowwise as runtime diff --git a/packages/microcosm-build/tests/test_uk_rowwise_candidate.py b/packages/microcosm-build/tests/test_uk_rowwise_candidate.py index b4623de8c..15baf8f17 100644 --- a/packages/microcosm-build/tests/test_uk_rowwise_candidate.py +++ b/packages/microcosm-build/tests/test_uk_rowwise_candidate.py @@ -1855,6 +1855,8 @@ def test_size_candidate_exports_compact_links_and_cannot_claim_dense_release( "2", "--dataset-households", "300", + "--selection-seed", + "11", "--epochs", "2", "--skip-holdout", @@ -1884,6 +1886,82 @@ def test_size_candidate_exports_compact_links_and_cannot_claim_dense_release( assert set(persons.person_benunit_id) == set(benunits.benunit_id) assert len(_spool_rows(out)) == 1 + # The selection seed moves the draw only; the manifest records both seeds. + assert manifest["parameters"]["seed"] == 7 + assert manifest["parameters"]["selection_seed"] == 11 + assert size["seed"] == 11 + + # The dense solve the selection was cut from ships as evidence. + dense = size["dense_reference"] + assert dense["final_loss"] == size["dense_loss"] + assert dense["n_households"] == 416 + assert dense["weights"]["n_records"] == 416 + assert {"effective_sample_size", "max_to_median_positive_weight"} <= set( + dense["weights"] + ) + assert dense["diagnostics_file"] == builder.DENSE_REFERENCE_DIAGNOSTICS_FILENAME + outputs = manifest["outputs"] + dense_csv = out / builder.DENSE_REFERENCE_DIAGNOSTICS_FILENAME + selection_csv = out / builder.DATASET_SIZE_SELECTION_FILENAME + assert Path(outputs["dense_reference_diagnostics"]["path"]) == dense_csv.resolve() + assert Path(outputs["dataset_size_selection"]["path"]) == selection_csv.resolve() + assert ( + outputs["dataset_size_selection"]["sha256"] + == hashlib.sha256(selection_csv.read_bytes()).hexdigest() + ) + dense_rows = pd.read_csv(dense_csv) + assert len(dense_rows) == manifest["solve"]["n_targets"] + assert dense_rows.columns[0] == "grain" + assert {"target", "final_estimate", "abs_relative_error"} <= set(dense_rows.columns) + selection = pd.read_csv(selection_csv) + assert list(selection.columns) == [ + "pool_row_index", + "household_id", + "clone_index", + "design_weight", + "inclusion_probability", + "certainty", + "ht_baseline_weight", + "refit_weight", + ] + assert len(selection) == 300 + assert selection["pool_row_index"].is_unique + assert selection["pool_row_index"].max() < 416 + assert set(selection["household_id"]) == set(households.household_id) + assert (selection["refit_weight"] > 0).all() + assert (selection["design_weight"] > 0).all() + assert ( + int(selection["certainty"].sum()) + == size["selection_receipt"]["certainty_count"] + ) + assert int(selection["certainty"].sum()) == size["protected_carriers"] + + +def test_selection_seed_requires_a_dataset_size(tmp_path): + builder = _load_builder_module() + args = builder._parse_args( + [ + "--input-h5", + str(tmp_path / "spine.h5"), + "--ladder", + str(tmp_path / "ladder.npz"), + "--out", + str(tmp_path / "out"), + "--selection-seed", + "11", + ] + ) + with pytest.raises(ValueError, match="requires --dataset-households"): + builder._validate_cli_args(args) + + +def test_dense_candidate_manifest_has_no_size_sidecars(tmp_path): + builder = _load_builder_module() + paths = builder._output_paths(tmp_path, source_year=2024, calibration_year=2025) + assert paths["dense_reference"].name == builder.DENSE_REFERENCE_DIAGNOSTICS_FILENAME + assert paths["selection"].name == builder.DATASET_SIZE_SELECTION_FILENAME + assert builder._SIZE_RUN_ONLY_OUTPUTS == {"dense_reference", "selection"} + def test_size_cli_refuses_promotion_without_separate_certification(tmp_path): builder = _load_builder_module() diff --git a/tools/build_uk_rowwise_candidate.py b/tools/build_uk_rowwise_candidate.py index a7a0a4dd5..e455b7ec1 100644 --- a/tools/build_uk_rowwise_candidate.py +++ b/tools/build_uk_rowwise_candidate.py @@ -99,6 +99,7 @@ uk_local_target_surface, uk_support_limited_misses, uk_time_period, + uk_weight_summary, write_uk_calibration_diagnostics, write_uk_rowwise_dataset, ) @@ -135,6 +136,11 @@ AREA_SUPPORT_FILENAME = "area_support_summary.csv" PAST_CAP_FILENAME = "past_cap_census.json" LOCAL_REGISTRY_FILENAME = "local_target_registry.json" +DENSE_REFERENCE_DIAGNOSTICS_FILENAME = "dense_reference_diagnostics.csv" +DATASET_SIZE_SELECTION_FILENAME = "dataset_size_selection.csv" + +#: Outputs a run writes only when ``--dataset-households`` is set. +_SIZE_RUN_ONLY_OUTPUTS = frozenset({"dense_reference", "selection"}) _CONSERVE_MASS = False _TARGET_RECORDS: int | None = None @@ -603,6 +609,16 @@ def _parse_args(argv: list[str] | None = None) -> argparse.Namespace: help="Dry-run only comma-separated candidate clone counts.", ) parser.add_argument("--seed", type=int, default=42) + parser.add_argument( + "--selection-seed", + type=int, + help=( + "Seed for the size selection only (informed L0 search, exact-count " + "draw, refit); defaults to --seed. The pool, ladder assignment and " + "dense reference stay on --seed, so two selections compare on one " + "pool. Requires --dataset-households." + ), + ) parser.add_argument( "--sample-fraction", type=float, @@ -993,6 +1009,7 @@ def _run_candidate( l0_lambda=_L0_LAMBDA, budget_iters=_BUDGET_ITERS, seed=args.seed, + selection_seed=args.selection_seed, ) _validate_solve_result(solve, problem=problem) append_phase(state, "solved") @@ -1100,6 +1117,7 @@ def _run_candidate( l0_lambda=_L0_LAMBDA, budget_iters=_BUDGET_ITERS, solve_seed=args.seed, + selection_seed=args.selection_seed, ) args._rotated_holdout = rotated_holdout @@ -2064,6 +2082,13 @@ def _write_output_bundle( ) write_uk_rowwise_dataset(candidate, staged["dataset"]) solve.diagnostics.to_csv(staged["diagnostics"], index=False) + if solve.dense_reference is not None: + _dense_reference_diagnostics_frame(solve).to_csv( + staged["dense_reference"], index=False + ) + _dataset_size_selection_frame(solve, problem=problem, clone=clone).to_csv( + staged["selection"], index=False + ) support = support.copy() support["support_below_floor"] = ( (support["assigned_households"] < 50) @@ -2128,6 +2153,15 @@ def _write_output_bundle( reported_path=output_paths["local_registry"], ), } + if solve.dense_reference is not None: + outputs["dense_reference_diagnostics"] = _artifact_info( + staged["dense_reference"], + reported_path=output_paths["dense_reference"], + ) + outputs["dataset_size_selection"] = _artifact_info( + staged["selection"], + reported_path=output_paths["selection"], + ) manifest = _manifest( args, candidate=candidate, @@ -2329,7 +2363,10 @@ def _manifest( "pool_households": int(problem.n_households), "dataset_size": None if solve.size_receipt is None - else dict(solve.size_receipt), + else { + **dict(solve.size_receipt), + "dense_reference": _dense_reference_summary(solve), + }, "initial_loss": float(solve.initial_loss), "final_loss": float(solve.final_loss), "max_abs_relative_error": float(abs_errors.max()), @@ -2454,6 +2491,9 @@ def _parameters(args: argparse.Namespace, *, source_year: int) -> dict[str, Any] "n_clones": int(args.n_clones), "dataset_households": args.dataset_households, "seed": int(args.seed), + "selection_seed": None + if args.dataset_households is None + else int(args.seed if args.selection_seed is None else args.selection_seed), "source_year": source_year, "source_lineage_modulus": args.source_lineage_modulus, "sample_fraction": float(args.sample_fraction), @@ -2498,6 +2538,91 @@ def _fit_by_family(diagnostics: pd.DataFrame) -> list[dict[str, object]]: return rows +def _dense_reference_summary(solve: UKRowwiseDoctrineSolve) -> dict[str, Any] | None: + """Manifest-sized evidence of the dense solve a size run was cut from.""" + + dense = solve.dense_reference + if dense is None: + return None + local_errors = dense.diagnostics["abs_relative_error"].to_numpy(dtype=np.float64) + national_errors = dense.national_diagnostics["abs_relative_error"].to_numpy( + dtype=np.float64 + ) + past_cap = dict(dense.past_cap_census or {}) + return { + "initial_loss": float(dense.initial_loss), + "final_loss": float(dense.final_loss), + "n_nonzero": int(dense.n_nonzero), + "n_households": int(dense.weights.size), + "max_abs_relative_error": float(local_errors.max()) + if local_errors.size + else None, + "median_abs_relative_error": float(np.median(local_errors)) + if local_errors.size + else None, + "national_max_abs_relative_error": float(national_errors.max()) + if national_errors.size + else None, + "past_cap": {key: int(past_cap[key]) for key in _PAST_CAP_COUNT_KEYS}, + "weights": uk_weight_summary(dense.weights), + "local_by_family": _fit_by_family(dense.diagnostics), + "national_by_family": _fit_by_family(dense.national_diagnostics), + "diagnostics_file": DENSE_REFERENCE_DIAGNOSTICS_FILENAME, + } + + +def _dense_reference_diagnostics_frame(solve: UKRowwiseDoctrineSolve) -> pd.DataFrame: + """Every target's dense-reference estimate, local rows then national rows.""" + + dense = solve.dense_reference + assert dense is not None + local = dense.diagnostics.copy() + local.insert(0, "grain", local["area_type"].astype(str)) + national = dense.national_diagnostics.copy() + national.insert(0, "grain", "national") + return pd.concat([local, national], ignore_index=True, sort=False) + + +def _dataset_size_selection_frame( + solve: UKRowwiseDoctrineSolve, + *, + problem: UKRowwiseLocalMatrix, + clone: UKLadderRowwiseDatasetResult, +) -> pd.DataFrame: + """One row per selected pool household: identity, design, draw and refit.""" + + dense = solve.dense_reference + receipt = solve.size_receipt + assert dense is not None and receipt is not None + support = np.asarray(solve.selected_support, dtype=np.int64) + household = clone.frame.table("household") + clone_column = ladder_clone_index_column("household") + ids = household["household_id"].to_numpy()[support] + expected = np.asarray([problem.household_ids[i] for i in support]) + if not np.array_equal(ids, expected): + raise RuntimeError( + "the cloned pool's household order does not match the solve's " + "matrix columns; the selection sidecar would misattribute rows." + ) + inclusion = np.asarray(receipt["inclusion_probabilities"], dtype=np.float64) + if inclusion.shape != support.shape: + raise RuntimeError("selection receipt inclusion probabilities are misaligned.") + return pd.DataFrame( + { + "pool_row_index": support, + "household_id": ids, + "clone_index": household[clone_column].to_numpy()[support] + if clone_column in household.columns + else np.zeros(support.size, dtype=np.int64), + "design_weight": dense.initial_weights[support], + "inclusion_probability": inclusion, + "certainty": inclusion >= 1.0, + "ht_baseline_weight": np.asarray(solve.initial_weights, dtype=np.float64), + "refit_weight": np.asarray(solve.weights, dtype=np.float64), + } + ) + + def _local_output_registry( problem: UKRowwiseLocalMatrix, *, @@ -2622,6 +2747,8 @@ def _validate_cli_args(args: argparse.Namespace) -> None: raise ValueError( "the joint registry path requires --input-sha256 and --ladder-sha256." ) + if args.selection_seed is not None and args.dataset_households is None: + raise ValueError("--selection-seed requires --dataset-households.") if args.dataset_households is not None: if args.dataset_households <= 0: raise ValueError("--dataset-households must be positive.") @@ -2709,6 +2836,8 @@ def _output_paths( "local_gates": out_dir / LOCAL_GATE_REPORT_FILENAME_TEMPLATE.format(calibration_year=calibration_year), "local_registry": out_dir / LOCAL_REGISTRY_FILENAME, + "dense_reference": out_dir / DENSE_REFERENCE_DIAGNOSTICS_FILENAME, + "selection": out_dir / DATASET_SIZE_SELECTION_FILENAME, } @@ -2749,12 +2878,16 @@ def _publish_staged_files( "past_cap", "calibration_diagnostics", "local_registry", + "dense_reference", + "selection", "manifest", ) published: list[Path] = [] succeeded = False try: for key in publish_order: + if key in _SIZE_RUN_ONLY_OUTPUTS and not staged[key].exists(): + continue destination = output_paths[key] if destination.exists(): raise FileExistsError( From 115bc9184cf23c15779b9ffa92a0febb6ef7d519 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Tue, 8 Sep 2026 14:33:27 +0200 Subject: [PATCH 05/51] Carry the exact-count feasibility measurement with the size refusal and receipt (#355) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first licensed size rehearsal (55,000 of 792,690 at 100 epochs) was refused inside select_exact_k: "degenerate boundary mass … adjust pi_hi or k". With pi_hi=1.0 every gate below one stays in the boundary and its open probability is scaled to the remaining draw size m, so the design is feasible only when m * max(pi_boundary) <= sum(pi_boundary); the L0 budget search stops on the count of not-fully-closed gates, which sits above the open-probability mass while gates are only partly polarised. The refusal carried no numbers. refit_uk_dataset_size now measures selection_feasibility before the draw — certainties and boundary draw at pi_hi=1, boundary mass and largest boundary gate, feasibility, the largest household count feasible at pi_hi=1, the smallest feasible pi_hi on a fixed grid, gate quantiles and counts, the budget search's n_nonzero and lambda — and attaches it to the ValueError on refusal and to the size receipt on success. Nothing is clamped or promoted; the ruling on pi_hi stays a reviewed decision, now made from measured mass. Co-Authored-By: Claude Fable 5.1 --- .../build/uk_runtime/dataset_size.py | 118 +++++++++++++++++- .../tests/test_uk_local_rowwise.py | 68 ++++++++++ 2 files changed, 184 insertions(+), 2 deletions(-) diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/dataset_size.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/dataset_size.py index 1a61ff4f8..eb7457670 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/dataset_size.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/dataset_size.py @@ -1,5 +1,6 @@ """Exact household-count UK candidates on a fixed, materialized target surface.""" +import json from dataclasses import dataclass, replace import numpy as np @@ -110,9 +111,24 @@ def refit_uk_dataset_size( raise RuntimeError("informed L0 returned no selection probabilities.") # Exact-one certainties are protected by the gates themselves; no top-k # ranking or post-hoc promotion of learned boundary scores is performed. - support, sampling, q = select_exact_k( - probabilities, households, pi_hi=1.0, seed=seed + feasibility = selection_feasibility( + probabilities, + households, + protected=init.protected, + n_nonzero=int(selection.n_nonzero), + l0_lambda=float(selection.l0_lambda), ) + try: + support, sampling, q = select_exact_k( + probabilities, households, pi_hi=1.0, seed=seed + ) + except ValueError as error: + # The draw refuses rather than clamps; carry the measured gate mass + # with the refusal so the ruling it needs can be made from the receipt. + raise ValueError( + f"{error} Selection feasibility at pi_hi=1.0: " + f"{json.dumps(feasibility, sort_keys=True)}" + ) from error support = assert_exact_k_support(support, households, pool_size=n) if not np.isin(np.flatnonzero(init.protected), support).all(): raise RuntimeError("exact-count selection lost a protected carrier.") @@ -148,6 +164,7 @@ def refit_uk_dataset_size( "seed": seed, "protected_carriers": int(init.protected.sum()), "selection_receipt": sampling, + "selection_feasibility": feasibility, "selection_l0_lambda": selection.l0_lambda, "selection_epochs": epochs, "refit_epochs": epochs, @@ -165,6 +182,103 @@ def refit_uk_dataset_size( ) +_FEASIBILITY_PI_HI_GRID = (0.999, 0.99, 0.98, 0.95, 0.9, 0.8, 0.7, 0.5) + + +def selection_feasibility( + probabilities: np.ndarray, + households: int, + *, + protected: np.ndarray, + n_nonzero: int, + l0_lambda: float, +) -> dict[str, object]: + """Measure whether an exact-count draw is feasible on these gate probabilities. + + ``select_exact_k`` with ``pi_hi=1.0`` keeps every gate below one in the + boundary and scales its open probabilities to the remaining draw size + ``m``; the scaled value of the largest boundary gate must not exceed one, + i.e. ``m * max(pi_boundary) <= sum(pi_boundary)``. The L0 budget search + stops on the count of not-fully-closed gates, which can sit well above the + open-probability mass when gates are only partly polarised, so the draw + can refuse. This records the measured mass and the two ways out — the + smallest certainty threshold on a fixed grid that makes the design + feasible, and the largest household count feasible at ``pi_hi=1.0`` — so + the refusal is a ruling with numbers, never a silent clamp. + """ + + pi = np.asarray(probabilities, dtype=np.float64) + protected_mask = np.asarray(protected, dtype=bool) + if pi.shape != protected_mask.shape: + raise ValueError("selection feasibility needs aligned probabilities and mask.") + certainty = pi >= 1.0 + boundary = pi[~certainty] + positive = boundary[boundary > 0.0] + m = int(households) - int(certainty.sum()) + boundary_mass = float(positive.sum()) if positive.size else 0.0 + boundary_max = float(positive.max()) if positive.size else 0.0 + feasible = m <= 0 or ( + positive.size >= m and boundary_max * m <= boundary_mass * (1.0 + 1e-12) + ) + max_feasible_k = ( + int(certainty.sum()) + int(np.floor(boundary_mass / boundary_max)) + if boundary_max > 0.0 + else int(certainty.sum()) + ) + scan: dict[str, object] = {} + smallest_feasible_pi_hi: float | None = 1.0 if feasible else None + for threshold in _FEASIBILITY_PI_HI_GRID: + certain_t = pi >= threshold + c_t = int(certain_t.sum()) + m_t = int(households) - c_t + boundary_t = pi[~certain_t] + positive_t = boundary_t[boundary_t > 0.0] + mass_t = float(positive_t.sum()) if positive_t.size else 0.0 + max_t = float(positive_t.max()) if positive_t.size else 0.0 + ok = m_t >= 0 and ( + m_t == 0 + or (positive_t.size >= m_t and max_t * m_t <= mass_t * (1.0 + 1e-12)) + ) + scan[f"{threshold:g}"] = { + "certainties": c_t, + "boundary_draw": m_t, + "boundary_mass": mass_t, + "boundary_max": max_t, + "feasible": bool(ok), + } + if ok and smallest_feasible_pi_hi is None: + smallest_feasible_pi_hi = threshold + quantiles = ( + { + f"p{q * 100:g}": float(np.quantile(pi, q)) + for q in (0.5, 0.9, 0.99, 0.999) + } + if pi.size + else {} + ) + return { + "requested_households": int(households), + "pool_households": int(pi.size), + "protected_carriers": int(protected_mask.sum()), + "certainties_at_pi_hi_1": int(certainty.sum()), + "boundary_draw": m, + "boundary_positive_gates": int(positive.size), + "boundary_mass": boundary_mass, + "boundary_max": boundary_max, + "feasible_at_pi_hi_1": bool(feasible), + "max_feasible_households_at_pi_hi_1": max_feasible_k, + "smallest_feasible_pi_hi_on_grid": smallest_feasible_pi_hi, + "pi_hi_scan": scan, + "pi_sum": float(pi.sum()), + "pi_quantiles": quantiles, + "gates_above": { + f"{t:g}": int((pi >= t).sum()) for t in (0.5, 0.9, 0.99, 0.999) + }, + "budget_search_n_nonzero": int(n_nonzero), + "selection_l0_lambda": float(l0_lambda), + } + + def _frozen_targets( frame: Frame, dense: CalibrationResult, support: np.ndarray ) -> TargetSet: diff --git a/packages/microcosm-build/tests/test_uk_local_rowwise.py b/packages/microcosm-build/tests/test_uk_local_rowwise.py index 896b9b052..35f9002aa 100644 --- a/packages/microcosm-build/tests/test_uk_local_rowwise.py +++ b/packages/microcosm-build/tests/test_uk_local_rowwise.py @@ -1347,6 +1347,74 @@ def test_size_solve_keeps_its_dense_reference_and_selection_seed_moves_only_the_ assert other.size_receipt["dense_loss"] == sized.size_receipt["dense_loss"] +def test_selection_feasibility_measures_boundary_mass_and_the_ways_out(): + from microcosm.build.uk_runtime.dataset_size import selection_feasibility + + # Two protected certainties, then a boundary whose largest gate (0.9) + # cannot be scaled to a draw of 5 out of mass 0.9 + 4 * 0.2 = 1.7. + pi = np.array([1.0, 1.0, 0.9, 0.2, 0.2, 0.2, 0.2, 0.0]) + protected = np.array([True, True, False, False, False, False, False, False]) + report = selection_feasibility( + pi, 7, protected=protected, n_nonzero=7, l0_lambda=0.5 + ) + assert report["certainties_at_pi_hi_1"] == 2 + assert report["boundary_draw"] == 5 + assert report["boundary_positive_gates"] == 5 + assert report["boundary_mass"] == pytest.approx(1.7) + assert report["boundary_max"] == pytest.approx(0.9) + assert report["feasible_at_pi_hi_1"] is False + # At pi_hi=1 the boundary supports floor(1.7 / 0.9) = 1 draw: k <= 3. + assert report["max_feasible_households_at_pi_hi_1"] == 3 + # Promoting the 0.9 gate to a certainty (pi_hi <= 0.9) leaves a draw of 4 + # over four equal 0.2 gates: feasible; 0.99 and above are not. + assert report["pi_hi_scan"]["0.99"]["feasible"] is False + assert report["pi_hi_scan"]["0.9"]["feasible"] is True + assert report["smallest_feasible_pi_hi_on_grid"] == 0.9 + assert report["gates_above"]["0.5"] == 3 + assert report["budget_search_n_nonzero"] == 7 + assert report["selection_l0_lambda"] == 0.5 + + feasible = selection_feasibility( + np.array([1.0, 0.5, 0.5, 0.5, 0.5]), + 3, + protected=np.array([True, False, False, False, False]), + n_nonzero=5, + l0_lambda=0.1, + ) + assert feasible["feasible_at_pi_hi_1"] is True + assert feasible["smallest_feasible_pi_hi_on_grid"] == 1.0 + + +def test_size_refusal_at_the_draw_carries_the_feasibility_numbers(monkeypatch): + import microcosm.build.uk_runtime.dataset_size as sizing + from microcosm.calibrate import Target, TargetSet, calibrate + + frame = _clone_frame() + dense = calibrate( + frame, + TargetSet( + [Target("count", "household", lambda f: np.ones(f.n("household")), 3)] + ), + epochs=2, + ) + + def refuse(*_args, **_kwargs): + raise ValueError("degenerate boundary mass: synthetic refusal.") + + monkeypatch.setattr(sizing, "select_exact_k", refuse) + with pytest.raises(ValueError) as caught: + sizing.refit_uk_dataset_size( + frame, dense, households=2, epochs=2, learning_rate=0.02, seed=7 + ) + message = str(caught.value) + assert message.startswith("degenerate boundary mass: synthetic refusal.") + assert "Selection feasibility at pi_hi=1.0:" in message + payload = json.loads(message.split("Selection feasibility at pi_hi=1.0: ", 1)[1]) + assert payload["requested_households"] == 2 + assert payload["pool_households"] == 3 + assert "pi_hi_scan" in payload + + def test_size_holdout_uses_compact_support_and_reselects_each_fold(monkeypatch): import microcosm.build.uk_runtime.local_rowwise as runtime From 544f957cc3878db2d836e977648ab0b92d958c88 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mar=C3=ADa=20Juaristi?= <127882282+juaristi22@users.noreply.github.com> Date: Tue, 8 Sep 2026 14:52:49 +0200 Subject: [PATCH 06/51] Evaluate a size candidate with one command: size_evaluation library, evaluate_uk_dataset_size steps 00-90, uk_fit_by_family lifted (#355) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every evaluation piece for a --dataset-households candidate now has one name and one place: - uk_runtime/size_evaluation.py — pure measurement over a run directory: load_run (tolerant of pre-#877 dense runs), run_acceptance (the size-run checklist, not_applicable on dense runs), fit_tables (by grain and family), weight_tables (Kish ESS, distinct sources, stretch vs the Horvitz–Thompson baseline and vs pool design from the selection sidecar or a spine, self-checked against the manifest), area_support_tables (per-grain floors and breaches), gate_table (six ids with criticality), paired_targets (join on name; wins/ties/losses; the reference's red rows tracked through family/area/metric), dense_reference_deltas (size-only effect from the run's own dense reference), frozen_vs_recomputed (national rows against the incumbent-surface evaluator), footprint, and summarize with PRE_REGISTERED_OUTCOMES_V1 flags. No subprocess, no engine, no licensed data in tests. - tools/evaluate_uk_dataset_size.py — the one command: steps 00-run-acceptance, 10-dense-reference, 20-vs-reference/