From 0b853706675e0b25f7f992a724d173439a2e264c Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 29 Aug 2026 18:12:32 -0400 Subject: [PATCH 1/4] Validate the Belgian calibration target contract --- changelog.d/be-calibration-contract.added.md | 1 + .../microcosm/build/be/target_references.json | 154 +++++++- .../src/microcosm/build/country_spec.py | 347 ++++++++++++++++++ .../src/microcosm/build/ledger_targets.py | 60 ++- .../tests/golden/be_country_spec.json | 4 +- .../tests/test_country_spec.py | 296 +++++++++++++++ .../tests/test_ledger_targets.py | 124 +++++++ 7 files changed, 964 insertions(+), 22 deletions(-) create mode 100644 changelog.d/be-calibration-contract.added.md diff --git a/changelog.d/be-calibration-contract.added.md b/changelog.d/be-calibration-contract.added.md new file mode 100644 index 000000000..190aa0c44 --- /dev/null +++ b/changelog.d/be-calibration-contract.added.md @@ -0,0 +1 @@ +Add a value-free Belgian calibration target contract with Chronicle selector pins, explicit BE-SILC income-reference periods, criticality-tier tolerances, exact projection handling, and geography-vintage validation. diff --git a/packages/microcosm-build/src/microcosm/build/be/target_references.json b/packages/microcosm-build/src/microcosm/build/be/target_references.json index 3efc33b66..f2847f6ef 100644 --- a/packages/microcosm-build/src/microcosm/build/be/target_references.json +++ b/packages/microcosm-build/src/microcosm/build/be/target_references.json @@ -1,108 +1,228 @@ { "country": "be", - "description": "Initial Belgian target references, by reference only — observed values live in Ledger. Source names bind to the ledger-be packages (PolicyEngine/ledger#69); the full profile with criticality tiers and basis-period declarations is the target-profile issue (PolicyEngine/ledger#70). Every reference here names its intended series and publisher so the profile work binds facts without any populace-side value copying.", - "allowed_value_operations": ["identity"], + "description": "Belgian calibration and validation references, by reference only. Values and source projections resolve from Chronicle consumer facts at build time. The consumer contract names every target's criticality tier, relative tolerance, basis period, and subnational geography vintage; it never copies a target value from Chronicle, a survey publication, or a validation oracle.", + "allowed_value_operations": [ + "identity" + ], "target_references": [ { "name": "statbel_population_by_age_sex_region", "ledger_selector": { "source_name": "statbel_population_structure", - "geography_level": "nuts1" + "source_measure_id": "people", + "period_type": "calendar_year", + "geography_level": "nuts1", + "geography_vintage": "NUTS_2024" }, "entity": "person", "measure": "people", + "period": 2023, "family": "demography", + "assertion_policy": "allow_source_projection", + "period_match_policy": "exact", "metadata": { + "basis_period": "population_reference_2023", "criticality": "release_blocking", + "criticality_tier": "demography_release", + "geography_vintage": "NUTS_2024", "publisher": "Statbel", - "series": "Structure of the population: age band by sex by region" + "series": "Structure of the population: age band by sex by region", + "target_role": "calibration" }, - "notes": "The demographic anchor (age band x sex x NUTS1), mirroring the US demographics posture. Reference-date population; basis period declared by the profile." + "notes": "The demographic anchor is an age-band x sex x NUTS1 count surface. An observation for another year may enter only as a Chronicle source_projection whose fact period is the declared 2023 reference year." }, { "name": "statbel_fiscal_income_by_commune", "ledger_selector": { "source_name": "statbel_fiscal_income", - "geography_level": "commune" + "source_measure_id": "taxable_income", + "period_type": "tax_year", + "geography_level": "commune", + "geography_vintage": "nis_2025" }, "entity": "household", "measure": "belgium_pit_taxable_income", + "period": 2022, "family": "fiscal_income", + "assertion_policy": "allow_source_projection", + "period_match_policy": "exact", "metadata": { + "basis_period": "assessment_income_year_2022", "criticality": "diagnostic", + "criticality_tier": "commune_diagnostic", + "geography_vintage": "nis_2025", + "nis_vintage": "2025", "publisher": "Statbel", "series": "Fiscal statistics of income: total net taxable income by municipality", - "nis_vintage": "2025" + "target_role": "calibration" }, - "notes": "The flagship open subnational series. Commune rows start diagnostic-only until the spine demonstrably supports them (the US congressional-district posture, populace#204); geography vintage must match the spine's declared NIS vintage or compilation fails." + "notes": "The open commune series is diagnostic until the synthetic spine demonstrates adequate support. The selector binds Chronicle's nis_2025 spelling to the spine's 2025 NIS code set, and another code set must be projected and crosswalked in Chronicle rather than partially joined." }, { "name": "spf_finances_pit_total", "ledger_selector": { "source_name": "spf_finances_pit", + "source_measure_id": "tax_before_withholding", + "period_type": "tax_year", "geography_level": "country" }, "entity": "person", "measure": "belgium_pit_federal_and_local_tax_before_withholding", + "period": 2022, "family": "income_tax", + "assertion_policy": "allow_source_projection", + "period_match_policy": "exact", "metadata": { + "basis_period": "assessment_income_year_2022", "criticality": "release_blocking", + "criticality_tier": "core_fiscal_release", "publisher": "SPF Finances / Statbel", - "series": "Personal income tax: assessed total" + "series": "Personal income tax: assessed total", + "target_role": "calibration" }, - "notes": "National PIT aggregate against the Axiom-computed assessed liability (rulespec-be federal tax after Article 134 plus Articles 465-468 local additions). Assessment-year vs income-year basis is declared by the profile; a stale-level rerun of the populace#212 lesson is what the projection basis (PolicyEngine/ledger#71) exists to prevent." + "notes": "National assessed PIT is compared with the Axiom-computed federal and local liability on the 2022 income-year basis. Assessment-year facts must identify the represented income year in Chronicle; a period mismatch is a projection, never a silent relabeling." }, { "name": "onss_employee_contribution_total", "ledger_selector": { "source_name": "onss_contributions", + "source_measure_id": "worker_article_17_uncapped_component_contribution", + "period_type": "calendar_year", "geography_level": "country" }, "entity": "person", "measure": "belgium_worker_article_17_uncapped_component_contribution", + "period": 2022, "family": "social_security", + "assertion_policy": "allow_source_projection", + "period_match_policy": "exact", "metadata": { + "basis_period": "calendar_year_2022", "criticality": "release_blocking", + "criticality_tier": "standard_admin_release", "publisher": "ONSS/RSZ", - "series": "Employee social-security contributions" + "series": "Employee social-security contributions", + "target_role": "calibration" }, - "notes": "Worker-side article 17 contributions on the encoded slice; employer-side and capped-regime totals join as the encodings and ledger-be packages grow." + "notes": "Worker-side contributions on the encoded slice. Employer-side and capped-regime totals can enter as distinct references when both the engine concepts and Ledger selectors exist." }, { "name": "onem_unemployment_caseload", "ledger_selector": { "source_name": "onem_rva_unemployment", + "source_measure_id": "receives_unemployment_benefit", + "period_type": "calendar_year", "geography_level": "country" }, "entity": "person", "measure": "receives_unemployment_benefit", + "period": 2022, "family": "caseloads", + "assertion_policy": "allow_source_projection", + "period_match_policy": "exact", "metadata": { + "basis_period": "calendar_year_2022", "criticality": "release_blocking", + "criticality_tier": "caseload_release", "publisher": "ONEM/RVA", - "series": "Unemployment benefit recipients" + "series": "Unemployment benefit recipients", + "target_role": "calibration" }, - "notes": "First caseload family; pensions (SFPD) and the post-2019 regional child-benefit caseloads (per region) are declared with the ledger-be packages and enter through the profile." + "notes": "The initial caseload family uses the calendar-year recipient count. Pension and regional child-benefit caseload references can join this family when their Chronicle packages and model mappings land." }, { "name": "nbb_household_disposable_income", "ledger_selector": { "source_name": "nbb_national_accounts", + "source_measure_id": "household_disposable_income", + "period_type": "calendar_year", "geography_level": "country" }, "entity": "household", "measure": "household_disposable_income", + "period": 2022, "family": "national_accounts", + "assertion_policy": "allow_source_projection", + "period_match_policy": "exact", "metadata": { + "basis_period": "calendar_year_2022", "criticality": "diagnostic", + "criticality_tier": "validation_only", "publisher": "NBB", - "series": "Household-sector disposable income (national accounts)" + "series": "Household-sector disposable income (national accounts)", + "target_role": "validation" }, - "notes": "Validation-tier levels anchor, never hard-calibrated: the concept bridge between survey/fiscal income and national accounts makes exact hits meaningless (the populace#212 levels-sanity lesson); the macro-realism gate reads it as a band." + "notes": "Validation-tier level anchor only. The macro-realism gate may use it as a broad band, but neither the exact NBB level nor any survey- or EUROMOD-derived quantity enters the calibration objective." } ], "target_profile": { - "schema_version": 1, + "schema_version": 2, + "required_families": [ + "demography", + "fiscal_income", + "income_tax", + "social_security", + "caseloads" + ], + "criticality_tiers": { + "demography_release": { + "criticality": "release_blocking", + "relative_tolerance": 0.02, + "description": "Population-structure cells: plus or minus 2 percent." + }, + "core_fiscal_release": { + "criticality": "release_blocking", + "relative_tolerance": 0.05, + "description": "Core fiscal totals, including assessed PIT: plus or minus 5 percent." + }, + "standard_admin_release": { + "criticality": "release_blocking", + "relative_tolerance": 0.1, + "description": "Administrative contribution totals: plus or minus 10 percent." + }, + "caseload_release": { + "criticality": "release_blocking", + "relative_tolerance": 0.15, + "description": "Administrative caseload totals: plus or minus 15 percent." + }, + "commune_diagnostic": { + "criticality": "diagnostic", + "relative_tolerance": 0.1, + "description": "Commune fiscal-income cells: plus or minus 10 percent, reported but not release-blocking." + }, + "validation_only": { + "criticality": "diagnostic", + "relative_tolerance": null, + "description": "Validation oracle only; no calibration tolerance." + } + }, + "basis_periods": { + "population_reference_2023": { + "period": 2023, + "basis": "reference_date", + "fact_period_type": "calendar_year", + "mismatch_policy": "requires_source_projection", + "description": "Population stock in the BE-SILC 2023 survey year." + }, + "assessment_income_year_2022": { + "period": 2022, + "basis": "income_year", + "fact_period_type": "tax_year", + "survey_year": 2023, + "income_reference_offset_years": -1, + "mismatch_policy": "requires_source_projection", + "description": "BE-SILC survey year 2023 carries 2022 incomes; assessed fiscal facts bind by represented income year." + }, + "calendar_year_2022": { + "period": 2022, + "basis": "calendar_year", + "fact_period_type": "calendar_year", + "survey_year": 2023, + "income_reference_offset_years": -1, + "mismatch_policy": "requires_source_projection", + "description": "Calendar-year administrative or national-accounts fact aligned to the 2022 income reference year." + } + }, "hierarchy_reconciliations": [] } } diff --git a/packages/microcosm-build/src/microcosm/build/country_spec.py b/packages/microcosm-build/src/microcosm/build/country_spec.py index 4d44f1969..0d8af4d50 100644 --- a/packages/microcosm-build/src/microcosm/build/country_spec.py +++ b/packages/microcosm-build/src/microcosm/build/country_spec.py @@ -33,6 +33,7 @@ from __future__ import annotations import json +import math import re from collections.abc import Callable, Mapping from dataclasses import dataclass, field @@ -819,6 +820,338 @@ def _validate_target_references( return tuple(references) +def _validate_target_profile( + raw: Mapping[str, Any], + references: tuple[LedgerTargetReference, ...], + *, + country: str, + geography_spine: GeographySpineManifest | None, +) -> Mapping[str, Any]: + """Validate the versioned consumer-side calibration selection contract. + + Schema 1 remains the historical hierarchy-only profile. Schema 2 adds the + policy fields needed to make a target surface auditable without carrying + target values: named criticality/tolerance tiers, named basis periods, an + exact-period posture, and explicit geography vintages on every + subnational selector. + """ + + profile = raw.get("target_profile", {}) + if not isinstance(profile, Mapping): + raise ValueError("target_references.json: target_profile must be an object.") + schema_version = profile.get("schema_version", 1) + if schema_version == 1: + return _freeze_gate_parameter(profile, path="target_profile") + if schema_version != 2: + raise ValueError( + "target_references.json: target_profile.schema_version must be 1 " + f"or 2, got {schema_version!r}." + ) + + allowed_profile_keys = { + "schema_version", + "required_families", + "criticality_tiers", + "basis_periods", + "hierarchy_reconciliations", + } + unknown_profile_keys = sorted(set(profile) - allowed_profile_keys) + if unknown_profile_keys: + raise ValueError( + "target_references.json: target_profile has unknown key(s) " + f"{unknown_profile_keys}." + ) + + raw_families = profile.get("required_families") + if not isinstance(raw_families, list) or not raw_families: + raise ValueError( + "target_references.json: target_profile.required_families must be " + "a non-empty list." + ) + required_families = tuple( + _require_non_empty_string( + family, + field_name="required_families entry", + context="target_references.json", + ) + for family in raw_families + ) + if len(set(required_families)) != len(required_families): + raise ValueError( + "target_references.json: target_profile.required_families must be unique." + ) + + tiers = profile.get("criticality_tiers") + if not isinstance(tiers, Mapping) or not tiers: + raise ValueError( + "target_references.json: target_profile.criticality_tiers must be " + "a non-empty object." + ) + normalized_tiers: dict[str, tuple[str, float | None]] = {} + for raw_tier_id, raw_tier in tiers.items(): + tier_id = _require_non_empty_string( + raw_tier_id, + field_name="criticality tier id", + context="target_references.json", + ) + if not isinstance(raw_tier, Mapping): + raise ValueError( + f"target_references.json: criticality tier {tier_id!r} must be " + "an object." + ) + unknown_tier_keys = sorted( + set(raw_tier) - {"criticality", "relative_tolerance", "description"} + ) + if unknown_tier_keys: + raise ValueError( + f"target_references.json: criticality tier {tier_id!r} has " + f"unknown key(s) {unknown_tier_keys}." + ) + criticality = _require_non_empty_string( + raw_tier.get("criticality"), + field_name="criticality", + context=f"criticality tier {tier_id!r}", + ) + if criticality not in ALLOWED_GATE_CRITICALITIES: + raise ValueError( + f"target_references.json: criticality tier {tier_id!r} uses " + f"unknown criticality {criticality!r}." + ) + raw_tolerance = raw_tier.get("relative_tolerance") + if raw_tolerance is None: + tolerance = None + elif ( + isinstance(raw_tolerance, bool) + or not isinstance(raw_tolerance, (int, float)) + or not math.isfinite(float(raw_tolerance)) + or not 0.0 < float(raw_tolerance) <= 1.0 + ): + raise ValueError( + f"target_references.json: criticality tier {tier_id!r} " + "relative_tolerance must be null or a finite number in (0, 1]." + ) + else: + tolerance = float(raw_tolerance) + normalized_tiers[tier_id] = (criticality, tolerance) + + basis_periods = profile.get("basis_periods") + if not isinstance(basis_periods, Mapping) or not basis_periods: + raise ValueError( + "target_references.json: target_profile.basis_periods must be a " + "non-empty object." + ) + normalized_periods: dict[str, tuple[object, str]] = {} + for raw_basis_id, raw_basis in basis_periods.items(): + basis_id = _require_non_empty_string( + raw_basis_id, + field_name="basis period id", + context="target_references.json", + ) + if not isinstance(raw_basis, Mapping): + raise ValueError( + f"target_references.json: basis period {basis_id!r} must be an object." + ) + unknown_basis_keys = sorted( + set(raw_basis) + - { + "period", + "basis", + "fact_period_type", + "mismatch_policy", + "survey_year", + "income_reference_offset_years", + "description", + } + ) + if unknown_basis_keys: + raise ValueError( + f"target_references.json: basis period {basis_id!r} has unknown " + f"key(s) {unknown_basis_keys}." + ) + period = raw_basis.get("period") + if ( + period is None + or isinstance(period, bool) + or not isinstance(period, (int, str)) + ): + raise ValueError( + f"target_references.json: basis period {basis_id!r} must declare " + "period as an integer or non-empty string." + ) + if isinstance(period, str) and not period.strip(): + raise ValueError( + f"target_references.json: basis period {basis_id!r} must declare " + "period as an integer or non-empty string." + ) + _require_non_empty_string( + raw_basis.get("basis"), + field_name="basis", + context=f"basis period {basis_id!r}", + ) + fact_period_type = _require_non_empty_string( + raw_basis.get("fact_period_type"), + field_name="fact_period_type", + context=f"basis period {basis_id!r}", + ) + if raw_basis.get("mismatch_policy") != "requires_source_projection": + raise ValueError( + f"target_references.json: basis period {basis_id!r} must set " + "mismatch_policy='requires_source_projection'." + ) + survey_year = raw_basis.get("survey_year") + income_offset = raw_basis.get("income_reference_offset_years") + if survey_year is not None or income_offset is not None: + if ( + isinstance(survey_year, bool) + or not isinstance(survey_year, int) + or isinstance(income_offset, bool) + or not isinstance(income_offset, int) + ): + raise ValueError( + f"target_references.json: basis period {basis_id!r} must " + "declare integer survey_year and " + "income_reference_offset_years together." + ) + try: + numeric_period = int(period) + except ValueError as error: + raise ValueError( + f"target_references.json: basis period {basis_id!r} with " + "an income-reference offset must use a numeric period." + ) from error + if numeric_period != survey_year + income_offset: + raise ValueError( + f"target_references.json: basis period {basis_id!r} period " + f"{period!r} does not equal survey_year {survey_year!r} plus " + f"income_reference_offset_years {income_offset!r}." + ) + normalized_periods[basis_id] = (period, fact_period_type) + + names = [reference.name for reference in references] + duplicates = sorted({name for name in names if names.count(name) > 1}) + if duplicates: + raise ValueError( + f"target_references.json: duplicate reference name(s) {duplicates}." + ) + + for reference in references: + context = f"target reference {reference.name!r}" + role = reference.metadata.get("target_role") + if role not in {"calibration", "validation"}: + raise ValueError( + f"target_references.json: {context} metadata.target_role must be " + "'calibration' or 'validation'." + ) + tier_id = reference.metadata.get("criticality_tier", "") + if tier_id not in normalized_tiers: + raise ValueError( + f"target_references.json: {context} names unknown " + f"criticality_tier {tier_id!r}." + ) + tier_criticality, tier_tolerance = normalized_tiers[tier_id] + if reference.metadata.get("criticality") != tier_criticality: + raise ValueError( + f"target_references.json: {context} criticality does not match " + f"tier {tier_id!r}." + ) + if role == "calibration" and tier_tolerance is None: + raise ValueError( + f"target_references.json: calibration {context} uses tier " + f"{tier_id!r} without a relative tolerance." + ) + if role == "validation" and ( + tier_criticality != "diagnostic" or tier_tolerance is not None + ): + raise ValueError( + f"target_references.json: validation {context} must use a " + "diagnostic tier with no calibration tolerance." + ) + + basis_id = reference.metadata.get("basis_period", "") + if basis_id not in normalized_periods: + raise ValueError( + f"target_references.json: {context} names unknown basis_period " + f"{basis_id!r}." + ) + basis_period, fact_period_type = normalized_periods[basis_id] + if str(reference.period) != str(basis_period): + raise ValueError( + f"target_references.json: {context} period {reference.period!r} " + f"does not match basis period {basis_id!r}." + ) + selector_period_type = str(reference.ledger_selector.get("period_type", "")) + if selector_period_type != fact_period_type: + raise ValueError( + f"target_references.json: {context} selector period_type " + f"{selector_period_type!r} does not match basis period " + f"{basis_id!r} fact_period_type {fact_period_type!r}." + ) + if reference.period_match_policy != "exact": + raise ValueError( + f"target_references.json: {context} must set " + "period_match_policy='exact'; stale observations require an " + "explicit Chronicle source_projection." + ) + if reference.assertion_policy != "allow_source_projection": + raise ValueError( + f"target_references.json: {context} must set " + "assertion_policy='allow_source_projection' so an explicitly " + "projected fact can satisfy the declared basis period." + ) + + geography_level = str(reference.ledger_selector.get("geography_level", "")) + if geography_level and geography_level not in {"country", "national"}: + selector_vintage = str( + reference.ledger_selector.get("geography_vintage", "") + ) + metadata_vintage = reference.metadata.get("geography_vintage", "") + if not selector_vintage or selector_vintage != metadata_vintage: + raise ValueError( + f"target_references.json: subnational {context} must bind the " + "same non-empty geography_vintage in its selector and metadata." + ) + if geography_spine is not None and ( + geography_level == geography_spine.geography_spine.geography_level + ): + spine = geography_spine.geography_spine + short_code_system = spine.code_system.removeprefix(f"{country}_") + accepted_spine_vintages = { + spine.vintage, + f"{short_code_system}_{spine.vintage}", + } + if selector_vintage not in accepted_spine_vintages: + raise ValueError( + f"target_references.json: {context} geography vintage " + f"{selector_vintage!r} differs from the " + f"{geography_level!r} spine vintage " + f"{spine.vintage!r} under code system " + f"{spine.code_system!r}." + ) + legacy_nis_vintage = reference.metadata.get("nis_vintage") + if legacy_nis_vintage is not None and legacy_nis_vintage not in { + spine.vintage, + selector_vintage, + }: + raise ValueError( + f"target_references.json: {context} legacy " + f"nis_vintage {legacy_nis_vintage!r} does not identify " + f"the declared spine vintage {spine.vintage!r}." + ) + + calibrated_families = { + reference.family + for reference in references + if reference.metadata.get("target_role") == "calibration" + } + missing_families = sorted(set(required_families) - calibrated_families) + if missing_families: + raise ValueError( + "target_references.json: target_profile.required_families has no " + f"calibration reference for {missing_families}." + ) + return _freeze_gate_parameter(profile, path="target_profile") + + def _validate_local_target_references( raw: Mapping[str, Any], *, @@ -1023,6 +1356,8 @@ class ResolvedCountrySpec: support_spine: The support-spine manifest, when declared. geography_spine: The geography-spine manifest, when declared. target_references: Ledger target references, when declared. + target_profile: The validated value-free target selection policy carried + by ``target_references.json``. gates: The gate selection, when declared. release_contract: The release contract, when declared. take_up_contract: The constants-era take-up compatibility view. For a @@ -1043,6 +1378,7 @@ class ResolvedCountrySpec: support_spine: SupportSpineManifest | None geography_spine: GeographySpineManifest | None target_references: tuple[LedgerTargetReference, ...] + target_profile: Mapping[str, Any] local_target_references: tuple[LedgerTargetReference, ...] gates: GatesManifest | None release_contract: ReleaseContractManifest | None @@ -1662,6 +1998,16 @@ def load_country_spec(country: str | Path) -> ResolvedCountrySpec: if "target_references.json" in payloads else () ) + target_profile = ( + _validate_target_profile( + payloads["target_references.json"], + target_references, + country=declared_country, + geography_spine=geography_spine, + ) + if "target_references.json" in payloads + else MappingProxyType({}) + ) local_target_references = ( _validate_local_target_references( payloads["local_target_references.json"], @@ -1697,6 +2043,7 @@ def load_country_spec(country: str | Path) -> ResolvedCountrySpec: support_spine=support_spine, geography_spine=geography_spine, target_references=target_references, + target_profile=target_profile, local_target_references=local_target_references, gates=gates, release_contract=release_contract, diff --git a/packages/microcosm-build/src/microcosm/build/ledger_targets.py b/packages/microcosm-build/src/microcosm/build/ledger_targets.py index f191cfd0d..ddaa5a6a4 100644 --- a/packages/microcosm-build/src/microcosm/build/ledger_targets.py +++ b/packages/microcosm-build/src/microcosm/build/ledger_targets.py @@ -21,6 +21,7 @@ SUPPORTED_LEDGER_AGGREGATIONS = frozenset(("sum",)) ALLOWED_ASSERTION_POLICIES = frozenset(("observed_only", "allow_source_projection")) +ALLOWED_PERIOD_MATCH_POLICIES = frozenset(("latest_not_after", "exact")) ALLOWED_VALUE_OPERATIONS = frozenset( ("identity", "sum", "calendar_year_average", "latest_plateau", "count_x_mean") ) @@ -81,6 +82,7 @@ class LedgerTargetReference: notes: str = "" metadata: Mapping[str, str] = field(default_factory=dict) assertion_policy: str = "observed_only" + period_match_policy: str = "latest_not_after" uprating_index: str | None = None uprating_from_period: int | str | None = None uprating_to_period: int | str | None = None @@ -113,6 +115,17 @@ def __post_init__(self) -> None: f"LedgerTargetReference {self.name!r}: unsupported " f"assertion_policy {self.assertion_policy!r}." ) + if self.period_match_policy not in ALLOWED_PERIOD_MATCH_POLICIES: + raise ValueError( + f"LedgerTargetReference {self.name!r}: unsupported " + f"period_match_policy {self.period_match_policy!r}; expected " + f"one of {sorted(ALLOWED_PERIOD_MATCH_POLICIES)!r}." + ) + if self.period_match_policy == "exact" and self.period is None: + raise ValueError( + f"LedgerTargetReference {self.name!r}: period_match_policy=" + "'exact' requires an explicit target period." + ) if not self.entity: raise ValueError( f"LedgerTargetReference {self.name!r}: entity must be non-empty." @@ -403,9 +416,10 @@ def apply_ledger_target_profile( if not profile: return registry schema_version = profile.get("schema_version", 1) - if schema_version != 1: + if schema_version not in {1, 2}: raise ValueError( - f"Ledger target profile schema_version must be 1, got {schema_version!r}." + "Ledger target profile schema_version must be 1 or 2, got " + f"{schema_version!r}." ) result = registry for raw_rule in profile.get("hierarchy_reconciliations") or (): @@ -524,6 +538,7 @@ def target_spec_from_ledger_reference( ) numeric_values = [] for member in facts: + _validate_reference_period(member, reference) numeric_values.append(_numeric_fact_value(member, reference)) _validate_fact_aggregation(member, reference) @@ -912,9 +927,14 @@ def _resolve_reference_fact( f"Ledger fact selector: {dict(reference.ledger_selector)!r}." ) if not eligible_matches: + period_requirement = ( + "at exact target period" + if reference.period_match_policy == "exact" + else "at or before target period" + ) raise ValueError( f"Ledger target reference {reference.name!r} did not match a " - "Ledger fact at or before target period " + f"Ledger fact {period_requirement} " f"{reference.period!r} for selector: " f"{dict(reference.ledger_selector)!r}." ) @@ -1169,6 +1189,13 @@ def _eligible_selector_matches( matches: list[object], ) -> list[object]: target_period_key = _period_key_from_value(reference.period) + if reference.period_match_policy == "exact": + return [ + fact + for fact in matches + if _period_key(fact) == target_period_key + and _assertion_allowed(reference, fact) + ] return [ fact for fact in matches @@ -1184,6 +1211,30 @@ def _assertion_allowed(reference: LedgerTargetReference, fact: object) -> bool: return True +def _validate_reference_period(fact: object, reference: LedgerTargetReference) -> None: + """Enforce an explicit exact-period consumer contract after resolution. + + An exact-period reference may consume either an observation at that period + or a Ledger ``source_projection`` whose published fact period is that same + target period. It may not silently substitute an older observation. The + projection's source period belongs in Ledger lineage; the consumer still + binds the projected fact to the period it estimates. + """ + + if reference.period_match_policy != "exact": + return + expected = _period_key_from_value(reference.period) + actual = _period_key(fact) + if actual != expected: + raise ValueError( + f"Ledger target reference {reference.name!r} requires exact period " + f"{reference.period!r}, but resolved fact period " + f"{_at(fact, 'period', 'value')!r}. A period mismatch must resolve " + "through a Ledger source_projection at the target period, never " + "through a silently stale observation." + ) + + def _fact_assertion(fact: object) -> str: return _str_at(fact, "assertion") or "observation" @@ -1374,6 +1425,7 @@ def _reference_metadata(reference: LedgerTargetReference) -> dict[str, str]: metadata = dict(reference.metadata) metadata["ledger_value_operation"] = reference.value_operation metadata["ledger_assertion_policy"] = reference.assertion_policy + metadata["ledger_period_match_policy"] = reference.period_match_policy for key, value in sorted(reference.ledger_selector.items()): if isinstance(value, Mapping): continue @@ -1505,6 +1557,8 @@ def _selector_candidates(fact: object, key: str) -> tuple[str, ...]: return (_str_at(fact, "geography", "level"),) if key == "geography_id": return (_str_at(fact, "geography", "id"),) + if key == "geography_vintage": + return (_str_at(fact, "geography", "vintage"),) if key == "entity_name": return (_str_at(fact, "entity", "name"),) if key in {"record_set_id", "layout_record_set_id"}: diff --git a/packages/microcosm-build/tests/golden/be_country_spec.json b/packages/microcosm-build/tests/golden/be_country_spec.json index b12ed4223..859250080 100644 --- a/packages/microcosm-build/tests/golden/be_country_spec.json +++ b/packages/microcosm-build/tests/golden/be_country_spec.json @@ -1,6 +1,6 @@ { "country": "be", - "fingerprint": "16c586e4599c8f1aeef05f1d91a852d0ca284f066266616cee2aaf2e9e0e0229", + "fingerprint": "3ebf6c9b64795c4e6ecb10d6b031721e0de14a843358793116dded05314df9a4", "gate_ids": [ "calibration_per_family_fit", "national_and_nuts1_admin_aggregates", @@ -39,7 +39,7 @@ "spec/sources.yaml": "1ca516ad27ee815cb60f451901dcdc6b8989c4f1731d0f6f3f1e50714e29e35e", "spec/spine.yaml": "908d99488c3905806ec6878d96b0e4a4cbb67c6ec69d53fe5f1d160246fdb513", "spec/vintages.yaml": "c842490287404cf306907800b0bcadec5d954e7e9c510f4b293786c5bedd0910", - "target_references.json": "0f49fe2cdb024607e00558dfb87655dbaad06219ba05dd5b72905efc1e2916af" + "target_references.json": "09ee9cd8618751c02e2da5708178f8c6a7fa6afbe6f524d632daae3483407eed" }, "resources": [ "spec/bundle.yaml", diff --git a/packages/microcosm-build/tests/test_country_spec.py b/packages/microcosm-build/tests/test_country_spec.py index 26a8cb76c..19bb5cfb8 100644 --- a/packages/microcosm-build/tests/test_country_spec.py +++ b/packages/microcosm-build/tests/test_country_spec.py @@ -171,6 +171,65 @@ def _minimal_package(**overrides) -> dict[str, dict]: return files +def _schema2_target_resource() -> dict[str, object]: + return { + "country": "xx", + "allowed_value_operations": ["identity"], + "target_references": [ + { + "name": "population_anchor", + "ledger_selector": { + "source_name": "official_population", + "source_measure_id": "people", + "period_type": "calendar_year", + "geography_level": "country", + }, + "entity": "person", + "measure": "people", + "period": 2023, + "family": "demography", + "assertion_policy": "allow_source_projection", + "period_match_policy": "exact", + "metadata": { + "basis_period": "population_2023", + "criticality": "release_blocking", + "criticality_tier": "demography_release", + "publisher": "Official statistics office", + "target_role": "calibration", + }, + } + ], + "target_profile": { + "schema_version": 2, + "required_families": ["demography"], + "criticality_tiers": { + "demography_release": { + "criticality": "release_blocking", + "relative_tolerance": 0.02, + "description": "Population cells.", + } + }, + "basis_periods": { + "population_2023": { + "period": 2023, + "basis": "reference_date", + "fact_period_type": "calendar_year", + "mismatch_policy": "requires_source_projection", + "description": "Population reference year.", + } + }, + "hierarchy_reconciliations": [], + }, + } + + +def _package_with_schema2_targets() -> dict[str, dict]: + files = _minimal_package() + files["country_package.json"]["resources"].append("target_references.json") + files["target_references.json"] = _schema2_target_resource() + return files + + class TestArmenianPackage: @pytest.fixture(scope="class") def spec(self): @@ -461,8 +520,125 @@ def test_targets_arrive_by_reference_with_no_values(self, spec) -> None: by_name = {reference.name: reference for reference in spec.target_references} commune = by_name["statbel_fiscal_income_by_commune"] assert commune.metadata["nis_vintage"] == "2025" + assert commune.metadata["geography_vintage"] == "nis_2025" + assert commune.ledger_selector["geography_vintage"] == "nis_2025" assert commune.metadata["criticality"] == "diagnostic" + payload = json.loads( + (COUNTRY_PACKAGE_ROOT / "be/target_references.json").read_text( + encoding="utf-8" + ) + ) + assert FORBIDDEN_TARGET_VALUE_KEYS.isdisjoint( + _nested_mapping_keys(payload["target_references"]) + ) + + def test_target_selectors_match_the_chronicle_fact_vocabulary(self, spec) -> None: + references = {reference.name: reference for reference in spec.target_references} + expected = { + "statbel_population_by_age_sex_region": ( + "statbel_population_structure", + "people", + "calendar_year", + 2023, + "nuts1", + "NUTS_2024", + ), + "statbel_fiscal_income_by_commune": ( + "statbel_fiscal_income", + "taxable_income", + "tax_year", + 2022, + "commune", + "nis_2025", + ), + "spf_finances_pit_total": ( + "spf_finances_pit", + "tax_before_withholding", + "tax_year", + 2022, + "country", + None, + ), + "onss_employee_contribution_total": ( + "onss_contributions", + "worker_article_17_uncapped_component_contribution", + "calendar_year", + 2022, + "country", + None, + ), + "onem_unemployment_caseload": ( + "onem_rva_unemployment", + "receives_unemployment_benefit", + "calendar_year", + 2022, + "country", + None, + ), + "nbb_household_disposable_income": ( + "nbb_national_accounts", + "household_disposable_income", + "calendar_year", + 2022, + "country", + None, + ), + } + + for name, ( + source_name, + source_measure_id, + period_type, + period, + geography_level, + geography_vintage, + ) in expected.items(): + reference = references[name] + assert reference.ledger_selector["source_name"] == source_name + assert reference.ledger_selector["source_measure_id"] == source_measure_id + assert reference.ledger_selector["period_type"] == period_type + assert reference.period == period + assert reference.period_match_policy == "exact" + assert reference.assertion_policy == "allow_source_projection" + assert reference.ledger_selector["geography_level"] == geography_level + assert ( + reference.ledger_selector.get("geography_vintage") == geography_vintage + ) + + def test_target_profile_declares_tiers_and_income_basis(self, spec) -> None: + profile = spec.target_profile + assert profile["schema_version"] == 2 + assert tuple(profile["required_families"]) == ( + "demography", + "fiscal_income", + "income_tax", + "social_security", + "caseloads", + ) + tiers = profile["criticality_tiers"] + assert tiers["core_fiscal_release"]["relative_tolerance"] == 0.05 + assert tiers["caseload_release"]["relative_tolerance"] == 0.15 + assert tiers["validation_only"]["relative_tolerance"] is None + + income_basis = profile["basis_periods"]["assessment_income_year_2022"] + assert income_basis["period"] == 2022 + assert income_basis["fact_period_type"] == "tax_year" + assert income_basis["survey_year"] == 2023 + assert income_basis["income_reference_offset_years"] == -1 + assert income_basis["mismatch_policy"] == "requires_source_projection" + + references = {reference.name: reference for reference in spec.target_references} + assert ( + references["nbb_household_disposable_income"].metadata["target_role"] + == "validation" + ) + assert { + reference.family + for reference in references.values() + if reference.metadata["target_role"] == "calibration" + } >= set(profile["required_families"]) + def test_gates_select_no_incumbent_comparison(self, spec) -> None: selected = {gate.gate for gate in spec.gates.gates} assert "parity" not in selected # no incumbent: oracles replace it @@ -1307,6 +1483,126 @@ def test_target_reference_carrying_a_nested_value_is_refused( with pytest.raises(ValueError, match="values live in Ledger"): load_country_spec(package_dir) + def test_schema2_target_profile_loads_as_value_free_policy(self, tmp_path) -> None: + files = _package_with_schema2_targets() + package_dir = _write_package(tmp_path, files) + + spec = load_country_spec(package_dir) + + assert spec.target_profile["schema_version"] == 2 + assert ( + spec.target_profile["criticality_tiers"]["demography_release"][ + "relative_tolerance" + ] + == 0.02 + ) + assert spec.target_references[0].period_match_policy == "exact" + + def test_schema2_target_profile_refuses_invalid_tolerance(self, tmp_path) -> None: + files = _package_with_schema2_targets() + files["target_references.json"]["target_profile"]["criticality_tiers"][ + "demography_release" + ]["relative_tolerance"] = 0.0 + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="finite number in \\(0, 1\\]"): + load_country_spec(package_dir) + + def test_schema2_target_profile_refuses_unknown_reference_tier( + self, tmp_path + ) -> None: + files = _package_with_schema2_targets() + files["target_references.json"]["target_references"][0]["metadata"][ + "criticality_tier" + ] = "undeclared" + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="unknown criticality_tier"): + load_country_spec(package_dir) + + def test_schema2_target_profile_refuses_income_offset_drift(self, tmp_path) -> None: + files = _package_with_schema2_targets() + basis = files["target_references.json"]["target_profile"]["basis_periods"][ + "population_2023" + ] + basis["survey_year"] = 2023 + basis["income_reference_offset_years"] = -1 + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="does not equal survey_year"): + load_country_spec(package_dir) + + def test_schema2_target_profile_refuses_reference_period_drift( + self, tmp_path + ) -> None: + files = _package_with_schema2_targets() + files["target_references.json"]["target_references"][0]["period"] = 2022 + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="does not match basis period"): + load_country_spec(package_dir) + + def test_schema2_target_profile_refuses_fact_period_type_drift( + self, tmp_path + ) -> None: + files = _package_with_schema2_targets() + files["target_references.json"]["target_references"][0]["ledger_selector"][ + "period_type" + ] = "tax_year" + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="fact_period_type"): + load_country_spec(package_dir) + + def test_schema2_target_profile_requires_explicit_projection_policy( + self, tmp_path + ) -> None: + files = _package_with_schema2_targets() + files["target_references.json"]["target_references"][0]["assertion_policy"] = ( + "observed_only" + ) + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="allow_source_projection"): + load_country_spec(package_dir) + + def test_schema2_target_profile_requires_subnational_vintage_binding( + self, tmp_path + ) -> None: + files = _package_with_schema2_targets() + reference = files["target_references.json"]["target_references"][0] + reference["ledger_selector"]["geography_level"] = "nuts1" + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="same non-empty geography_vintage"): + load_country_spec(package_dir) + + def test_schema2_target_profile_refuses_subnational_vintage_drift( + self, tmp_path + ) -> None: + files = _package_with_schema2_targets() + reference = files["target_references.json"]["target_references"][0] + reference["ledger_selector"].update( + {"geography_level": "nuts1", "geography_vintage": "NUTS_2024"} + ) + reference["metadata"]["geography_vintage"] = "NUTS_2021" + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="same non-empty geography_vintage"): + load_country_spec(package_dir) + + def test_schema2_target_profile_requires_every_calibration_family( + self, tmp_path + ) -> None: + files = _package_with_schema2_targets() + files["target_references.json"]["target_profile"]["required_families"].append( + "income_tax" + ) + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="no calibration reference"): + load_country_spec(package_dir) + def test_sum_target_reference_roundtrips(self, tmp_path) -> None: files = _minimal_package() files["country_package.json"]["resources"].append("target_references.json") diff --git a/packages/microcosm-build/tests/test_ledger_targets.py b/packages/microcosm-build/tests/test_ledger_targets.py index 7ca08363a..46993b981 100644 --- a/packages/microcosm-build/tests/test_ledger_targets.py +++ b/packages/microcosm-build/tests/test_ledger_targets.py @@ -144,6 +144,28 @@ def _consumer_fact_row_for_period(source_period: int, *, value: float): ) +def _exact_agi_reference(**overrides) -> LedgerTargetReference: + values = { + "name": "exact SOI AGI total", + "ledger_selector": { + "source_name": "irs_soi", + "source_measure_id": "adjusted_gross_income", + "period_type": "tax_year", + "geography_level": "country", + "geography_id": "0100000US", + "entity_name": "tax_unit", + "layout_groupby_value_id": "all", + }, + "entity": "tax_unit", + "measure": "adjusted_gross_income", + "period": 2023, + "family": "irs_soi", + "period_match_policy": "exact", + } + values.update(overrides) + return LedgerTargetReference(**values) + + def _monthly_consumer_fact_row(source_period: str, *, value: float): normalized_period = source_period.replace("-", "_") return _consumer_fact_row( @@ -643,6 +665,87 @@ def test__given_selector_matches_multiple_years__then_latest_source_period_is_us ) +def test__given_exact_period_policy__then_only_the_target_period_is_used() -> None: + registry = compile_ledger_target_references( + [ + _consumer_fact_row_for_period(2022, value=14_000_000_000_000), + _consumer_fact_row_for_period(2023, value=15_000_000_000_000), + ], + [_exact_agi_reference()], + country="us", + ) + + (spec,) = registry.specs + assert spec.value == 15_000_000_000_000 + assert spec.period == 2023 + assert spec.metadata["ledger_period_match_policy"] == "exact" + + +def test__given_exact_period_policy__then_stale_observation_is_refused() -> None: + with pytest.raises(ValueError, match="exact target period 2023"): + compile_ledger_target_references( + [_consumer_fact_row_for_period(2022, value=14_000_000_000_000)], + [_exact_agi_reference()], + country="us", + ) + + +def test__given_exact_period_projection__then_explicit_projection_is_used() -> None: + projection = _consumer_fact_row_for_period(2023, value=15_000_000_000_000) + projection["assertion"] = "source_projection" + registry = compile_ledger_target_references( + [projection], + [ + _exact_agi_reference( + assertion_policy="allow_source_projection", + ) + ], + country="us", + ) + + (spec,) = registry.specs + assert spec.value == 15_000_000_000_000 + assert spec.metadata["ledger_resolved_assertion"] == "source_projection" + assert spec.metadata["ledger_assertion_policy"] == "allow_source_projection" + assert spec.metadata["ledger_period_match_policy"] == "exact" + + +def test__given_geography_vintage_selector__then_only_that_vintage_matches() -> None: + old_vintage = _consumer_fact_row( + aggregate_fact_key="ledger.aggregate_fact.v2:old-vintage", + legacy_fact_key="ledger.fact.v1:old-vintage", + value=14_000_000_000_000, + geography={ + "level": "country", + "id": "0100000US", + "name": "United States", + "vintage": "2010_census", + }, + ) + current_vintage = _consumer_fact_row(value=15_000_000_000_000) + reference = _exact_agi_reference( + ledger_selector={ + **dict(_exact_agi_reference().ledger_selector), + "geography_vintage": "2020_census", + } + ) + + registry = compile_ledger_target_references( + [old_vintage, current_vintage], + [reference], + country="us", + ) + + (spec,) = registry.specs + assert spec.value == 15_000_000_000_000 + assert spec.metadata["ledger_selector_geography_vintage"] == "2020_census" + + +def test__given_exact_period_policy_without_period__then_reference_is_refused() -> None: + with pytest.raises(ValueError, match="requires an explicit target period"): + _exact_agi_reference(period=None) + + def test__given_period_bearing_groupby_value__then_latest_source_period_is_used() -> ( None ): @@ -1969,6 +2072,27 @@ def test__given_disabled_hierarchy_profile__then_children_are_unchanged() -> Non assert "hierarchy_reconciliation_rule" not in reconciled.specs[0].metadata +def test__given_schema2_profile_without_hierarchy__then_registry_is_unchanged() -> None: + registry = TargetRegistry( + [ + _hierarchy_target( + "be_population", + value=11_500_000.0, + geography_level="country", + geography_id="BE", + ) + ], + country="be", + ) + + result = apply_ledger_target_profile( + registry, + {"schema_version": 2, "hierarchy_reconciliations": []}, + ) + + assert result is registry + + def test__given_nonzero_parent_and_zero_children__then_hierarchy_fails() -> None: # Given registry = TargetRegistry( From 79848fe28bee6575f9f4ff19c3c605616354da7c Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 29 Aug 2026 18:39:52 -0400 Subject: [PATCH 2/4] Harden target profile period contracts --- .../src/microcosm/build/country_spec.py | 5 + .../src/microcosm/build/ledger_targets.py | 102 ++++++++++++++++-- .../tests/test_country_spec.py | 13 +++ .../tests/test_ledger_targets.py | 71 ++++++++++++ 4 files changed, 184 insertions(+), 7 deletions(-) diff --git a/packages/microcosm-build/src/microcosm/build/country_spec.py b/packages/microcosm-build/src/microcosm/build/country_spec.py index 0d8af4d50..5890bbee0 100644 --- a/packages/microcosm-build/src/microcosm/build/country_spec.py +++ b/packages/microcosm-build/src/microcosm/build/country_spec.py @@ -840,6 +840,11 @@ def _validate_target_profile( if not isinstance(profile, Mapping): raise ValueError("target_references.json: target_profile must be an object.") schema_version = profile.get("schema_version", 1) + if isinstance(schema_version, bool) or not isinstance(schema_version, int): + raise ValueError( + "target_references.json: target_profile.schema_version must be an " + f"integer 1 or 2, got {schema_version!r}." + ) if schema_version == 1: return _freeze_gate_parameter(profile, path="target_profile") if schema_version != 2: diff --git a/packages/microcosm-build/src/microcosm/build/ledger_targets.py b/packages/microcosm-build/src/microcosm/build/ledger_targets.py index ddaa5a6a4..1e35cbd93 100644 --- a/packages/microcosm-build/src/microcosm/build/ledger_targets.py +++ b/packages/microcosm-build/src/microcosm/build/ledger_targets.py @@ -416,6 +416,11 @@ def apply_ledger_target_profile( if not profile: return registry schema_version = profile.get("schema_version", 1) + if isinstance(schema_version, bool) or not isinstance(schema_version, int): + raise ValueError( + "Ledger target profile schema_version must be an integer 1 or 2, " + f"got {schema_version!r}." + ) if schema_version not in {1, 2}: raise ValueError( "Ledger target profile schema_version must be 1 or 2, got " @@ -918,7 +923,11 @@ def _resolve_reference_fact( return _resolve_count_x_mean_reference_facts(reference, eligible_matches) if len(eligible_matches) == 1: return eligible_matches[0] - latest_match = _latest_period_selector_match(reference, eligible_matches) + latest_match = ( + None + if reference.period_match_policy == "exact" + else _latest_period_selector_match(reference, eligible_matches) + ) if latest_match is not None: return latest_match if not matches: @@ -959,7 +968,10 @@ def _resolve_sum_reference_facts( ) -> tuple[object, ...]: partitions: dict[tuple[tuple[str, ...], tuple[int, int, str]], list[object]] = {} for fact in eligible_matches: - key = (_selector_sum_partition_key(fact), _period_key(fact)) + key = ( + _selector_sum_partition_key(fact), + _reference_period_partition_key(fact, reference), + ) partitions.setdefault(key, []).append(fact) if not partitions: raise ValueError( @@ -1038,7 +1050,10 @@ def _resolve_count_x_mean_reference_facts( role = _count_mean_fact_role(fact) if not role: continue - key = (_selector_count_mean_partition_key(fact), _period_key(fact)) + key = ( + _selector_count_mean_partition_key(fact), + _reference_period_partition_key(fact, reference), + ) partitions.setdefault(key, {}).setdefault(role, []).append(fact) valid = { @@ -1193,7 +1208,7 @@ def _eligible_selector_matches( return [ fact for fact in matches - if _period_key(fact) == target_period_key + if _exact_period_matches(fact, reference) and _assertion_allowed(reference, fact) ] return [ @@ -1223,12 +1238,11 @@ def _validate_reference_period(fact: object, reference: LedgerTargetReference) - if reference.period_match_policy != "exact": return - expected = _period_key_from_value(reference.period) - actual = _period_key(fact) - if actual != expected: + if not _exact_period_matches(fact, reference): raise ValueError( f"Ledger target reference {reference.name!r} requires exact period " f"{reference.period!r}, but resolved fact period " + f"{_at(fact, 'period', 'type')!r}:" f"{_at(fact, 'period', 'value')!r}. A period mismatch must resolve " "through a Ledger source_projection at the target period, never " "through a silently stale observation." @@ -1369,6 +1383,80 @@ def _period_key(fact: object) -> tuple[int, int, str]: return _period_key_from_value(_at(fact, "period", "value")) +def _exact_period_matches(fact: object, reference: LedgerTargetReference) -> bool: + """Match exact periods by semantic value while retaining period-kind pins.""" + + expected_value = reference.period + actual_value = _at(fact, "period", "value") + if not _period_values_semantically_equal(expected_value, actual_value): + return False + + actual_type = _str_at(fact, "period", "type") + selector_type = str(reference.ledger_selector.get("period_type", "")) + expected_type_hint = _period_type_hint(expected_value) + if selector_type and expected_type_hint and selector_type != expected_type_hint: + return False + expected_type = selector_type or expected_type_hint + if expected_type and actual_type != expected_type: + return False + actual_type_hint = _period_type_hint(actual_value) + return not actual_type_hint or actual_type_hint == actual_type + + +def _reference_period_partition_key( + fact: object, + reference: LedgerTargetReference, +) -> tuple[int, int, str]: + period_key = _period_key(fact) + if reference.period_match_policy == "exact" and period_key[0]: + return (period_key[0], period_key[1], "") + return period_key + + +def _period_values_semantically_equal(left: object, right: object) -> bool: + """Treat supported typed and scalar spellings of one period as equal.""" + + left_key = _period_key_from_value(left) + right_key = _period_key_from_value(right) + if left_key[0] and right_key[0]: + return left_key[:2] == right_key[:2] + return left_key == right_key + + +def _period_type_hint(value: object) -> str: + if not isinstance(value, str): + return "" + normalized = value.lower().replace("-", "_") + long_prefixes = ( + ("tax_year_", "tax_year"), + ("calendar_year_", "calendar_year"), + ("fiscal_year_", "fiscal_year"), + ("academic_year_", "academic_year"), + ("month", "month"), + ) + long_hint = next( + ( + period_type + for prefix, period_type in long_prefixes + if normalized.startswith(prefix) + ), + "", + ) + if long_hint: + return long_hint + aliases = { + "ty": "tax_year", + "cy": "calendar_year", + "fy": "fiscal_year", + "ay": "academic_year", + } + alias = normalized[:2] + suffix = normalized[2:].lstrip("_") + if alias in aliases and suffix[:1].isdigit(): + return aliases[alias] + return "" + + def _period_key_from_value(value: object) -> tuple[int, int, str]: label = "" if value is None else str(value) normalized = label.lower().replace("-", "_") diff --git a/packages/microcosm-build/tests/test_country_spec.py b/packages/microcosm-build/tests/test_country_spec.py index 19bb5cfb8..d2493ab18 100644 --- a/packages/microcosm-build/tests/test_country_spec.py +++ b/packages/microcosm-build/tests/test_country_spec.py @@ -1498,6 +1498,19 @@ def test_schema2_target_profile_loads_as_value_free_policy(self, tmp_path) -> No ) assert spec.target_references[0].period_match_policy == "exact" + @pytest.mark.parametrize("schema_version", [True, 1.0]) + def test_target_profile_refuses_non_integer_schema_version( + self, tmp_path, schema_version + ) -> None: + files = _package_with_schema2_targets() + files["target_references.json"]["target_profile"]["schema_version"] = ( + schema_version + ) + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="schema_version must be an integer"): + load_country_spec(package_dir) + def test_schema2_target_profile_refuses_invalid_tolerance(self, tmp_path) -> None: files = _package_with_schema2_targets() files["target_references.json"]["target_profile"]["criticality_tiers"][ diff --git a/packages/microcosm-build/tests/test_ledger_targets.py b/packages/microcosm-build/tests/test_ledger_targets.py index 46993b981..07b611bd3 100644 --- a/packages/microcosm-build/tests/test_ledger_targets.py +++ b/packages/microcosm-build/tests/test_ledger_targets.py @@ -681,6 +681,54 @@ def test__given_exact_period_policy__then_only_the_target_period_is_used() -> No assert spec.metadata["ledger_period_match_policy"] == "exact" +@pytest.mark.parametrize( + ("reference_period", "fact_period"), + [ + (2023, "tax_year_2023"), + ("tax_year_2023", 2023), + ], +) +def test__given_equivalent_exact_period_labels__then_target_period_is_used( + reference_period, fact_period +) -> None: + fact = _consumer_fact_row_for_period(2023, value=15_000_000_000_000) + fact["period"]["value"] = fact_period + + registry = compile_ledger_target_references( + [fact], + [_exact_agi_reference(period=reference_period)], + country="us", + ) + + (spec,) = registry.specs + assert spec.value == 15_000_000_000_000 + assert spec.period == reference_period + + +def test__given_equivalent_exact_period_label__then_period_type_still_matches() -> None: + fact = _consumer_fact_row_for_period(2023, value=15_000_000_000_000) + fact["period"] = {"type": "calendar_year", "value": "tax_year_2023"} + + with pytest.raises(ValueError, match="did not match a Ledger fact selector"): + compile_ledger_target_references( + [fact], + [_exact_agi_reference(period=2023)], + country="us", + ) + + +def test__given_exact_identifier__then_declared_period_type_still_matches() -> None: + fact = _consumer_fact_row_for_period(2023, value=15_000_000_000_000) + fact["period"]["type"] = "calendar_year" + reference = _exact_agi_reference( + ledger_fact_key=fact["aggregate_fact_key"], + period="tax_year_2023", + ) + + with pytest.raises(ValueError, match="requires exact period"): + compile_ledger_target_references([fact], [reference], country="us") + + def test__given_exact_period_policy__then_stale_observation_is_refused() -> None: with pytest.raises(ValueError, match="exact target period 2023"): compile_ledger_target_references( @@ -2093,6 +2141,29 @@ def test__given_schema2_profile_without_hierarchy__then_registry_is_unchanged() assert result is registry +@pytest.mark.parametrize("schema_version", [True, 1.0]) +def test__given_non_integer_target_profile_schema__then_profile_is_refused( + schema_version, +) -> None: + registry = TargetRegistry( + [ + _hierarchy_target( + "be_population", + value=11_500_000.0, + geography_level="country", + geography_id="BE", + ) + ], + country="be", + ) + + with pytest.raises(ValueError, match="schema_version must be an integer"): + apply_ledger_target_profile( + registry, + {"schema_version": schema_version, "hierarchy_reconciliations": []}, + ) + + def test__given_nonzero_parent_and_zero_children__then_hierarchy_fails() -> None: # Given registry = TargetRegistry( From 5bff43fb73dc6d34dde9461bcb2d505eef89cbc2 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 29 Aug 2026 19:33:03 -0400 Subject: [PATCH 3/4] Harden Belgian target declarations and shared resolver contracts --- changelog.d/be-calibration-contract.added.md | 2 +- .../src/microcosm/build/be/gates.json | 6 +- .../microcosm/build/be/target_references.json | 12 +- .../src/microcosm/build/country_spec.py | 154 +++++++++++-- .../src/microcosm/build/ledger_targets.py | 217 ++++++++++++------ .../tests/golden/be_country_spec.json | 6 +- .../tests/test_country_spec.py | 215 ++++++++++++++++- .../tests/test_ledger_targets.py | 197 ++++++++++++++++ 8 files changed, 697 insertions(+), 112 deletions(-) diff --git a/changelog.d/be-calibration-contract.added.md b/changelog.d/be-calibration-contract.added.md index 190aa0c44..93d489028 100644 --- a/changelog.d/be-calibration-contract.added.md +++ b/changelog.d/be-calibration-contract.added.md @@ -1 +1 @@ -Add a value-free Belgian calibration target contract with Chronicle selector pins, explicit BE-SILC income-reference periods, criticality-tier tolerances, exact projection handling, and geography-vintage validation. +Add value-free Belgian calibration schema groundwork with Chronicle selector shapes, explicit BE-SILC income-reference periods, declaration-only criticality tiers and target roles, fail-closed placeholder and exact-period handling, and typed geography-vintage validation. Current Chronicle Belgian facts do not yet satisfy the full target surface; runtime tolerance/gate wiring and the #264 external-oracle work remain out of scope. diff --git a/packages/microcosm-build/src/microcosm/build/be/gates.json b/packages/microcosm-build/src/microcosm/build/be/gates.json index 13d3cd63d..024a626d1 100644 --- a/packages/microcosm-build/src/microcosm/build/be/gates.json +++ b/packages/microcosm-build/src/microcosm/build/be/gates.json @@ -1,7 +1,7 @@ { "version": 2, "country": "be", - "policy": "Belgium has no incumbent dataset, so the incumbent-comparison gates (parity, export_surface, target_surface) are deliberately not selected: external oracles replace self-parity (EUROMOD-BE baselines and Federal Planning Bureau reform scores, populace#264). National and NUTS1 targets hard-gate; commune-grain rows are diagnostic until the spine demonstrably supports them.", + "policy": "Declaration-only intended Belgian gate posture. Belgium has no incumbent dataset, so incumbent-comparison gates (parity, export_surface, target_surface) are not selected. National and NUTS1 references are declared release-blocking and commune-grain references diagnostic, but this schema change does not wire target-profile tiers or roles into runtime gates. External-oracle implementation (EUROMOD-BE baselines and Federal Planning Bureau reform scores) remains separate work in #264 and is not implemented here.", "phases": [ "terminal" ], @@ -29,7 +29,7 @@ ], "default_rtol": 0.05 }, - "notes": "Weighted aggregates against admin anchors on the hard surface. Criticality tiers per target arrive with the Ledger profile (PolicyEngine/ledger#70): e.g. PIT total tighter than caseloads." + "notes": "Intended weighted aggregates against admin anchors on the future hard surface. The tier metadata in target_references.json is declaration-only in this change; a later runner integration must explicitly bridge it into gate tolerances." }, { "id": "commune_fiscal_income_fit", @@ -58,7 +58,7 @@ "caseloads" ] }, - "notes": "The active target profile must cover every named family; a build that silently dropped a family must not release." + "notes": "The eventual active target profile must cover every named family; the schema declarations and placeholders in this change do not themselves activate that surface." }, { "id": "demographics_vs_statbel", diff --git a/packages/microcosm-build/src/microcosm/build/be/target_references.json b/packages/microcosm-build/src/microcosm/build/be/target_references.json index f2847f6ef..4517f4f5f 100644 --- a/packages/microcosm-build/src/microcosm/build/be/target_references.json +++ b/packages/microcosm-build/src/microcosm/build/be/target_references.json @@ -1,6 +1,6 @@ { "country": "be", - "description": "Belgian calibration and validation references, by reference only. Values and source projections resolve from Chronicle consumer facts at build time. The consumer contract names every target's criticality tier, relative tolerance, basis period, and subnational geography vintage; it never copies a target value from Chronicle, a survey publication, or a validation oracle.", + "description": "Belgian calibration and validation schema groundwork, by reference only. Values and source projections would resolve from Chronicle consumer facts at build time, but the current Chronicle Belgian catalog does not satisfy this full selector and period surface. Criticality tiers, relative tolerances, and target_role are validated declaration metadata only: this package does not yet wire them into runtime calibration objectives or release gates. Multi-cell series remain non-executable until Chronicle fanout produces cell-pinned references. No external-oracle implementation from #264 is included. The file never copies a target value from Chronicle, a survey publication, or a validation oracle.", "allowed_value_operations": [ "identity" ], @@ -12,7 +12,7 @@ "source_measure_id": "people", "period_type": "calendar_year", "geography_level": "nuts1", - "geography_vintage": "NUTS_2024" + "geography_vintage": "nuts1_2025" }, "entity": "person", "measure": "people", @@ -21,15 +21,16 @@ "assertion_policy": "allow_source_projection", "period_match_policy": "exact", "metadata": { + "activation_status": "requires_harvested_cell_references", "basis_period": "population_reference_2023", "criticality": "release_blocking", "criticality_tier": "demography_release", - "geography_vintage": "NUTS_2024", + "geography_vintage": "nuts1_2025", "publisher": "Statbel", "series": "Structure of the population: age band by sex by region", "target_role": "calibration" }, - "notes": "The demographic anchor is an age-band x sex x NUTS1 count surface. An observation for another year may enter only as a Chronicle source_projection whose fact period is the declared 2023 reference year." + "notes": "Non-executable series placeholder for an age-band x sex x NUTS1 count surface. Chronicle must fan this into dimension- and geography-pinned scalar references before compilation. The selector uses the typed NUTS1-2025 authority alias; a differently-vintaged source requires an explicit Chronicle projection/crosswalk. An observation for another year may enter only as a Chronicle source_projection whose fact period is the declared 2023 reference year." }, { "name": "statbel_fiscal_income_by_commune", @@ -47,6 +48,7 @@ "assertion_policy": "allow_source_projection", "period_match_policy": "exact", "metadata": { + "activation_status": "requires_harvested_cell_references", "basis_period": "assessment_income_year_2022", "criticality": "diagnostic", "criticality_tier": "commune_diagnostic", @@ -56,7 +58,7 @@ "series": "Fiscal statistics of income: total net taxable income by municipality", "target_role": "calibration" }, - "notes": "The open commune series is diagnostic until the synthetic spine demonstrates adequate support. The selector binds Chronicle's nis_2025 spelling to the spine's 2025 NIS code set, and another code set must be projected and crosswalked in Chronicle rather than partially joined." + "notes": "Non-executable series placeholder for commune fiscal-income cells; Chronicle must fan it into one commune-pinned scalar reference per selected cell before compilation. The eventual open commune series is diagnostic until the synthetic spine demonstrates adequate support. The selector binds the typed nis_2025 alias to the spine's 2025 NIS code set, and another code set must be projected and crosswalked in Chronicle rather than partially joined." }, { "name": "spf_finances_pit_total", diff --git a/packages/microcosm-build/src/microcosm/build/country_spec.py b/packages/microcosm-build/src/microcosm/build/country_spec.py index 5890bbee0..060c38960 100644 --- a/packages/microcosm-build/src/microcosm/build/country_spec.py +++ b/packages/microcosm-build/src/microcosm/build/country_spec.py @@ -42,7 +42,11 @@ from types import MappingProxyType from typing import Any -from microcosm.build.ledger_targets import LedgerTargetReference +from microcosm.build.ledger_targets import ( + LedgerTargetReference, + period_type_hint, + period_values_semantically_equal, +) from microcosm.build.plan import DonorSpec, Stage, StagePlan from microcosm.build.source_manifest import ( SourceManifest, @@ -820,20 +824,104 @@ def _validate_target_references( return tuple(references) +def _typed_geography_vintage_aliases( + resolved_spec: object | None, + *, + country: str, +) -> Mapping[str, frozenset[str]]: + """Resolve each typed geography layer to its closed authority aliases. + + The geography domain owns the layer-to-vintage-reference binding, and the + resolved vintage registry owns that reference's content-pinned value. A + consumer selector may use only the full typed id, its country-shortened id, + or that resolved authority value; no year-shaped alias is inferred. + """ + + if resolved_spec is None: + return MappingProxyType({}) + try: + geography_domain = resolved_spec.domain("geography") # type: ignore[attr-defined] + except KeyError: + return MappingProxyType({}) + geography = _require_mapping( + geography_domain.to_wire(), context="typed geography domain" + ) + assignment = _require_mapping( + geography.get("assignment"), context="typed geography assignment" + ) + layer_vintages = _require_mapping( + assignment.get("layer_vintages", {}), + context="typed geography assignment.layer_vintages", + ) + authorities = _require_mapping( + getattr(resolved_spec, "vintage_authorities", {}), + context="resolved vintage authorities", + ) + records = _require_mapping( + authorities.get("records", {}), + context="resolved vintage authorities.records", + ) + aliases_by_layer: dict[str, frozenset[str]] = {} + for raw_layer, raw_reference in layer_vintages.items(): + layer = _require_non_empty_string( + raw_layer, + field_name="geography layer", + context="typed geography assignment.layer_vintages", + ) + reference = _require_non_empty_string( + raw_reference, + field_name=f"{layer} vintage reference", + context="typed geography assignment.layer_vintages", + ) + if not reference.startswith("vintage:"): + raise ValueError( + f"typed geography layer {layer!r} must bind a vintage: reference, " + f"got {reference!r}." + ) + record_id = reference.removeprefix("vintage:") + record = records.get(record_id.casefold()) + if not isinstance(record, Mapping): + raise ValueError( + f"typed geography layer {layer!r} has dangling vintage " + f"reference {reference!r}." + ) + if record.get("kind") != "geography_vintage_ref": + raise ValueError( + f"typed geography layer {layer!r} reference {reference!r} " + "does not resolve to a geography_vintage_ref." + ) + authority_value = record.get("value") + if not isinstance(authority_value, (str, int)) or isinstance( + authority_value, bool + ): + raise ValueError( + f"typed geography layer {layer!r} reference {reference!r} has " + "no string/integer authority value." + ) + aliases = {record_id, str(authority_value)} + country_prefix = f"{country}_" + if record_id.startswith(country_prefix): + aliases.add(record_id.removeprefix(country_prefix)) + aliases_by_layer[layer] = frozenset(aliases) + return MappingProxyType(aliases_by_layer) + + def _validate_target_profile( raw: Mapping[str, Any], references: tuple[LedgerTargetReference, ...], *, country: str, geography_spine: GeographySpineManifest | None, + resolved_spec: object | None, ) -> Mapping[str, Any]: """Validate the versioned consumer-side calibration selection contract. Schema 1 remains the historical hierarchy-only profile. Schema 2 adds the - policy fields needed to make a target surface auditable without carrying - target values: named criticality/tolerance tiers, named basis periods, an - exact-period posture, and explicit geography vintages on every - subnational selector. + declaration fields needed to make a future target surface auditable + without carrying target values: named criticality/tolerance tiers, named + basis periods, an exact-period posture, and explicit geography vintages on + every subnational selector. This validates schema policy only; it does not + apply tier tolerances or ``target_role`` to runtime calibration or gates. """ profile = raw.get("target_profile", {}) @@ -1032,6 +1120,11 @@ def _validate_target_profile( ) normalized_periods[basis_id] = (period, fact_period_type) + geography_vintage_aliases = _typed_geography_vintage_aliases( + resolved_spec, + country=country, + ) + names = [reference.name for reference in references] duplicates = sorted({name for name in names if names.count(name) > 1}) if duplicates: @@ -1079,11 +1172,25 @@ def _validate_target_profile( f"{basis_id!r}." ) basis_period, fact_period_type = normalized_periods[basis_id] - if str(reference.period) != str(basis_period): + if not period_values_semantically_equal(reference.period, basis_period): raise ValueError( f"target_references.json: {context} period {reference.period!r} " f"does not match basis period {basis_id!r}." ) + basis_type_hint = period_type_hint(basis_period) + if basis_type_hint and basis_type_hint != fact_period_type: + raise ValueError( + f"target_references.json: basis period {basis_id!r} value " + f"{basis_period!r} implies period type {basis_type_hint!r}, not " + f"declared fact_period_type {fact_period_type!r}." + ) + reference_type_hint = period_type_hint(reference.period) + if reference_type_hint and reference_type_hint != fact_period_type: + raise ValueError( + f"target_references.json: {context} period {reference.period!r} " + f"implies period type {reference_type_hint!r}, not basis " + f"fact_period_type {fact_period_type!r}." + ) selector_period_type = str(reference.ledger_selector.get("period_type", "")) if selector_period_type != fact_period_type: raise ValueError( @@ -1115,27 +1222,28 @@ def _validate_target_profile( f"target_references.json: subnational {context} must bind the " "same non-empty geography_vintage in its selector and metadata." ) + accepted_typed_aliases = geography_vintage_aliases.get(geography_level) + if accepted_typed_aliases is None: + raise ValueError( + f"target_references.json: subnational {context} uses geography " + f"layer {geography_level!r}, but the authoritative typed " + "geography layer-vintage registry does not declare it." + ) + if selector_vintage not in accepted_typed_aliases: + raise ValueError( + f"target_references.json: {context} geography vintage " + f"{selector_vintage!r} is not an exact typed authority alias " + f"for layer {geography_level!r}; expected one of " + f"{sorted(accepted_typed_aliases)!r}." + ) if geography_spine is not None and ( geography_level == geography_spine.geography_spine.geography_level ): spine = geography_spine.geography_spine - short_code_system = spine.code_system.removeprefix(f"{country}_") - accepted_spine_vintages = { - spine.vintage, - f"{short_code_system}_{spine.vintage}", - } - if selector_vintage not in accepted_spine_vintages: - raise ValueError( - f"target_references.json: {context} geography vintage " - f"{selector_vintage!r} differs from the " - f"{geography_level!r} spine vintage " - f"{spine.vintage!r} under code system " - f"{spine.code_system!r}." - ) legacy_nis_vintage = reference.metadata.get("nis_vintage") if legacy_nis_vintage is not None and legacy_nis_vintage not in { spine.vintage, - selector_vintage, + *accepted_typed_aliases, }: raise ValueError( f"target_references.json: {context} legacy " @@ -1361,8 +1469,9 @@ class ResolvedCountrySpec: support_spine: The support-spine manifest, when declared. geography_spine: The geography-spine manifest, when declared. target_references: Ledger target references, when declared. - target_profile: The validated value-free target selection policy carried - by ``target_references.json``. + target_profile: The validated value-free target declaration carried by + ``target_references.json``. Tier tolerances and target roles remain + metadata until a separate runtime integration consumes them. gates: The gate selection, when declared. release_contract: The release contract, when declared. take_up_contract: The constants-era take-up compatibility view. For a @@ -2009,6 +2118,7 @@ def load_country_spec(country: str | Path) -> ResolvedCountrySpec: target_references, country=declared_country, geography_spine=geography_spine, + resolved_spec=resolved_spec, ) if "target_references.json" in payloads else MappingProxyType({}) diff --git a/packages/microcosm-build/src/microcosm/build/ledger_targets.py b/packages/microcosm-build/src/microcosm/build/ledger_targets.py index 1e35cbd93..fd61d8f23 100644 --- a/packages/microcosm-build/src/microcosm/build/ledger_targets.py +++ b/packages/microcosm-build/src/microcosm/build/ledger_targets.py @@ -28,6 +28,7 @@ MULTI_FACT_VALUE_OPERATIONS = frozenset( ("sum", "calendar_year_average", "latest_plateau", "count_x_mean") ) +EXACT_PERIOD_VALUE_OPERATIONS = frozenset(("identity", "sum", "count_x_mean")) DEFAULT_HIERARCHY_MATCH_SPEC_FIELDS = ("entity", "period", "family", "filter") @@ -126,6 +127,18 @@ def __post_init__(self) -> None: f"LedgerTargetReference {self.name!r}: period_match_policy=" "'exact' requires an explicit target period." ) + if ( + self.period_match_policy == "exact" + and self.value_operation not in EXACT_PERIOD_VALUE_OPERATIONS + ): + raise ValueError( + f"LedgerTargetReference {self.name!r}: period_match_policy=" + f"'exact' does not support value_operation {self.value_operation!r}; " + "calendar_year_average and latest_plateau consume subperiod " + "series whose selection semantics are not an exact scalar-period " + "match. Use latest_not_after or add an explicit subperiod " + "materialization contract before activating this reference." + ) if not self.entity: raise ValueError( f"LedgerTargetReference {self.name!r}: entity must be non-empty." @@ -394,11 +407,32 @@ def compile_ledger_target_references( fact_index = _ledger_fact_index(facts) specs: list[TargetSpec] = [] for reference in references: + _require_executable_reference(reference) resolved = _resolve_reference_fact(reference, fact_index) specs.append(target_spec_from_ledger_reference(resolved, reference)) return TargetRegistry(specs, country=country) +def _require_executable_reference(reference: LedgerTargetReference) -> None: + """Refuse authoring placeholders until a scalar/fanout reference replaces them. + + Country packages may declare a future target surface before Chronicle has + harvested and pinned the corresponding facts. Those rows are useful schema + evidence, but they are not executable selectors: in particular, resolving a + multi-cell table through ``identity`` would either depend on accidental + singleton input or become ambiguous as soon as the next cell arrived. + """ + + activation_status = reference.metadata.get("activation_status", "") + if activation_status and activation_status != "active": + raise ValueError( + f"Ledger target reference {reference.name!r} is a non-executable " + f"placeholder with activation_status={activation_status!r}. Replace " + "it with harvested, cell-pinned references (or a reviewed scalar " + "fact reference) before compilation." + ) + + def apply_ledger_target_profile( registry: TargetRegistry, profile: Mapping[str, object] | None, @@ -532,6 +566,7 @@ def target_spec_from_ledger_reference( ) -> TargetSpec: """Compile one resolved Ledger fact plus Microcosm mapping into a target.""" + _require_executable_reference(reference) facts = fact if isinstance(fact, tuple) else (fact,) if not facts: raise ValueError(f"Ledger target reference {reference.name!r} has no facts.") @@ -543,7 +578,7 @@ def target_spec_from_ledger_reference( ) numeric_values = [] for member in facts: - _validate_reference_period(member, reference) + _validate_resolved_reference_fact(member, reference) numeric_values.append(_numeric_fact_value(member, reference)) _validate_fact_aggregation(member, reference) @@ -1221,13 +1256,35 @@ def _eligible_selector_matches( def _assertion_allowed(reference: LedgerTargetReference, fact: object) -> bool: assertion = _fact_assertion(fact) + if assertion == "observation": + return True if assertion == "source_projection": return reference.assertion_policy == "allow_source_projection" - return True + return False + + +def _validate_resolved_reference_fact( + fact: object, + reference: LedgerTargetReference, +) -> None: + """Apply consumer policy after either identifier or selector resolution. + + Selector filtering still uses these predicates to choose an eligible row + from a series. This post-resolution check is the authority: exact keys and + source-record identifiers must not bypass the assertion or period policy. + """ + + _validate_reference_period(fact, reference) + if not _assertion_allowed(reference, fact): + raise ValueError( + f"Ledger target reference {reference.name!r} assertion_policy=" + f"{reference.assertion_policy!r} does not allow resolved fact " + f"assertion {_fact_assertion(fact)!r}." + ) def _validate_reference_period(fact: object, reference: LedgerTargetReference) -> None: - """Enforce an explicit exact-period consumer contract after resolution. + """Enforce the declared period consumer contract after either resolution path. An exact-period reference may consume either an observation at that period or a Ledger ``source_projection`` whose published fact period is that same @@ -1237,6 +1294,14 @@ def _validate_reference_period(fact: object, reference: LedgerTargetReference) - """ if reference.period_match_policy != "exact": + if not _not_after_target_period( + _period_key(fact), _period_key_from_value(reference.period) + ): + raise ValueError( + f"Ledger target reference {reference.name!r} requires a fact " + f"at or before target period {reference.period!r}, but resolved " + f"fact period {_at(fact, 'period', 'value')!r}." + ) return if not _exact_period_matches(fact, reference): raise ValueError( @@ -1368,15 +1433,9 @@ def _is_period_fragment(value: str) -> bool: def _is_period_token(value: str) -> bool: - normalized = value.lower().replace("-", "_") - if normalized.startswith("month"): - normalized = normalized[len("month") :] - if normalized[:2] in {"ty", "cy", "fy", "ay"}: - normalized = normalized[2:] - parts = normalized.split("_", maxsplit=1) - if len(parts) == 2 and all(part.isdigit() for part in parts): - return len(parts[0]) == 4 and len(parts[1]) in {1, 2, 4} - return normalized.isdigit() and len(normalized) == 4 + period_key = _period_key_from_value(value) + # Source-table numbers such as table_1_2 are identity, not year tokens. + return bool(period_key[0]) and 1000 <= period_key[1] // 100 <= 9999 def _period_key(fact: object) -> tuple[int, int, str]: @@ -1388,18 +1447,18 @@ def _exact_period_matches(fact: object, reference: LedgerTargetReference) -> boo expected_value = reference.period actual_value = _at(fact, "period", "value") - if not _period_values_semantically_equal(expected_value, actual_value): + if not period_values_semantically_equal(expected_value, actual_value): return False actual_type = _str_at(fact, "period", "type") selector_type = str(reference.ledger_selector.get("period_type", "")) - expected_type_hint = _period_type_hint(expected_value) + expected_type_hint = period_type_hint(expected_value) if selector_type and expected_type_hint and selector_type != expected_type_hint: return False expected_type = selector_type or expected_type_hint if expected_type and actual_type != expected_type: return False - actual_type_hint = _period_type_hint(actual_value) + actual_type_hint = period_type_hint(actual_value) return not actual_type_hint or actual_type_hint == actual_type @@ -1413,76 +1472,98 @@ def _reference_period_partition_key( return period_key -def _period_values_semantically_equal(left: object, right: object) -> bool: +def period_values_semantically_equal(left: object, right: object) -> bool: """Treat supported typed and scalar spellings of one period as equal.""" - left_key = _period_key_from_value(left) - right_key = _period_key_from_value(right) + left_key, left_hint, left_range_end = _normalize_period_value(left) + right_key, right_hint, right_range_end = _normalize_period_value(right) if left_key[0] and right_key[0]: - return left_key[:2] == right_key[:2] + return left_key[:2] == right_key[:2] and left_range_end == right_range_end + if left_hint or right_hint: + # Recognized but malformed typed values must not match, even when the + # same invalid label appears on both sides of the contract. + return False return left_key == right_key -def _period_type_hint(value: object) -> str: - if not isinstance(value, str): - return "" - normalized = value.lower().replace("-", "_") - long_prefixes = ( - ("tax_year_", "tax_year"), - ("calendar_year_", "calendar_year"), - ("fiscal_year_", "fiscal_year"), - ("academic_year_", "academic_year"), - ("month", "month"), - ) - long_hint = next( - ( - period_type - for prefix, period_type in long_prefixes - if normalized.startswith(prefix) - ), - "", - ) - if long_hint: - return long_hint - aliases = { - "ty": "tax_year", - "cy": "calendar_year", - "fy": "fiscal_year", - "ay": "academic_year", - } - alias = normalized[:2] - suffix = normalized[2:].lstrip("_") - if alias in aliases and suffix[:1].isdigit(): - return aliases[alias] - return "" +def period_type_hint(value: object) -> str: + """Return the period-kind prefix recognized by the shared normalizer.""" + + return _normalize_period_value(value)[1] def _period_key_from_value(value: object) -> tuple[int, int, str]: + return _normalize_period_value(value)[0] + + +def _normalize_period_value( + value: object, +) -> tuple[tuple[int, int, str], str, int | None]: + """Normalize one scalar/typed period spelling once for all consumers. + + Annual aliases (``ty``, ``cy``, ``fy``, and ``ay``) and their long forms + share one parser. A two-part value under an annual prefix is an annual + range (for example ``academic_year_2023_24``), while an untyped or + ``month``-prefixed ``2023_04`` remains a monthly point. Annual ranges must + end in the following year. Keep that explicit end year in the semantic + identity: an annual range is not interchangeable with a scalar year. + """ + label = "" if value is None else str(value) - normalized = label.lower().replace("-", "_") - if normalized[:2] in {"ty", "cy", "fy"}: - normalized = normalized[2:] - for prefix in ("month", "tax_year_", "calendar_year_", "fiscal_year_"): + normalized = label.strip().lower().replace("-", "_") + type_hint = "" + long_prefixes = ( + ("tax_year", "tax_year"), + ("calendar_year", "calendar_year"), + ("fiscal_year", "fiscal_year"), + ("academic_year", "academic_year"), + ("month", "month"), + ) + for prefix, period_type in long_prefixes: if normalized.startswith(prefix): - normalized = normalized[len(prefix) :] + suffix = normalized[len(prefix) :].lstrip("_") + if suffix[:1].isdigit(): + normalized = suffix + type_hint = period_type break + if not type_hint: + aliases = { + "ty": "tax_year", + "cy": "calendar_year", + "fy": "fiscal_year", + "ay": "academic_year", + } + alias = normalized[:2] + suffix = normalized[2:].lstrip("_") + if alias in aliases and suffix[:1].isdigit(): + normalized = suffix + type_hint = aliases[alias] + parts = normalized.split("_", maxsplit=1) if len(parts) == 2: year, suffix = parts - if year.isdigit() and suffix.isdigit(): - if len(suffix) == 4: - suffix_value = 99 - elif len(suffix) <= 2 and int(suffix) <= 12: - suffix_value = int(suffix) - elif len(suffix) <= 2: - suffix_value = 99 - else: - suffix_value = int(suffix) - return (1, int(year) * 100 + suffix_value, label) + if year.isdigit() and suffix.isdigit() and len(suffix) in {1, 2, 4}: + annual_range = type_hint in { + "tax_year", + "calendar_year", + "fiscal_year", + "academic_year", + } or (not type_hint and (len(suffix) == 4 or int(suffix) > 12)) + if annual_range: + range_end = int(year) + 1 + expected_suffix = range_end if len(suffix) == 4 else range_end % 100 + if len(suffix) not in {2, 4} or int(suffix) != expected_suffix: + return (0, 0, label), type_hint, None + return (1, int(year) * 100 + 99, label), type_hint, range_end + if len(suffix) <= 2 and 1 <= int(suffix) <= 12: + return (1, int(year) * 100 + int(suffix), label), type_hint, None + return (0, 0, label), type_hint, None + if type_hint == "month": + return (0, 0, label), type_hint, None try: - return (1, int(normalized) * 100 + 99, label) + return (1, int(normalized) * 100 + 99, label), type_hint, None except ValueError: - return (0, 0, label) + return (0, 0, label), type_hint, None def _prefer_period_candidate( diff --git a/packages/microcosm-build/tests/golden/be_country_spec.json b/packages/microcosm-build/tests/golden/be_country_spec.json index 859250080..5cfc5cea4 100644 --- a/packages/microcosm-build/tests/golden/be_country_spec.json +++ b/packages/microcosm-build/tests/golden/be_country_spec.json @@ -1,6 +1,6 @@ { "country": "be", - "fingerprint": "3ebf6c9b64795c4e6ecb10d6b031721e0de14a843358793116dded05314df9a4", + "fingerprint": "6f1dab96ccedb21fc261acf9ebbccfc036b2e354a78cf9314e394707a1888cf6", "gate_ids": [ "calibration_per_family_fit", "national_and_nuts1_admin_aggregates", @@ -29,7 +29,7 @@ }, "resource_hashes": { "country_package.json": "1dff9052dbdcdbceeb077d1212008226cf783aa5d33cbda52d07437aef69d30b", - "gates.json": "49635a32881aec3e914e74ed1bd3a408b427caac902c8e3d01118e99e8c9d1c8", + "gates.json": "85c07f713168058a081b40cf94da6a33035108ba7807e04818486032fb2ad580", "geography_spine.json": "0b39217386dbc9b053a036766004f10a3a34434a293c1e956e0f4f053a08e1bf", "release_contract.json": "af956e05782cd2aa5df108b216f0ff2a835f2a4e87d68419239b537550082c25", "source_stages.json": "64e45f48b5c3bb00a710e56f805c7c8c441ddc1050cfb61d846d45e9cbbb16b2", @@ -39,7 +39,7 @@ "spec/sources.yaml": "1ca516ad27ee815cb60f451901dcdc6b8989c4f1731d0f6f3f1e50714e29e35e", "spec/spine.yaml": "908d99488c3905806ec6878d96b0e4a4cbb67c6ec69d53fe5f1d160246fdb513", "spec/vintages.yaml": "c842490287404cf306907800b0bcadec5d954e7e9c510f4b293786c5bedd0910", - "target_references.json": "09ee9cd8618751c02e2da5708178f8c6a7fa6afbe6f524d632daae3483407eed" + "target_references.json": "a8c9f875e3a04d082fbacfa8511c856b3879077d8a4276df50ea38c345e702ea" }, "resources": [ "spec/bundle.yaml", diff --git a/packages/microcosm-build/tests/test_country_spec.py b/packages/microcosm-build/tests/test_country_spec.py index d2493ab18..145477bb3 100644 --- a/packages/microcosm-build/tests/test_country_spec.py +++ b/packages/microcosm-build/tests/test_country_spec.py @@ -11,7 +11,8 @@ from __future__ import annotations import json -from dataclasses import FrozenInstanceError +import shutil +from dataclasses import FrozenInstanceError, replace from pathlib import Path import pytest @@ -376,18 +377,30 @@ def test_target_selector_vocabulary_resolves_scalar_facts(self, spec) -> None: ) # Isolate each scalar probe: this exercises the real closed resolver - # vocabulary without implying that a table placeholder performs fanout. + # vocabulary through the shape a harvested replacement would carry, + # without activating the packaged authoring placeholder. for ordinal, reference in enumerate(references, start=1): + active_reference = replace( + reference, + metadata={ + key: value + for key, value in reference.metadata.items() + if key != "activation_status" + }, + ) registry = compile_ledger_target_references( - [_armenia_scalar_ledger_fact(reference, ordinal)], - [reference], + [_armenia_scalar_ledger_fact(active_reference, ordinal)], + [active_reference], country="am", ) assert len(registry.specs) == 1 assert registry.specs[0].name == reference.name assert registry.specs[0].value > 0 - def test_table_placeholder_refuses_multiple_matching_cells(self, spec) -> None: + @pytest.mark.parametrize("fact_count", [1, 2]) + def test_table_placeholder_refuses_compilation_before_cell_fanout( + self, spec, fact_count + ) -> None: reference = next( reference for reference in spec.target_references @@ -406,8 +419,10 @@ def test_table_placeholder_refuses_multiple_matching_cells(self, spec) -> None: ), ] - with pytest.raises(ValueError, match="matched multiple Ledger facts"): - compile_ledger_target_references(facts, [reference], country="am") + with pytest.raises(ValueError, match="non-executable placeholder"): + compile_ledger_target_references( + facts[:fact_count], [reference], country="am" + ) def test_gates_use_greenfield_and_weight_health_posture(self, spec) -> None: selected = {gate.gate for gate in spec.gates.gates} @@ -523,6 +538,13 @@ def test_targets_arrive_by_reference_with_no_values(self, spec) -> None: assert commune.metadata["geography_vintage"] == "nis_2025" assert commune.ledger_selector["geography_vintage"] == "nis_2025" assert commune.metadata["criticality"] == "diagnostic" + assert { + by_name[name].metadata["activation_status"] + for name in { + "statbel_population_by_age_sex_region", + "statbel_fiscal_income_by_commune", + } + } == {"requires_harvested_cell_references"} payload = json.loads( (COUNTRY_PACKAGE_ROOT / "be/target_references.json").read_text( @@ -533,7 +555,9 @@ def test_targets_arrive_by_reference_with_no_values(self, spec) -> None: _nested_mapping_keys(payload["target_references"]) ) - def test_target_selectors_match_the_chronicle_fact_vocabulary(self, spec) -> None: + def test_target_selectors_declare_the_intended_chronicle_vocabulary( + self, spec + ) -> None: references = {reference.name: reference for reference in spec.target_references} expected = { "statbel_population_by_age_sex_region": ( @@ -542,7 +566,7 @@ def test_target_selectors_match_the_chronicle_fact_vocabulary(self, spec) -> Non "calendar_year", 2023, "nuts1", - "NUTS_2024", + "nuts1_2025", ), "statbel_fiscal_income_by_commune": ( "statbel_fiscal_income", @@ -639,9 +663,114 @@ def test_target_profile_declares_tiers_and_income_basis(self, spec) -> None: if reference.metadata["target_role"] == "calibration" } >= set(profile["required_families"]) + def test_target_profile_tiers_and_roles_are_declaration_only(self, spec) -> None: + assert all(reference.tolerance is None for reference in spec.target_references) + description = json.loads( + (COUNTRY_PACKAGE_ROOT / "be/target_references.json").read_text( + encoding="utf-8" + ) + )["description"] + assert "validated declaration metadata only" in description + assert "does not yet wire them into runtime calibration" in description + assert "current Chronicle Belgian catalog does not satisfy" in description + assert "#264" in description + assert "Declaration-only intended Belgian gate posture" in spec.gates.policy + assert "not implemented here" in spec.gates.policy + + @pytest.mark.parametrize( + ("reference_name", "aliases"), + [ + ( + "statbel_population_by_age_sex_region", + ("be_nuts1_2025", "nuts1_2025", "2025_nuts1"), + ), + ( + "statbel_fiscal_income_by_commune", + ("be_nis_2025", "nis_2025", "2025_nis"), + ), + ], + ) + def test_subnational_targets_accept_only_declared_typed_vintage_aliases( + self, tmp_path, reference_name, aliases + ) -> None: + for index, alias in enumerate(aliases): + package_dir = tmp_path / f"case-{index}" / "be" + shutil.copytree(COUNTRY_PACKAGE_ROOT / "be", package_dir) + target_path = package_dir / "target_references.json" + payload = json.loads(target_path.read_text(encoding="utf-8")) + reference = next( + row + for row in payload["target_references"] + if row["name"] == reference_name + ) + reference["ledger_selector"]["geography_vintage"] = alias + reference["metadata"]["geography_vintage"] = alias + target_path.write_text(json.dumps(payload), encoding="utf-8") + + loaded = load_country_spec(package_dir) + loaded_reference = next( + row for row in loaded.target_references if row.name == reference_name + ) + assert loaded_reference.ledger_selector["geography_vintage"] == alias + + @pytest.mark.parametrize( + ("reference_name", "invalid_vintage"), + [ + ("statbel_population_by_age_sex_region", "NUTS_2024"), + ("statbel_fiscal_income_by_commune", "nis_2024"), + ], + ) + def test_subnational_targets_refuse_vintages_outside_typed_registry( + self, tmp_path, reference_name, invalid_vintage + ) -> None: + package_dir = tmp_path / "be" + shutil.copytree(COUNTRY_PACKAGE_ROOT / "be", package_dir) + target_path = package_dir / "target_references.json" + payload = json.loads(target_path.read_text(encoding="utf-8")) + reference = next( + row for row in payload["target_references"] if row["name"] == reference_name + ) + reference["ledger_selector"]["geography_vintage"] = invalid_vintage + reference["metadata"]["geography_vintage"] = invalid_vintage + target_path.write_text(json.dumps(payload), encoding="utf-8") + + with pytest.raises(ValueError, match="not an exact typed authority alias"): + load_country_spec(package_dir) + + def test_subnational_target_requires_a_typed_geography_layer( + self, tmp_path + ) -> None: + package_dir = tmp_path / "be" + shutil.copytree(COUNTRY_PACKAGE_ROOT / "be", package_dir) + target_path = package_dir / "target_references.json" + payload = json.loads(target_path.read_text(encoding="utf-8")) + reference = payload["target_references"][0] + reference["ledger_selector"]["geography_level"] = "province" + target_path.write_text(json.dumps(payload), encoding="utf-8") + + with pytest.raises(ValueError, match="does not declare it"): + load_country_spec(package_dir) + + @pytest.mark.parametrize( + "reference_name", + [ + "statbel_population_by_age_sex_region", + "statbel_fiscal_income_by_commune", + ], + ) + def test_multicell_be_placeholders_cannot_compile_before_fanout( + self, spec, reference_name + ) -> None: + reference = next( + row for row in spec.target_references if row.name == reference_name + ) + + with pytest.raises(ValueError, match="non-executable placeholder"): + compile_ledger_target_references([], [reference], country="be") + def test_gates_select_no_incumbent_comparison(self, spec) -> None: selected = {gate.gate for gate in spec.gates.gates} - assert "parity" not in selected # no incumbent: oracles replace it + assert "parity" not in selected # no incumbent; #264 remains separate assert "export_surface" not in selected assert "per_family_fit" in selected assert "formula_owned_export" in selected @@ -1555,6 +1684,72 @@ def test_schema2_target_profile_refuses_reference_period_drift( with pytest.raises(ValueError, match="does not match basis period"): load_country_spec(package_dir) + @pytest.mark.parametrize( + ("reference_period", "basis_period"), + [ + ("academic_year_2023_24", "ay2023_24"), + ("ay_2023_24", "academic-year-2023-2024"), + ("academic_year_1999_00", "ay1999_2000"), + ], + ) + def test_schema2_target_profile_uses_shared_academic_period_semantics( + self, tmp_path, reference_period, basis_period + ) -> None: + files = _package_with_schema2_targets() + reference = files["target_references.json"]["target_references"][0] + reference["period"] = reference_period + reference["ledger_selector"]["period_type"] = "academic_year" + basis = files["target_references.json"]["target_profile"]["basis_periods"][ + "population_2023" + ] + basis["period"] = basis_period + basis["fact_period_type"] = "academic_year" + package_dir = _write_package(tmp_path, files) + + spec = load_country_spec(package_dir) + + assert spec.target_references[0].period == reference_period + + @pytest.mark.parametrize( + ("reference_period", "basis_period"), + [ + ("academic_year_2023_24", "ay2023_25"), + ("academic_year_2023_24", "ay2023_04"), + ("academic_year_2023_24", "ay2023"), + ("academic_year_2023_25", "academic_year_2023_25"), + ], + ) + def test_schema2_target_profile_preserves_academic_period_range_end( + self, tmp_path, reference_period, basis_period + ) -> None: + files = _package_with_schema2_targets() + reference = files["target_references.json"]["target_references"][0] + reference["period"] = reference_period + reference["ledger_selector"]["period_type"] = "academic_year" + basis = files["target_references.json"]["target_profile"]["basis_periods"][ + "population_2023" + ] + basis["period"] = basis_period + basis["fact_period_type"] = "academic_year" + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="does not match basis period"): + load_country_spec(package_dir) + + def test_schema2_target_profile_refuses_typed_period_kind_drift( + self, tmp_path + ) -> None: + files = _package_with_schema2_targets() + files["target_references.json"]["target_references"][0]["period"] = "ay2023_24" + basis = files["target_references.json"]["target_profile"]["basis_periods"][ + "population_2023" + ] + basis["period"] = "academic_year_2023_24" + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="implies period type 'academic_year'"): + load_country_spec(package_dir) + def test_schema2_target_profile_refuses_fact_period_type_drift( self, tmp_path ) -> None: diff --git a/packages/microcosm-build/tests/test_ledger_targets.py b/packages/microcosm-build/tests/test_ledger_targets.py index 07b611bd3..2f1933ee9 100644 --- a/packages/microcosm-build/tests/test_ledger_targets.py +++ b/packages/microcosm-build/tests/test_ledger_targets.py @@ -705,6 +705,109 @@ def test__given_equivalent_exact_period_labels__then_target_period_is_used( assert spec.period == reference_period +@pytest.mark.parametrize( + ("reference_period", "fact_period"), + [ + ("academic_year_2023_24", "ay2023_24"), + ("ay_2023_24", "academic-year-2023-2024"), + ("academic_year_1999_00", "ay1999_2000"), + ], +) +def test__given_equivalent_academic_period_labels__then_exact_match_uses_shared_parser( + reference_period, fact_period +) -> None: + fact = _consumer_fact_row_for_period(2023, value=15_000_000_000_000) + fact["period"] = {"type": "academic_year", "value": fact_period} + reference = LedgerTargetReference( + name="academic-year total", + ledger_selector={ + "source_name": "irs_soi", + "source_measure_id": "adjusted_gross_income", + "period_type": "academic_year", + "geography_level": "country", + "geography_id": "0100000US", + }, + entity="tax_unit", + measure="adjusted_gross_income", + period=reference_period, + family="irs_soi", + period_match_policy="exact", + ) + + registry = compile_ledger_target_references([fact], [reference], country="us") + + (spec,) = registry.specs + assert spec.period == reference_period + + +@pytest.mark.parametrize( + ("reference_period", "fact_period"), + [ + ("academic_year_2023_24", "ay2023_25"), + ("academic_year_2023_24", "ay2023_04"), + ("academic_year_2023_24", "ay2023"), + ("academic_year_2023_25", "academic_year_2023_25"), + ], +) +@pytest.mark.parametrize( + "resolution_route", ["selector", "ledger_fact_key", "ledger_source_record_id"] +) +def test_exact_academic_periods_do_not_discard_the_range_end( + reference_period, fact_period, resolution_route +) -> None: + fact = _consumer_fact_row_for_period(2023, value=15_000_000_000_000) + fact["period"] = {"type": "academic_year", "value": fact_period} + identifiers = ( + {} + if resolution_route == "selector" + else { + resolution_route: ( + fact["aggregate_fact_key"] + if resolution_route == "ledger_fact_key" + else fact["lineage"]["source_record_id"] + ) + } + ) + reference = LedgerTargetReference( + name="academic-year total", + **identifiers, + ledger_selector={ + "source_name": "irs_soi", + "source_measure_id": "adjusted_gross_income", + "period_type": "academic_year", + }, + entity="tax_unit", + measure="adjusted_gross_income", + period=reference_period, + period_match_policy="exact", + ) + + with pytest.raises(ValueError, match="exact (target )?period"): + compile_ledger_target_references([fact], [reference], country="us") + + +@pytest.mark.parametrize( + "value_operation", + ["calendar_year_average", "latest_plateau"], +) +def test__given_exact_period_with_subperiod_operation__then_reference_is_refused( + value_operation, +) -> None: + with pytest.raises( + ValueError, + match="does not support value_operation", + ): + LedgerTargetReference( + name="unsupported exact aggregate", + ledger_selector={"source_name": "official_series"}, + value_operation=value_operation, + entity="person", + measure="people", + period=2025, + period_match_policy="exact", + ) + + def test__given_equivalent_exact_period_label__then_period_type_still_matches() -> None: fact = _consumer_fact_row_for_period(2023, value=15_000_000_000_000) fact["period"] = {"type": "calendar_year", "value": "tax_year_2023"} @@ -856,6 +959,36 @@ def test__given_period_bearing_groupby_value__then_latest_source_period_is_used( assert registry.specs[0].value == 65_900_000_000 +@pytest.mark.parametrize( + ("layout_field", "older_value", "newer_value"), + [ + ("record_set_id", "irs_soi.ty2022.table_1_1", "irs_soi.ty2023.table_1_2"), + ("record_set_id", "irs_soi.ty2022.1", "irs_soi.ty2023.2"), + ("groupby_value_id", "1", "2"), + ], +) +def test_period_normalization_preserves_non_year_numeric_series_identifiers( + layout_field, older_value, newer_value +) -> None: + older = _consumer_fact_row_for_period(2022, value=14_000_000_000_000) + newer = _consumer_fact_row_for_period(2023, value=15_000_000_000_000) + older["layout"][layout_field] = older_value + newer["layout"][layout_field] = newer_value + reference = LedgerTargetReference( + name="ambiguous source tables", + ledger_selector={ + "source_name": "irs_soi", + "source_measure_id": "adjusted_gross_income", + }, + entity="tax_unit", + measure="adjusted_gross_income", + period=2023, + ) + + with pytest.raises(ValueError, match="multiple Ledger facts"): + compile_ledger_target_references([older, newer], [reference], country="us") + + def test__given_academic_year_record_sets__then_latest_source_period_is_used() -> None: """The SLC series key one record set per academic year (…ay2023, …ay2024). @@ -2499,6 +2632,70 @@ def test_ledger_reference_projection_fact_excluded_by_default(): ) +@pytest.mark.parametrize( + "identifier_field", + ["ledger_fact_key", "ledger_source_record_id"], +) +def test_ledger_reference_identifier_cannot_bypass_observed_only_policy( + identifier_field, +) -> None: + fact = _consumer_fact_row( + aggregate_fact_key="ledger.aggregate_fact.v2:projected-identifier", + assertion="source_projection", + ) + identifier = ( + fact["aggregate_fact_key"] + if identifier_field == "ledger_fact_key" + else fact["lineage"]["source_record_id"] + ) + reference = LedgerTargetReference( + name="identifier-bound observed target", + **{identifier_field: identifier}, + entity="household", + measure="adjusted_gross_income", + family="irs_soi", + period=2023, + period_match_policy="exact", + ) + + with pytest.raises( + ValueError, + match="assertion_policy='observed_only'.*source_projection", + ): + compile_ledger_target_references([fact], [reference], country="us") + + +@pytest.mark.parametrize( + "identifier_field", ["ledger_fact_key", "ledger_source_record_id"] +) +@pytest.mark.parametrize("fact_period", [2022, 2023, 2024]) +def test_ledger_reference_identifier_enforces_latest_not_after_policy( + identifier_field, fact_period +) -> None: + fact = _consumer_fact_row_for_period(fact_period, value=15_000_000_000_000) + identifier = ( + fact["aggregate_fact_key"] + if identifier_field == "ledger_fact_key" + else fact["lineage"]["source_record_id"] + ) + reference = LedgerTargetReference( + name="identifier-bound latest target", + **{identifier_field: identifier}, + entity="household", + measure="adjusted_gross_income", + family="irs_soi", + period=2023, + period_match_policy="latest_not_after", + ) + + if fact_period > 2023: + with pytest.raises(ValueError, match="at or before target period"): + compile_ledger_target_references([fact], [reference], country="us") + else: + registry = compile_ledger_target_references([fact], [reference], country="us") + assert registry.specs[0].value == 15_000_000_000_000 + + def test_ledger_reference_projection_fact_allowed_by_policy(): reference = LedgerTargetReference( name="cbo_projected_agi", From 7a1d306a6351f4cdeaadfda3547432234046af57 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sat, 29 Aug 2026 21:21:46 -0400 Subject: [PATCH 4/4] Fix issues from review: period and geography-vintage contracts --- .../src/microcosm/build/country_spec.py | 4 +- .../src/microcosm/build/ledger_targets.py | 110 ++++++++---- .../tests/test_country_spec.py | 126 +++++++++++++ .../tests/test_ledger_targets.py | 166 +++++++++++++++++- 4 files changed, 372 insertions(+), 34 deletions(-) diff --git a/packages/microcosm-build/src/microcosm/build/country_spec.py b/packages/microcosm-build/src/microcosm/build/country_spec.py index 060c38960..1eb45e39c 100644 --- a/packages/microcosm-build/src/microcosm/build/country_spec.py +++ b/packages/microcosm-build/src/microcosm/build/country_spec.py @@ -1172,7 +1172,9 @@ def _validate_target_profile( f"{basis_id!r}." ) basis_period, fact_period_type = normalized_periods[basis_id] - if not period_values_semantically_equal(reference.period, basis_period): + if not period_values_semantically_equal( + reference.period, basis_period, declared_type=fact_period_type + ): raise ValueError( f"target_references.json: {context} period {reference.period!r} " f"does not match basis period {basis_id!r}." diff --git a/packages/microcosm-build/src/microcosm/build/ledger_targets.py b/packages/microcosm-build/src/microcosm/build/ledger_targets.py index fd61d8f23..83bca9306 100644 --- a/packages/microcosm-build/src/microcosm/build/ledger_targets.py +++ b/packages/microcosm-build/src/microcosm/build/ledger_targets.py @@ -1271,7 +1271,8 @@ def _validate_resolved_reference_fact( Selector filtering still uses these predicates to choose an eligible row from a series. This post-resolution check is the authority: exact keys and - source-record identifiers must not bypass the assertion or period policy. + source-record identifiers must not bypass assertion, period, or declared + geography-vintage policy. """ _validate_reference_period(fact, reference) @@ -1281,6 +1282,16 @@ def _validate_resolved_reference_fact( f"{reference.assertion_policy!r} does not allow resolved fact " f"assertion {_fact_assertion(fact)!r}." ) + vintage_pin = reference.ledger_selector.get("geography_vintage") + if vintage_pin is not None and vintage_pin != "": + if not _str_at(fact, "geography", "vintage") or not _fact_matches_selector( + fact, {"geography_vintage": vintage_pin} + ): + raise ValueError( + f"Ledger target reference {reference.name!r} requires geography " + f"vintage {vintage_pin!r}, but resolved fact has vintage " + f"{_at(fact, 'geography', 'vintage')!r}." + ) def _validate_reference_period(fact: object, reference: LedgerTargetReference) -> None: @@ -1447,9 +1458,6 @@ def _exact_period_matches(fact: object, reference: LedgerTargetReference) -> boo expected_value = reference.period actual_value = _at(fact, "period", "value") - if not period_values_semantically_equal(expected_value, actual_value): - return False - actual_type = _str_at(fact, "period", "type") selector_type = str(reference.ledger_selector.get("period_type", "")) expected_type_hint = period_type_hint(expected_value) @@ -1459,30 +1467,51 @@ def _exact_period_matches(fact: object, reference: LedgerTargetReference) -> boo if expected_type and actual_type != expected_type: return False actual_type_hint = period_type_hint(actual_value) - return not actual_type_hint or actual_type_hint == actual_type + if actual_type_hint and actual_type_hint != actual_type: + return False + return period_values_semantically_equal( + expected_value, + actual_value, + declared_type=expected_type or actual_type, + ) def _reference_period_partition_key( fact: object, reference: LedgerTargetReference, ) -> tuple[int, int, str]: - period_key = _period_key(fact) + period_key = ( + _normalize_period_value( + _at(fact, "period", "value"), + declared_type=_str_at(fact, "period", "type"), + )[0] + if reference.period_match_policy == "exact" + else _period_key(fact) + ) if reference.period_match_policy == "exact" and period_key[0]: return (period_key[0], period_key[1], "") return period_key -def period_values_semantically_equal(left: object, right: object) -> bool: - """Treat supported typed and scalar spellings of one period as equal.""" +def period_values_semantically_equal( + left: object, right: object, *, declared_type: str = "" +) -> bool: + """Compare valid period spellings, using the declared type for untyped values. + + Opaque publisher labels retain literal equality. Malformed numeric period + shapes do not, even when the same invalid spelling appears on both sides. + """ - left_key, left_hint, left_range_end = _normalize_period_value(left) - right_key, right_hint, right_range_end = _normalize_period_value(right) + left_key, _, left_range_end, left_malformed = _normalize_period_value( + left, declared_type=declared_type + ) + right_key, _, right_range_end, right_malformed = _normalize_period_value( + right, declared_type=declared_type + ) + if left_malformed or right_malformed: + return False if left_key[0] and right_key[0]: return left_key[:2] == right_key[:2] and left_range_end == right_range_end - if left_hint or right_hint: - # Recognized but malformed typed values must not match, even when the - # same invalid label appears on both sides of the contract. - return False return left_key == right_key @@ -1498,15 +1527,19 @@ def _period_key_from_value(value: object) -> tuple[int, int, str]: def _normalize_period_value( value: object, -) -> tuple[tuple[int, int, str], str, int | None]: + *, + declared_type: str = "", +) -> tuple[tuple[int, int, str], str, int | None, bool]: """Normalize one scalar/typed period spelling once for all consumers. Annual aliases (``ty``, ``cy``, ``fy``, and ``ay``) and their long forms share one parser. A two-part value under an annual prefix is an annual - range (for example ``academic_year_2023_24``), while an untyped or - ``month``-prefixed ``2023_04`` remains a monthly point. Annual ranges must - end in the following year. Keep that explicit end year in the semantic - identity: an annual range is not interchangeable with a scalar year. + range (for example ``academic_year_2023_24``). A declared annual type also + disambiguates untyped ``2003_04`` as a range rather than a monthly point. + Annual ranges must end in the following year. Keep that explicit end year + in the semantic identity: a range is not interchangeable with a scalar year. + The final flag distinguishes malformed numeric shapes from opaque labels, + so only the latter can fall back to literal equality. """ label = "" if value is None else str(value) @@ -1539,31 +1572,44 @@ def _normalize_period_value( normalized = suffix type_hint = aliases[alias] + annual_types = {"tax_year", "calendar_year", "fiscal_year", "academic_year"} + if ( + not type_hint + and declared_type in annual_types | {"month"} + and normalized[:1].isdigit() + ): + type_hint = declared_type + parts = normalized.split("_", maxsplit=1) if len(parts) == 2: year, suffix = parts if year.isdigit() and suffix.isdigit() and len(suffix) in {1, 2, 4}: - annual_range = type_hint in { - "tax_year", - "calendar_year", - "fiscal_year", - "academic_year", - } or (not type_hint and (len(suffix) == 4 or int(suffix) > 12)) + annual_range = type_hint in annual_types or ( + not type_hint and (len(suffix) == 4 or int(suffix) > 12) + ) if annual_range: range_end = int(year) + 1 expected_suffix = range_end if len(suffix) == 4 else range_end % 100 if len(suffix) not in {2, 4} or int(suffix) != expected_suffix: - return (0, 0, label), type_hint, None - return (1, int(year) * 100 + 99, label), type_hint, range_end + return (0, 0, label), type_hint, None, True + return (1, int(year) * 100 + 99, label), type_hint, range_end, False if len(suffix) <= 2 and 1 <= int(suffix) <= 12: - return (1, int(year) * 100 + int(suffix), label), type_hint, None - return (0, 0, label), type_hint, None + return ( + (1, int(year) * 100 + int(suffix), label), + type_hint, + None, + False, + ) + numeric_shape = year.isdigit() and all( + part.isdigit() or not part for part in suffix.split("_") + ) + return (0, 0, label), type_hint, None, bool(type_hint) or numeric_shape if type_hint == "month": - return (0, 0, label), type_hint, None + return (0, 0, label), type_hint, None, True try: - return (1, int(normalized) * 100 + 99, label), type_hint, None + return (1, int(normalized) * 100 + 99, label), type_hint, None, False except ValueError: - return (0, 0, label), type_hint, None + return (0, 0, label), type_hint, None, bool(type_hint) def _prefer_period_candidate( diff --git a/packages/microcosm-build/tests/test_country_spec.py b/packages/microcosm-build/tests/test_country_spec.py index 145477bb3..df94c31f6 100644 --- a/packages/microcosm-build/tests/test_country_spec.py +++ b/packages/microcosm-build/tests/test_country_spec.py @@ -2064,3 +2064,129 @@ def test_geography_vintage_policy_must_be_error(self, tmp_path) -> None: package_dir = _write_package(tmp_path, files) with pytest.raises(ValueError, match="vintage_policy"): load_country_spec(package_dir) + + +@pytest.mark.parametrize( + ("period_type", "invalid_period"), + [ + ("academic_year", "2023_25"), + ("academic_year", "2023_2025"), + ("academic_year", "2023_04"), + ("tax_year", "2023_25"), + ("calendar_year", "2023_2025"), + ("fiscal_year", "2023_00"), + ("month", "2023_00"), + ("month", "2023_13"), + ("month", "2023_24"), + ("month", "2023_2024"), + ("month", "2023"), + ], +) +def test_schema2_period_contract_refuses_equal_malformed_untyped_labels( + tmp_path, period_type, invalid_period +) -> None: + files = _package_with_schema2_targets() + reference = files["target_references.json"]["target_references"][0] + reference["period"] = invalid_period + reference["ledger_selector"]["period_type"] = period_type + basis = files["target_references.json"]["target_profile"]["basis_periods"][ + "population_2023" + ] + basis["period"] = invalid_period + basis["fact_period_type"] = period_type + package_dir = _write_package(tmp_path, files) + + with pytest.raises(ValueError, match="does not match basis period"): + load_country_spec(package_dir) + + +@pytest.mark.parametrize( + ("period_type", "reference_period", "basis_period"), + [ + ("tax_year", 2023, "ty2023"), + ("calendar_year", "2023", "calendar_year_2023"), + ("fiscal_year", "2023_24", "fy2023_2024"), + ("academic_year", "2003_04", "ay2003_04"), + ("academic_year", "1999_00", "academic_year_1999_2000"), + ("academic_year", 2023, "ay2023"), + ("month", "2003_04", "month_2003_04"), + ("month", "2023_1", "month_2023_01"), + ("academic_year", "publisher_revision_a", "publisher_revision_a"), + ("reporting_window", "publisher_release_a", "publisher_release_a"), + ], +) +def test_schema2_period_contract_preserves_valid_aliases_and_opaque_labels( + tmp_path, period_type, reference_period, basis_period +) -> None: + files = _package_with_schema2_targets() + reference = files["target_references.json"]["target_references"][0] + reference["period"] = reference_period + reference["ledger_selector"]["period_type"] = period_type + basis = files["target_references.json"]["target_profile"]["basis_periods"][ + "population_2023" + ] + basis["period"] = basis_period + basis["fact_period_type"] = period_type + package_dir = _write_package(tmp_path, files) + + loaded = load_country_spec(package_dir) + + assert loaded.target_references[0].period == reference_period + + +@pytest.mark.parametrize( + "identifier_field", ["ledger_fact_key", "ledger_source_record_id"] +) +@pytest.mark.parametrize("fact_vintage", ["nis_2025", "nis_2024", None]) +def test_schema2_be_geography_vintage_contract_survives_identifier_resolution( + tmp_path, identifier_field, fact_vintage +) -> None: + package_dir = tmp_path / "be" + shutil.copytree(COUNTRY_PACKAGE_ROOT / "be", package_dir) + target_path = package_dir / "target_references.json" + payload = json.loads(target_path.read_text(encoding="utf-8")) + raw_reference = next( + row + for row in payload["target_references"] + if row["name"] == "statbel_fiscal_income_by_commune" + ) + raw_reference["metadata"]["activation_status"] = "active" + raw_reference["ledger_selector"]["geography_id"] = "21004" + fact_key = "ledger.aggregate_fact.v2:synthetic-be-commune" + record_id = "statbel_fiscal_income.synthetic-commune" + raw_reference[identifier_field] = ( + fact_key if identifier_field == "ledger_fact_key" else record_id + ) + target_path.write_text(json.dumps(payload), encoding="utf-8") + loaded = load_country_spec(package_dir) + reference = next( + row + for row in loaded.target_references + if row.name == "statbel_fiscal_income_by_commune" + ) + fact = { + "aggregate_fact_key": fact_key, + "lineage": {"source_record_id": record_id}, + "value": 10.0, # Synthetic resolver probe, not a Belgian source value. + "period": {"type": "tax_year", "value": 2022}, + "geography": {"level": "commune", "id": "21004"}, + "entity": {"name": "household"}, + "observed_measure": { + "source_name": "statbel_fiscal_income", + "source_measure_id": "taxable_income", + "unit": "eur", + }, + "aggregation": {"method": "sum"}, + } + if fact_vintage is not None: + fact["geography"]["vintage"] = fact_vintage + + if fact_vintage != "nis_2025": + with pytest.raises(ValueError, match="vintage"): + compile_ledger_target_references([fact], [reference], country="be") + else: + (spec,) = compile_ledger_target_references( + [fact], [reference], country="be" + ).specs + assert spec.metadata["ledger_geography_vintage"] == "nis_2025" + assert spec.value == 10.0 diff --git a/packages/microcosm-build/tests/test_ledger_targets.py b/packages/microcosm-build/tests/test_ledger_targets.py index 2f1933ee9..cd46e2cad 100644 --- a/packages/microcosm-build/tests/test_ledger_targets.py +++ b/packages/microcosm-build/tests/test_ledger_targets.py @@ -1,5 +1,5 @@ import json -from dataclasses import asdict +from dataclasses import asdict, replace import pytest @@ -13,8 +13,10 @@ apply_ledger_target_profile, compile_ledger_target_references, ledger_target_registry_parity_report, + period_values_semantically_equal, select_ledger_targets, select_ledger_targets_from_jsonl, + target_spec_from_ledger_reference, ) from microcosm.calibrate import TargetRegistry, TargetSpec @@ -2775,3 +2777,165 @@ def test_ledger_reference_latest_selection_uses_policy_eligible_facts(): ) assert registry.specs[0].value == 15_000_000_000_000 + + +def _target_spec_via_contract_route(fact, reference, resolution_route): + if resolution_route == "direct_helper": + return target_spec_from_ledger_reference(fact, reference) + if resolution_route == "ledger_fact_key": + reference = replace(reference, ledger_fact_key=fact["aggregate_fact_key"]) + elif resolution_route == "ledger_source_record_id": + reference = replace( + reference, + ledger_source_record_id=fact["lineage"]["source_record_id"], + ) + else: + assert resolution_route == "selector" + (spec,) = compile_ledger_target_references([fact], [reference], country="us").specs + return spec + + +@pytest.mark.parametrize( + "resolution_route", + ["selector", "ledger_fact_key", "ledger_source_record_id", "direct_helper"], +) +@pytest.mark.parametrize( + ("period_type", "invalid_period"), + [ + ("academic_year", "2023_25"), + ("academic_year", "2023_2025"), + ("academic_year", "2023_04"), + ("tax_year", "2023_25"), + ("calendar_year", "2023_2025"), + ("fiscal_year", "2023_00"), + ("month", "2023_00"), + ("month", "2023_13"), + ("month", "2023_24"), + ("month", "2023_2024"), + ("month", "2023"), + ], +) +def test_exact_period_contract_refuses_equal_malformed_untyped_labels( + resolution_route, period_type, invalid_period +) -> None: + fact = _consumer_fact_row(period={"type": period_type, "value": invalid_period}) + reference = _exact_agi_reference( + ledger_selector={"period_type": period_type}, + period=invalid_period, + ) + + with pytest.raises(ValueError, match="exact (target )?period"): + _target_spec_via_contract_route(fact, reference, resolution_route) + + +@pytest.mark.parametrize( + "resolution_route", + ["selector", "ledger_fact_key", "ledger_source_record_id", "direct_helper"], +) +@pytest.mark.parametrize( + ("period_type", "reference_period", "fact_period"), + [ + ("tax_year", 2023, "ty2023"), + ("calendar_year", "2023", "calendar_year_2023"), + ("fiscal_year", "2023_24", "fy2023_2024"), + ("academic_year", "2003_04", "ay2003_04"), + ("academic_year", "1999_00", "academic_year_1999_2000"), + ("academic_year", 2023, "ay2023"), + ("month", "2003_04", "month_2003_04"), + ("month", "2023_1", "month_2023_01"), + ("academic_year", "publisher_revision_a", "publisher_revision_a"), + ("reporting_window", "publisher_release_a", "publisher_release_a"), + ], +) +def test_exact_period_contract_preserves_valid_aliases_and_opaque_labels( + resolution_route, period_type, reference_period, fact_period +) -> None: + fact = _consumer_fact_row(period={"type": period_type, "value": fact_period}) + reference = _exact_agi_reference( + ledger_selector={"period_type": period_type}, + period=reference_period, + ) + + spec = _target_spec_via_contract_route(fact, reference, resolution_route) + + assert spec.value == fact["value"] + assert spec.period == reference_period + + +@pytest.mark.parametrize( + "resolution_route", + ["selector", "ledger_fact_key", "ledger_source_record_id", "direct_helper"], +) +@pytest.mark.parametrize("period_match_policy", ["exact", "latest_not_after"]) +@pytest.mark.parametrize("vintage_pin", ["2010_census", ["2010_census", "2000_census"]]) +@pytest.mark.parametrize("fact_vintage", ["2010_census", "2020_census", None, ""]) +def test_geography_vintage_contract_applies_to_every_resolution_route( + resolution_route, period_match_policy, vintage_pin, fact_vintage +) -> None: + fact = _consumer_fact_row() + if fact_vintage is None: + fact["geography"].pop("vintage") + else: + fact["geography"]["vintage"] = fact_vintage + reference = _exact_agi_reference( + ledger_selector={"geography_vintage": vintage_pin}, + period_match_policy=period_match_policy, + ) + + if fact_vintage != "2010_census": + with pytest.raises(ValueError, match="(vintage|fact selector)"): + _target_spec_via_contract_route(fact, reference, resolution_route) + else: + spec = _target_spec_via_contract_route(fact, reference, resolution_route) + assert spec.metadata["ledger_geography_vintage"] == "2010_census" + assert spec.value == fact["value"] + + +@pytest.mark.parametrize( + "resolution_route", + ["selector", "ledger_fact_key", "ledger_source_record_id", "direct_helper"], +) +def test_geography_vintage_contract_does_not_add_an_undeclared_pin( + resolution_route, +) -> None: + fact = _consumer_fact_row() + fact["geography"].pop("vintage") + + spec = _target_spec_via_contract_route( + fact, _exact_agi_reference(), resolution_route + ) + + assert spec.value == fact["value"] + + +@pytest.mark.parametrize( + "invalid_period", ["2023_25", "2023_2025", "2023_00", "2023_13"] +) +def test_period_contract_comparator_rejects_malformed_numeric_literal_equality( + invalid_period, +) -> None: + assert not period_values_semantically_equal(invalid_period, invalid_period) + + +def test_exact_period_contract_sum_keeps_equivalent_untyped_annual_range_cells() -> ( + None +): + first = _consumer_fact_row_for_period(2003, value=10.0) + first["period"] = {"type": "academic_year", "value": "2003_04"} + second = _consumer_fact_row_for_period(2003, value=20.0) + second["aggregate_fact_key"] += "-second" + second["legacy_fact_key"] += "-second" + second["semantic_fact_key"] += "-second" + second["lineage"]["source_record_id"] += ".second" + second["period"] = {"type": "academic_year", "value": "ay2003_04"} + reference = _exact_agi_reference( + ledger_selector={"period_type": "academic_year"}, + period="academic_year_2003_2004", + value_operation="sum", + ) + + (spec,) = compile_ledger_target_references( + [first, second], [reference], country="us" + ).specs + + assert spec.value == 30.0 # Both synthetic cells belong to the same academic year.