diff --git a/packages/microcosm-build/src/microcosm/build/uk/target_references.json b/packages/microcosm-build/src/microcosm/build/uk/target_references.json index d7cdb07f0..114952a29 100644 --- a/packages/microcosm-build/src/microcosm/build/uk/target_references.json +++ b/packages/microcosm-build/src/microcosm/build/uk/target_references.json @@ -22,6 +22,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.income_tax", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -40,6 +41,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.ni", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -58,6 +60,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.ni_employee", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -76,6 +79,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.ni_employer", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -94,6 +98,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.ni_self_employed", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -112,6 +117,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.vat", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -130,6 +136,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.fuel_duties", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -148,6 +155,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.capital_gains_tax", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -166,6 +174,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.sdlt", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -184,6 +193,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.attendance_allowance", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -202,6 +212,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.carers_allowance", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -220,6 +231,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.child_benefit", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -238,6 +250,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.council_tax", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -256,6 +269,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.esa", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection", @@ -275,6 +289,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.housing_benefit", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -293,6 +308,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.jobseekers_allowance", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -311,6 +327,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.pension_credit", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -329,6 +346,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.pip", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -347,6 +365,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.state_pension", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -365,6 +384,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.statutory_maternity_pay", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -383,6 +403,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.tv_licence_fee", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -401,6 +422,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.universal_credit_in_cap", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -419,6 +441,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.universal_credit_outside_cap", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -437,6 +460,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.winter_fuel_allowance", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/diagnostics.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/diagnostics.py index 7907b4d71..befc156eb 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/diagnostics.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/diagnostics.py @@ -1,12 +1,11 @@ """Standard calibration diagnostics for UK release candidates. The shared :mod:`microcosm.calibrate.diagnostics` payload is the release -contract: it carries the target surface, every target row, solver options, and -the concentration scalars used by US releases. UK needs a little more release -evidence without changing that shared schema (and therefore without changing -US output): the effective-sample-size fraction, shipped-weight concentration, -zero-weight rows split by their declared support strata, and target fit by UK -geography level. +contract: it carries the target surface, every target row, solver options, +schema-7 source/variable/dimension identity, and the concentration scalars used +by US releases. UK adds the effective-sample-size fraction, shipped-weight +concentration, zero-weight rows split by their declared support strata, and +target fit by UK geography level. This module wraps the shared payload and places those additions under a separately versioned ``uk_diagnostics`` block. The common top-level diff --git a/packages/microcosm-build/tests/test_uk_diagnostics.py b/packages/microcosm-build/tests/test_uk_diagnostics.py index 5eb8e8f01..6c7145b6f 100644 --- a/packages/microcosm-build/tests/test_uk_diagnostics.py +++ b/packages/microcosm-build/tests/test_uk_diagnostics.py @@ -577,7 +577,7 @@ def test_payload_requires_a_valid_matching_uk_registry() -> None: target_geography_levels=geography, target_registry=TargetRegistry((), country="uk"), ) - with pytest.raises(ValueError, match="exactly partition"): + with pytest.raises(ValueError, match="does not contain compiled target row"): uk_calibration_diagnostics_payload( result, frame, diff --git a/packages/microcosm-build/tests/test_uk_rowwise_candidate.py b/packages/microcosm-build/tests/test_uk_rowwise_candidate.py index 2d5a8933b..5de3268bb 100644 --- a/packages/microcosm-build/tests/test_uk_rowwise_candidate.py +++ b/packages/microcosm-build/tests/test_uk_rowwise_candidate.py @@ -399,7 +399,7 @@ def test_candidate_build_writes_calibrated_h5_and_evidence( assert diagnostics["metric"].unique().tolist() == ["households"] assert len(support) == 8 assert past_cap["n_targets"] == 4 - assert calibration_diagnostics["schema_version"] == 6 + assert calibration_diagnostics["schema_version"] == 7 uk_diagnostics = calibration_diagnostics["uk_diagnostics"] assert len(uk_diagnostics["weakest_families"]) == 1 assert len(uk_diagnostics["weakest_areas_by_fit"]["bottom_by_fit"]) == 4 diff --git a/packages/microcosm-build/tests/test_uk_target_references.py b/packages/microcosm-build/tests/test_uk_target_references.py index d9ee9d216..7f3bef780 100644 --- a/packages/microcosm-build/tests/test_uk_target_references.py +++ b/packages/microcosm-build/tests/test_uk_target_references.py @@ -24,9 +24,12 @@ TargetReferenceAuthoringConfig, author_target_references, ) +from microcosm.calibrate.diagnostics import diagnostics_payload from microcosm.calibrate.matrix import build_constraint_matrix +from microcosm.calibrate.score import score_targets from microcosm.frame import EntitySchema, Frame, WeightKind, Weights from tools.generate_uk_target_references import ( + OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID, POLICYENGINE_BINDING_KEYS, _annual_uc_award_band_token, _geography_pins, @@ -128,10 +131,16 @@ def test_uk_target_references_follow_contract_derivation_rules() -> None: assert reference["measure"] == expected_measure assert reference["family"] == target["family"] assert reference["period"] == 2025 - assert reference["metadata"] == { + expected_metadata = { "contract_target_id": contract_target_id, "measure_kind": "prepared_column", } + diagnostic_variable_id = OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID.get( + contract_target_id + ) + if diagnostic_variable_id is not None: + expected_metadata["diagnostic_variable_id"] = diagnostic_variable_id + assert reference["metadata"] == expected_metadata # The measure is a prepared column, so the pointed-to contract binding # must carry what the microcosm#622 materializer needs to prepare it. assert ( @@ -152,6 +161,29 @@ def test_uk_target_references_do_not_bind_known_mismatched_property_amounts() -> ] +def test_uk_obr_references_declare_receipts_and_expenditure_categories() -> None: + references = _load_uk_resource("target_references.json")["target_references"] + obr_references = [ + reference + for reference in references + if reference["metadata"]["contract_target_id"].startswith("obr.") + ] + + assert len(obr_references) == len(OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID) + assert { + reference["metadata"]["contract_target_id"]: reference["metadata"][ + "diagnostic_variable_id" + ] + for reference in obr_references + } == OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID + assert not [ + reference["name"] + for reference in references + if not reference["metadata"]["contract_target_id"].startswith("obr.") + and "diagnostic_variable_id" in reference["metadata"] + ] + + def test_uk_target_references_do_not_emit_nan_uc_payment_bands() -> None: resource = _load_uk_resource("target_references.json") @@ -418,6 +450,7 @@ def test_uk_target_references_compile_from_real_staged_feed_rows() -> None: assert income_tax.period == 2025 assert income_tax.metadata["ledger_assertion"] == "source_projection" assert income_tax.metadata["ledger_assertion_policy"] == ("allow_source_projection") + assert income_tax.metadata["diagnostic_variable_id"] == "efo_receipts" tcl_households = targets["dwp.uc.two_child_limit.households_affected"] assert tcl_households.value == 469_780 @@ -558,6 +591,20 @@ def test_uk_target_references_constrain_a_frame_with_prepared_columns() -> None: for name, target_value in zip(problem.names, problem.target_vector, strict=True): assert target_value == fact_values_by_name[name], name + diagnostics = diagnostics_payload( + score_targets(frame, registry.to_target_set()), + target_registry=registry, + ) + obr_variables = { + row["name"]: row["variable"]["id"] + for row in diagnostics["targets"] + if row["source"]["id"] == "obr" + } + assert obr_variables == { + "obr.esa@2025": "efo_expenditure", + "obr.income_tax@2025": "efo_receipts", + } + def _real_uk_consumer_fact_rows() -> list[dict]: """Public rows copied from .codex-work/consumer_facts_uk.jsonl.""" diff --git a/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py b/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py index 0794037d1..11fa4beb4 100644 --- a/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py +++ b/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py @@ -1743,20 +1743,16 @@ def selected(path, *, allow_terminal_gate_failure): ) return - loaded_frame, receipt, loaded_identity = ( - builder._load_base_pool_if_identified( - pool_h5, - allow_gate_failed_base_pool=allow_gate_failed, - ) + loaded_frame, receipt, loaded_identity = builder._load_base_pool_if_identified( + pool_h5, + allow_gate_failed_base_pool=allow_gate_failed, ) assert loaded_frame is frame assert loaded_identity is authenticated assert receipt["status"] == status assert receipt["allow_gate_failed_base_pool"] is allow_gate_failed - assert receipt["agreement_gate_reference"]["failure_count"] == len( - gate_failures - ) + assert receipt["agreement_gate_reference"]["failure_count"] == len(gate_failures) def test_builder_refuses_actual_red_base_h5_pool_sidecar_without_opt_in( @@ -1777,7 +1773,9 @@ def test_builder_refuses_actual_red_base_h5_pool_sidecar_without_opt_in( ) out = tmp_path / "out" monkeypatch.setattr(builder, "_git_dirty", lambda: False) - monkeypatch.setattr(builder, "_refuse_certified_release_dir_reuse", lambda path: None) + monkeypatch.setattr( + builder, "_refuse_certified_release_dir_reuse", lambda path: None + ) monkeypatch.setattr( builder, "_load_frame", @@ -1826,7 +1824,9 @@ def test_builder_refuses_bare_stamped_pool_h5_before_generic_load( ) out = tmp_path / "out" monkeypatch.setattr(builder, "_git_dirty", lambda: False) - monkeypatch.setattr(builder, "_refuse_certified_release_dir_reuse", lambda path: None) + monkeypatch.setattr( + builder, "_refuse_certified_release_dir_reuse", lambda path: None + ) monkeypatch.setattr( builder, "_load_frame", @@ -4160,6 +4160,13 @@ def test_release_calibration_diagnostics_writes_nan_final_loss_as_null( measure="income", value=500_000.0, source="fixture", + metadata={ + "ledger_selector_source_name": "irs_soi", + "ledger_measure_concept": "irs_soi.income", + "ledger_measure_unit": "usd", + "ledger_geography_level": "state", + "ledger_geography_id": "0400000US06", + }, ), ), country="us", @@ -4197,6 +4204,19 @@ def test_release_calibration_diagnostics_writes_nan_final_loss_as_null( ) diagnostics = json.loads((tmp_path / "calibration_diagnostics.json").read_text()) + assert diagnostics["schema_version"] == 7 + assert diagnostics["targets"][0]["source"] == { + "id": "irs_soi", + "citation": "fixture", + } + assert diagnostics["targets"][0]["dimensions"] == {"geography_state": "0400000US06"} + assert diagnostics["dimensions"]["geography_state"] == { + "label": "State", + "role": "geography", + "level": "state", + "values": {"0400000US06": "CA"}, + "order": ["0400000US06"], + } assert diagnostics["final_loss"] is None assert diagnostics["build"]["default_dataset"]["final_loss"] is None @@ -9160,9 +9180,7 @@ def test_exact_k_receipt_stays_strict_even_when_base_h5_opt_in_is_present() -> N builder = _load_builder_module() with pytest.raises(RuntimeError, match="lost its passing agreement gate"): - builder._exact_k_ladder_manifest_payload( - **_gate_failed_exact_k_inputs(builder) - ) + builder._exact_k_ladder_manifest_payload(**_gate_failed_exact_k_inputs(builder)) def _gate_failed_base_pool_receipt() -> dict[str, object]: diff --git a/packages/microcosm-calibrate/README.md b/packages/microcosm-calibrate/README.md index 233f24847..30b0f9ae0 100644 --- a/packages/microcosm-calibrate/README.md +++ b/packages/microcosm-calibrate/README.md @@ -76,6 +76,27 @@ frontier can be read off any run's artifact. The standalone `effective_sample_size(weights)` scores any weight vector, e.g. a published artifact's. +Registry-backed diagnostics use schema 7. Each target publishes structured +`source`, `variable`, and `dimensions` objects, and the artifact publishes a +top-level dimension dictionary. Ledger geography metadata becomes one typed +geography dimension per level (for example, `geography_country` or +`geography_state`), with stable geography identifiers, producer-owned labels, +and deterministic value order. Ledger filter and layout dimensions remain +separate non-geographic dimensions. This applies to every country release that +passes its `TargetRegistry`, including the UK and US release builders. Calls +without a registry retain legacy target identity fields because they do not +provide enough declared information to construct structured identities. + +Schema 7 also separates the statistic category from its measurement. For +legacy Ledger concepts whose declared unit agrees with a trailing `_count` or +`_amount`, the suffix is represented as `variable.measure` (`count` or `total`) +instead of remaining in `variable.id`. For example, +`hmrc.spi_employment_income_count` and +`hmrc.spi_employment_income_amount` both use the variable identifier +`spi_employment_income`; their measure values remain distinct. An explicit +`diagnostic_variable_id` or `variable` metadata value always takes precedence +and is not rewritten. + ## Example ```python diff --git a/packages/microcosm-calibrate/src/microcosm/calibrate/diagnostics.py b/packages/microcosm-calibrate/src/microcosm/calibrate/diagnostics.py index f63913d68..88f23f808 100644 --- a/packages/microcosm-calibrate/src/microcosm/calibrate/diagnostics.py +++ b/packages/microcosm-calibrate/src/microcosm/calibrate/diagnostics.py @@ -21,6 +21,7 @@ import json import logging import math +import re from collections.abc import Mapping from pathlib import Path from typing import Any @@ -49,10 +50,89 @@ #: v6 added authoritative final per-target loss attribution and an explicit #: warning-only degradation state when that supplementary attribution cannot #: be validated. -CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION = 6 +#: v7 adds producer-defined source, variable, and dimension identity for +#: registry-backed release diagnostics. Geography is represented as a typed +#: dimension with stable identifiers and display labels. +CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION = 7 _LOGGER = logging.getLogger(__name__) +_UK_GEOGRAPHY_LABELS = { + "K02000001": "United Kingdom", + "K03000001": "Great Britain", + "E92000001": "England", + "W92000004": "Wales", + "S92000003": "Scotland", + "N92000002": "Northern Ireland", +} + +_US_STATE_POSTAL = { + "01": "AL", + "02": "AK", + "04": "AZ", + "05": "AR", + "06": "CA", + "08": "CO", + "09": "CT", + "10": "DE", + "11": "DC", + "12": "FL", + "13": "GA", + "15": "HI", + "16": "ID", + "17": "IL", + "18": "IN", + "19": "IA", + "20": "KS", + "21": "KY", + "22": "LA", + "23": "ME", + "24": "MD", + "25": "MA", + "26": "MI", + "27": "MN", + "28": "MS", + "29": "MO", + "30": "MT", + "31": "NE", + "32": "NV", + "33": "NH", + "34": "NJ", + "35": "NM", + "36": "NY", + "37": "NC", + "38": "ND", + "39": "OH", + "40": "OK", + "41": "OR", + "42": "PA", + "44": "RI", + "45": "SC", + "46": "SD", + "47": "TN", + "48": "TX", + "49": "UT", + "50": "VT", + "51": "VA", + "53": "WA", + "54": "WV", + "55": "WI", + "56": "WY", +} + +_COUNT_UNITS = frozenset( + { + "count", + "households", + "people", + "persons", + "returns", + "claims", + } +) +_TOTAL_UNITS = frozenset({"gbp", "usd", "dollars", "pounds"}) +_MEAN_UNITS = frozenset({"percent", "percentage", "rate", "ratio"}) + def _finite(value: float) -> float | None: """JSON has no NaN/inf; a non-finite diagnostic serializes as null.""" @@ -120,6 +200,316 @@ def _registry_spec_lookup(target_registry: object | None) -> dict[str, object]: return lookup +def _metadata_string(metadata: Mapping[str, object], key: str) -> str: + """Return a stripped metadata string, or an empty string.""" + + value = metadata.get(key) + return value.strip() if isinstance(value, str) else "" + + +def _source_id(spec: object, metadata: Mapping[str, object]) -> str: + """Return the publisher identifier declared by a registry-backed target.""" + + explicit = _metadata_string(metadata, "diagnostic_source_id") + if explicit: + return explicit + selector_source = _metadata_string(metadata, "ledger_selector_source_name") + if selector_source: + return selector_source + source_record = _metadata_string(metadata, "ledger_source_record_id") + if source_record: + return source_record.split(".", 1)[0] + family = str(getattr(spec, "family", "")).strip() + if family: + return family + name = str(getattr(spec, "name", "")).strip() + return re.split(r"[./]", name, maxsplit=1)[0] or "other" + + +def _variable_id( + spec: object, + metadata: Mapping[str, object], + *, + source_id: str, +) -> str: + """Return a stable statistic identifier for a registry-backed target. + + Explicit diagnostic identifiers are already producer declarations and are + preserved verbatim. Ledger measure concepts are older compound identifiers: + some append ``_count`` or ``_amount`` even though schema 7 represents that + distinction separately in ``variable.measure``. Remove only the suffix that + agrees with the declared unit so count and amount rows remain one dashboard + category without conflating their measurements. + """ + + def without_source_prefix(value: str) -> str: + for prefix in (f"{source_id}.", f"{source_id}:"): + if value.startswith(prefix): + return value[len(prefix) :] + return value + + for key in ("diagnostic_variable_id", "variable"): + value = _metadata_string(metadata, key) + if value: + return without_source_prefix(value) + + measure = _variable_measure(metadata) + measure_suffix = {"count": "_count", "total": "_amount"}.get(measure, "") + for key in ("ledger_measure_concept", "ledger_source_concept"): + value = _metadata_string(metadata, key) + if value: + identifier = without_source_prefix(value) + if measure_suffix and identifier.endswith(measure_suffix): + identifier = identifier[: -len(measure_suffix)] + return identifier + contract_id = _metadata_string(metadata, "contract_target_id") + if contract_id: + prefix = re.split(r"[./]", contract_id, maxsplit=1)[0] + remainder = contract_id[len(prefix) :].lstrip("./") + return remainder or contract_id + measure = str(getattr(spec, "measure", "")).strip() + return measure or str(getattr(spec, "name", "")).strip() or "unknown" + + +def _variable_measure(metadata: Mapping[str, object]) -> str: + """Classify a declared Ledger unit into the dashboard's measure vocabulary.""" + + unit = _metadata_string(metadata, "ledger_measure_unit").lower() + if unit in _COUNT_UNITS: + return "count" + if unit in _TOTAL_UNITS: + return "total" + if unit in _MEAN_UNITS: + return "mean" + return "" + + +def _humanize_identifier(value: str) -> str: + """Turn a machine identifier into a concise dimension label.""" + + tail = value.rsplit("#", 1)[-1].rsplit(".", 1)[-1] + return " ".join(part.capitalize() for part in re.split(r"[_:/-]+", tail) if part) + + +def _dimension_label(dimension_id: str) -> str: + """Return the established display label for a Ledger dimension id.""" + + if dimension_id == "us:statutes/26/62#adjusted_gross_income": + return "Income Band" + if dimension_id == "census_stc.item": + return "Item" + if dimension_id == "hhs_acf_tanf.spending_category": + return "Spending Category" + if dimension_id == "income_range": + return "Income Band" + if dimension_id == "filing_status": + return "Filing Status" + if dimension_id == "eitc_child_count": + return "Qualifying Children" + return _humanize_identifier(dimension_id) + + +def _normalized_geography_level(value: str) -> str: + """Normalize producer aliases used by the existing country contracts.""" + + normalized = value.strip().lower().replace("-", "_").replace(" ", "_") + return "local_authority" if normalized == "la" else normalized + + +def _geography_label( + *, + country: str, + level: str, + geography_id: str, + metadata: Mapping[str, object], +) -> str: + """Resolve a producer-owned geography label without consumer name parsing.""" + + explicit = _metadata_string(metadata, "ledger_geography_name") + if explicit: + return explicit + if country == "uk": + return _UK_GEOGRAPHY_LABELS.get(geography_id, geography_id) + if country == "us": + if geography_id == "0100000US" or level in {"country", "national"}: + return "United States" + match = re.search(r"US(\d{2})(\d{2})$", geography_id) + if level == "congressional_district" and match: + postal = _US_STATE_POSTAL.get(match.group(1)) + if postal: + return f"{postal}-{match.group(2)}" + match = re.search(r"US(\d{2})$", geography_id) + if level == "state" and match: + return _US_STATE_POSTAL.get(match.group(1), geography_id) + return geography_id + + +def _structured_dimensions( + metadata: Mapping[str, object], + *, + country: str, +) -> tuple[dict[str, str], dict[str, dict[str, object]]]: + """Build schema-7 row dimensions and their producer definitions.""" + + values: dict[str, str] = {} + definitions: dict[str, dict[str, object]] = {} + + geography_id = _metadata_string(metadata, "ledger_geography_id") + geography_level = _normalized_geography_level( + _metadata_string(metadata, "ledger_geography_level") + ) + if geography_id and geography_level: + dimension_id = f"geography_{geography_level}" + values[dimension_id] = geography_id + definitions[dimension_id] = { + "label": _humanize_identifier(geography_level), + "role": "geography", + "level": geography_level, + "values": { + geography_id: _geography_label( + country=country, + level=geography_level, + geography_id=geography_id, + metadata=metadata, + ) + }, + "order": [geography_id], + } + + filter_dimensions = [ + (key.removeprefix("ledger_filter_"), raw_value.strip()) + for key, raw_value in metadata.items() + if key.startswith("ledger_filter_") + and isinstance(raw_value, str) + and key.removeprefix("ledger_filter_") + and raw_value.strip() + ] + layout_dimension = _metadata_string(metadata, "ledger_layout_groupby_dimension") + layout_value = _metadata_string(metadata, "ledger_layout_groupby_value_id") + layout_label = _dimension_label(layout_dimension) + duplicate_filter = any( + _dimension_label(dimension_id) == layout_label and value == layout_value + for dimension_id, value in filter_dimensions + ) + resolved_geography_label = ( + _geography_label( + country=country, + level=geography_level, + geography_id=geography_id, + metadata=metadata, + ) + if geography_id and geography_level + else "" + ) + geography_layout = layout_dimension in { + "geography", + "state", + "cms_medicaid.state_abbreviation", + } + redundant_geography = layout_value.lower() in { + geography_id.lower(), + resolved_geography_label.lower(), + } + if ( + layout_dimension + and layout_value + and not duplicate_filter + and not geography_layout + and not redundant_geography + ): + values[layout_dimension] = layout_value + definitions[layout_dimension] = { + "label": layout_label, + "values": {layout_value: _humanize_identifier(layout_value)}, + "order": [layout_value], + } + + for dimension_id, value in filter_dimensions: + values[dimension_id] = value + definitions[dimension_id] = { + "label": _dimension_label(dimension_id), + "values": {value: _humanize_identifier(value)}, + "order": [value], + } + return values, definitions + + +def _merge_dimension_definitions( + destination: dict[str, dict[str, object]], + additions: Mapping[str, Mapping[str, object]], +) -> None: + """Merge per-row dimension declarations into one deterministic dictionary.""" + + for dimension_id, addition in additions.items(): + current = destination.get(dimension_id) + if current is None: + destination[dimension_id] = { + **addition, + "values": dict(addition.get("values", {})), + "order": list(addition.get("order", [])), + } + continue + for key in ("label", "role", "level"): + incoming = addition.get(key) + if incoming is not None and current.get(key) != incoming: + raise ValueError( + f"Diagnostics dimension {dimension_id!r} has conflicting " + f"{key} declarations {current.get(key)!r} and {incoming!r}." + ) + current_values = current.setdefault("values", {}) + current_order = current.setdefault("order", []) + if not isinstance(current_values, dict) or not isinstance(current_order, list): + raise TypeError("Diagnostics dimension aggregation state is malformed.") + for raw_value, label in addition.get("values", {}).items(): + existing = current_values.get(raw_value) + if existing is not None and existing != label: + raise ValueError( + f"Diagnostics dimension {dimension_id!r} value " + f"{raw_value!r} has conflicting labels {existing!r} and {label!r}." + ) + current_values[raw_value] = label + for raw_value in addition.get("order", []): + if raw_value not in current_order: + current_order.append(raw_value) + + +def _structured_target_fields( + target: object, + spec: object, + *, + country: str, +) -> tuple[dict[str, object], dict[str, dict[str, object]]]: + """Serialize complete schema-7 identity for one registry-backed target.""" + + metadata = dict(getattr(spec, "metadata", {}) or {}) + source_id = _source_id(spec, metadata) + variable_id = _variable_id(spec, metadata, source_id=source_id) + dimensions, definitions = _structured_dimensions(metadata, country=country) + citation = str(getattr(target, "source", "")).strip() + source: dict[str, str] = {"id": source_id} + if citation: + source["citation"] = citation + source_url = next( + ( + part.strip() + for part in citation.split("|") + if part.strip().startswith(("https://", "http://")) + ), + "", + ) + if source_url: + source["url"] = source_url + variable: dict[str, str] = {"id": variable_id} + measure = _variable_measure(metadata) + if measure: + variable["measure"] = measure + return { + "source": source, + "variable": variable, + "dimensions": dimensions, + }, definitions + + def _target_identity_rows(result: CalibrationResult) -> list[dict[str, object]]: """The target surface as structured rows suitable for hashing.""" rows: list[dict[str, object]] = [] @@ -328,17 +718,33 @@ def diagnostics_payload( floats become ``null``). """ registry_specs = _registry_spec_lookup(target_registry) - target_rows = [ - _target_row( + registry_country = str(getattr(target_registry, "country", "")).strip() + dimension_definitions: dict[str, dict[str, object]] = {} + target_rows: list[dict[str, object]] = [] + for index, (diagnostic, target) in enumerate( + zip(result.diagnostics, result.problem.targets, strict=True) + ): + spec = registry_specs.get(diagnostic.name) + if target_registry is not None and spec is None: + raise ValueError( + "The supplied target registry does not contain compiled target " + f"row {diagnostic.name!r}." + ) + row = _target_row( diagnostic, target, compiled_target=result.problem.target_vector[index], - spec=registry_specs.get(diagnostic.name), - ) - for index, (diagnostic, target) in enumerate( - zip(result.diagnostics, result.problem.targets, strict=True) + spec=spec, ) - ] + if spec is not None: + structured_fields, row_definitions = _structured_target_fields( + target, + spec, + country=registry_country, + ) + row.update(structured_fields) + _merge_dimension_definitions(dimension_definitions, row_definitions) + target_rows.append(row) payload = { "schema_version": CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION, "weight_entity": result.weight_entity, @@ -361,6 +767,8 @@ def diagnostics_payload( "diagnostic_warnings": [], "targets": target_rows, } + if target_registry is not None: + payload["dimensions"] = dimension_definitions try: attribution = assemble_target_loss_attribution(result) except TargetLossAttributionError as error: diff --git a/packages/microcosm-calibrate/tests/test_diagnostics.py b/packages/microcosm-calibrate/tests/test_diagnostics.py index e7e683285..325be9a61 100644 --- a/packages/microcosm-calibrate/tests/test_diagnostics.py +++ b/packages/microcosm-calibrate/tests/test_diagnostics.py @@ -115,7 +115,7 @@ def test_payload_reports_complete_uniform_final_loss_attribution( result = _result(feasible_frame, epochs=1) payload = diagnostics_payload(result) - assert payload["schema_version"] == 6 + assert payload["schema_version"] == 7 assert payload["diagnostic_warnings"] == [] basis = payload["target_loss_basis"] assert basis["formula"] == ( @@ -437,6 +437,16 @@ def test_payload_can_carry_target_registry_identity(feasible_frame) -> None: period=2024, source="IRS SOI 2024", family="irs_soi", + metadata={ + "ledger_selector_source_name": "irs_soi", + "ledger_measure_concept": "irs_soi.adjusted_gross_income", + "ledger_measure_unit": "usd", + "ledger_geography_level": "state", + "ledger_geography_id": "0400000US06", + "ledger_layout_groupby_dimension": "filing_status", + "ledger_layout_groupby_value_id": "single", + "ledger_filter_filing_status": "single", + }, ), ), country="us", @@ -451,10 +461,172 @@ def test_payload_can_carry_target_registry_identity(feasible_frame) -> None: } income = next(row for row in payload["targets"] if row["target_name"] == "income") assert income["period"] == 2024 - assert income["source"] == "IRS SOI 2024" + assert income["source"] == { + "id": "irs_soi", + "citation": "IRS SOI 2024", + } + assert income["variable"] == { + "id": "adjusted_gross_income", + "measure": "total", + } + assert income["dimensions"] == { + "geography_state": "0400000US06", + "filing_status": "single", + } + assert payload["dimensions"] == { + "geography_state": { + "label": "State", + "role": "geography", + "level": "state", + "values": {"0400000US06": "CA"}, + "order": ["0400000US06"], + }, + "filing_status": { + "label": "Filing Status", + "values": {"single": "Single"}, + "order": ["single"], + }, + } assert income["registry"]["family"] == "irs_soi" +def test_registry_diagnostics_publish_uk_geography_and_all_ledger_dimensions( + feasible_frame, +) -> None: + frame, truths = feasible_frame() + geographies = { + "K02000001": "United Kingdom", + "K03000001": "Great Britain", + "E92000001": "England", + "S92000003": "Scotland", + } + registry = TargetRegistry( + tuple( + TargetSpec( + name=f"population_female_{geography_id}", + entity="household", + measure="household_count", + value=truths["population"], + period=2025, + source="ons | Population table | https://example.test/ons", + family="ons_population", + metadata={ + "ledger_selector_source_name": "ons", + "ledger_measure_concept": "ons.population", + "ledger_measure_unit": "count", + "ledger_geography_level": "country", + "ledger_geography_id": geography_id, + "ledger_layout_groupby_dimension": "age_band", + "ledger_layout_groupby_value_id": "18_64", + "ledger_filter_sex": "female", + }, + ) + for geography_id in geographies + ), + country="uk", + ) + result = score_targets(frame, registry.to_target_set()) + + payload = diagnostics_payload(result, target_registry=registry) + + assert payload["targets"][2]["source"] == { + "id": "ons", + "citation": "ons | Population table | https://example.test/ons", + "url": "https://example.test/ons", + } + assert payload["targets"][2]["variable"] == { + "id": "population", + "measure": "count", + } + assert payload["targets"][2]["dimensions"] == { + "geography_country": "E92000001", + "age_band": "18_64", + "sex": "female", + } + assert payload["dimensions"]["geography_country"] == { + "label": "Country", + "role": "geography", + "level": "country", + "values": geographies, + "order": list(geographies), + } + assert payload["dimensions"]["age_band"]["values"] == {"18_64": "18 64"} + assert payload["dimensions"]["sex"]["values"] == {"female": "Female"} + + +def test_registry_diagnostics_separate_measure_from_variable_category( + feasible_frame, +) -> None: + frame, truths = feasible_frame() + registry = TargetRegistry( + ( + TargetSpec( + name="employment_income_amount_band_100000", + entity="household", + measure="income", + value=truths["income"], + period=2025, + source="HMRC SPI", + family="hmrc", + metadata={ + "ledger_selector_source_name": "hmrc", + "ledger_measure_concept": "hmrc.spi_employment_income_amount", + "ledger_measure_unit": "gbp", + "ledger_filter_total_income_lower_bound": "100000", + }, + ), + TargetSpec( + name="employment_income_count_band_100000", + entity="household", + measure="household_count", + value=truths["population"], + period=2025, + source="HMRC SPI", + family="hmrc", + metadata={ + "ledger_selector_source_name": "hmrc", + "ledger_measure_concept": "hmrc.spi_employment_income_count", + "ledger_measure_unit": "count", + "ledger_filter_total_income_lower_bound": "100000", + }, + ), + ), + country="uk", + ) + result = score_targets(frame, registry.to_target_set()) + + payload = diagnostics_payload(result, target_registry=registry) + + variables = [row["variable"] for row in payload["targets"]] + assert variables == [ + {"id": "spi_employment_income", "measure": "total"}, + {"id": "spi_employment_income", "measure": "count"}, + ] + assert payload["targets"][0]["dimensions"] == {"total_income_lower_bound": "100000"} + assert payload["targets"][1]["dimensions"] == {"total_income_lower_bound": "100000"} + + +def test_registry_diagnostics_reject_missing_compiled_target_identity( + feasible_frame, +) -> None: + result = _result(feasible_frame, epochs=1) + registry = TargetRegistry( + ( + TargetSpec( + name="different", + entity="household", + measure="household_count", + value=1.0, + source="Fixture", + ), + ), + country="us", + ) + + with pytest.raises(ValueError, match="does not contain compiled target row"): + diagnostics_payload(result, target_registry=registry) + + def test_payload_reports_weight_concentration(feasible_frame) -> None: """The accuracy-vs-spread coordinates ship with every calibration.""" result = _result(feasible_frame) diff --git a/packages/microcosm-data/src/microcosm/data/contract.py b/packages/microcosm-data/src/microcosm/data/contract.py index 91376f846..da8e87c44 100644 --- a/packages/microcosm-data/src/microcosm/data/contract.py +++ b/packages/microcosm-data/src/microcosm/data/contract.py @@ -125,11 +125,13 @@ ) # Lockstep with microcosm.calibrate.diagnostics.CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION -# (schema 6 = final per-target loss attribution plus warning-only degradation). +# (schema 7 = structured source, variable, and dimension identity for +# registry-backed release diagnostics). # microcosm-data cannot import # microcosm-calibrate (dependency direction), so the builder test suite pins the # two constants equal — see test_calibration_diagnostics_schema_lockstep. -CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION = 6 +CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION = 7 +_SUPPORTED_CALIBRATION_DIAGNOSTICS_SCHEMA_VERSIONS = frozenset({6, 7}) US_SOURCE_COVERAGE_DIAGNOSTICS_FILE = "us_source_coverage.json" SOURCE_COVERAGE_DIAGNOSTICS_SCHEMA_VERSION = 1 _SHA256_RE = re.compile(r"^[0-9a-f]{64}$") @@ -3165,14 +3167,16 @@ def _check_calibration_diagnostics( "calibration_diagnostics.json grandfathered June UK release " f"requires legacy schema version 2, got {schema_version!r}." ) - elif ( - not grandfathered_uk_june - and schema_version != CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION + elif not grandfathered_uk_june and ( + isinstance(schema_version, bool) + or not isinstance(schema_version, int) + or schema_version not in _SUPPORTED_CALIBRATION_DIAGNOSTICS_SCHEMA_VERSIONS ): failures.append( f"calibration_diagnostics.json 'schema_version' is {schema_version!r}; " - f"this library publishes version " - f"{CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION}." + "supported versions are " + f"{sorted(_SUPPORTED_CALIBRATION_DIAGNOSTICS_SCHEMA_VERSIONS)} and " + f"this library publishes version {CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION}." ) expected_sections = { @@ -3205,6 +3209,15 @@ def _check_calibration_diagnostics( ) targets = diagnostics.get("targets") + dimension_definitions = diagnostics.get("dimensions") + if schema_version == 7 and not isinstance(dimension_definitions, Mapping): + failures.append( + "calibration_diagnostics.json schema 7 requires a top-level " + "'dimensions' object." + ) + dimension_definitions = {} + if schema_version == 7 and isinstance(dimension_definitions, Mapping): + _check_diagnostics_dimension_definitions(dimension_definitions, failures) if isinstance(targets, list): surface = diagnostics.get("target_surface") if isinstance(surface, Mapping) and surface.get("n_targets") != len(targets): @@ -3241,6 +3254,17 @@ def _check_calibration_diagnostics( "calibration_diagnostics.json target row " f"{index} is missing non-empty 'source'." ) + if schema_version == 7: + _check_structured_diagnostics_target( + target, + index=index, + dimension_definitions=( + dimension_definitions + if isinstance(dimension_definitions, Mapping) + else {} + ), + failures=failures, + ) if not grandfathered_uk_june: if not isinstance(target.get("measure"), Mapping): failures.append( @@ -3260,6 +3284,115 @@ def _check_calibration_diagnostics( ) +def _check_diagnostics_dimension_definitions( + dimensions: Mapping, + failures: list[str], +) -> None: + """Validate the schema-7 dimension dictionary.""" + + for dimension_id, definition in dimensions.items(): + owner = f"calibration_diagnostics.json dimension {dimension_id!r}" + if not isinstance(dimension_id, str) or not dimension_id: + failures.append( + "calibration_diagnostics.json dimension ids must be non-empty strings." + ) + continue + if not isinstance(definition, Mapping): + failures.append(f"{owner} must be an object.") + continue + if ( + not isinstance(definition.get("label"), str) + or not str(definition.get("label")).strip() + ): + failures.append(f"{owner} requires a non-empty string 'label'.") + role = definition.get("role") + if role not in {None, "geography"}: + failures.append(f"{owner} has unsupported role {role!r}.") + if role == "geography" and ( + not isinstance(definition.get("level"), str) + or not str(definition.get("level")).strip() + ): + failures.append( + f"{owner} with role 'geography' requires a non-empty string 'level'." + ) + value_labels = definition.get("values") + if value_labels is not None and not isinstance(value_labels, Mapping): + failures.append(f"{owner} 'values' must be an object when provided.") + elif isinstance(value_labels, Mapping): + for raw_value, label in value_labels.items(): + if ( + not isinstance(raw_value, str) + or not raw_value + or not isinstance(label, str) + or not label.strip() + ): + failures.append( + f"{owner} value labels must map non-empty strings to " + "non-empty strings." + ) + break + order = definition.get("order") + if order is not None and ( + not isinstance(order, list) + or any(not isinstance(value, str) or not value for value in order) + or len(set(order)) != len(order) + ): + failures.append( + f"{owner} 'order' must be a list of unique non-empty strings." + ) + + +def _check_structured_diagnostics_target( + target: Mapping, + *, + index: int, + dimension_definitions: Mapping, + failures: list[str], +) -> None: + """Validate complete structured identity on one schema-7 target row.""" + + owner = f"calibration_diagnostics.json target row {index}" + for field in ("source", "variable"): + value = target.get(field) + if ( + not isinstance(value, Mapping) + or not isinstance(value.get("id"), str) + or not str(value.get("id")).strip() + ): + failures.append( + f"{owner} schema 7 requires {field!r} to be an object with a " + "non-empty string 'id'." + ) + values = target.get("dimensions") + if not isinstance(values, Mapping): + failures.append(f"{owner} schema 7 requires a 'dimensions' object.") + return + geography_count = 0 + for dimension_id, raw_value in values.items(): + if dimension_id not in dimension_definitions: + failures.append(f"{owner} references undefined dimension {dimension_id!r}.") + continue + if not isinstance(raw_value, str) or not raw_value.strip(): + failures.append( + f"{owner} dimension {dimension_id!r} must have a non-empty " + "string value." + ) + continue + definition = dimension_definitions.get(dimension_id) + if not isinstance(definition, Mapping): + continue + if definition.get("role") == "geography": + geography_count += 1 + labels = definition.get("values") + if isinstance(labels, Mapping) and raw_value not in labels: + failures.append( + f"{owner} geography value {raw_value!r} has no label in " + f"dimension {dimension_id!r}." + ) + if geography_count > 1: + failures.append(f"{owner} may populate at most one geography-role dimension.") + + def _uk_non_negative_int( value: object, *, diff --git a/packages/microcosm-data/tests/test_contract.py b/packages/microcosm-data/tests/test_contract.py index 9c536c332..b1224c7c6 100644 --- a/packages/microcosm-data/tests/test_contract.py +++ b/packages/microcosm-data/tests/test_contract.py @@ -2341,7 +2341,7 @@ def test_legacy_diagnostics_exemption_is_scoped_to_the_exact_june_id( payload=diagnostics, ) - with pytest.raises(ReleaseContractError, match="publishes version 6"): + with pytest.raises(ReleaseContractError, match="publishes version 7"): validate_release_dir(directory) @@ -3548,6 +3548,75 @@ def test_malformed_calibration_diagnostics_is_rejected( assert "targets" in failures +def test_schema_7_structured_calibration_diagnostics_are_accepted( + release_dir: Path, +) -> None: + diagnostics = _calibration_diagnostics() + diagnostics["schema_version"] = 7 + diagnostics["dimensions"] = { + "geography_country": { + "label": "Country", + "role": "geography", + "level": "country", + "values": {"0100000US": "United States"}, + "order": ["0100000US"], + } + } + for row in diagnostics["targets"]: + row["source"] = {"id": "fixture", "citation": row["source"]} + row["variable"] = {"id": row["target_name"]} + row["dimensions"] = {"geography_country": "0100000US"} + _write_json_and_refresh_manifest_hash( + release_dir, + filename="calibration_diagnostics.json", + artifact_key="calibration_diagnostics", + payload=diagnostics, + ) + + validate_release_dir(release_dir) + + +def test_schema_7_rejects_partial_identity_and_multiple_geographies( + release_dir: Path, +) -> None: + diagnostics = _calibration_diagnostics() + diagnostics["schema_version"] = 7 + diagnostics["dimensions"] = { + "geography_country": { + "label": "Country", + "role": "geography", + "level": "country", + "values": {"0100000US": "United States"}, + }, + "geography_state": { + "label": "State", + "role": "geography", + "level": "state", + "values": {"0400000US06": "CA"}, + }, + } + for row in diagnostics["targets"]: + row["source"] = {"id": "fixture"} + row["variable"] = {"label": "Missing id"} + row["dimensions"] = { + "geography_country": "0100000US", + "geography_state": "0400000US06", + } + _write_json_and_refresh_manifest_hash( + release_dir, + filename="calibration_diagnostics.json", + artifact_key="calibration_diagnostics", + payload=diagnostics, + ) + + with pytest.raises(ReleaseContractError) as excinfo: + validate_release_dir(release_dir) + + failures = "\n".join(excinfo.value.failures) + assert "'variable' to be an object with a non-empty string 'id'" in failures + assert "at most one geography-role dimension" in failures + + def test_malformed_us_source_coverage_diagnostics_is_rejected( release_dir: Path, ) -> None: diff --git a/tools/generate_uk_target_references.py b/tools/generate_uk_target_references.py index 79c3998ae..81b69282b 100644 --- a/tools/generate_uk_target_references.py +++ b/tools/generate_uk_target_references.py @@ -16,6 +16,7 @@ from typing import Any from microcosm.build.target_reference_authoring import ( + AuthoredTargetReferences, TargetReferenceAuthoringConfig, author_target_references, target_references_resource, @@ -91,10 +92,39 @@ "classes and geography-pin decisions are recorded in " "uk/target_reference_membership.json. metadata.measure_kind records that " "measures are prepared columns produced from the contract binding payload " - "referenced by metadata.contract_target_id." + "referenced by metadata.contract_target_id. OBR references also declare " + "metadata.diagnostic_variable_id as efo_receipts or efo_expenditure so " + "schema-7 consumers group the forecast lines by their source table." ) NATIONAL_GEOGRAPHY_LEVELS = frozenset({"country", "region"}) +OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID = { + "obr.income_tax": "efo_receipts", + "obr.ni": "efo_receipts", + "obr.ni_employee": "efo_receipts", + "obr.ni_employer": "efo_receipts", + "obr.ni_self_employed": "efo_receipts", + "obr.vat": "efo_receipts", + "obr.fuel_duties": "efo_receipts", + "obr.capital_gains_tax": "efo_receipts", + "obr.sdlt": "efo_receipts", + "obr.attendance_allowance": "efo_expenditure", + "obr.carers_allowance": "efo_expenditure", + "obr.child_benefit": "efo_expenditure", + "obr.council_tax": "efo_expenditure", + "obr.esa": "efo_expenditure", + "obr.housing_benefit": "efo_expenditure", + "obr.jobseekers_allowance": "efo_expenditure", + "obr.pension_credit": "efo_expenditure", + "obr.pip": "efo_expenditure", + "obr.state_pension": "efo_expenditure", + "obr.statutory_maternity_pay": "efo_expenditure", + "obr.tv_licence_fee": "efo_expenditure", + "obr.universal_credit_in_cap": "efo_expenditure", + "obr.universal_credit_outside_cap": "efo_expenditure", + "obr.winter_fuel_allowance": "efo_expenditure", +} + def main() -> None: args = _parser().parse_args() @@ -124,6 +154,7 @@ def main() -> None: source_fact_feed=str(args.ledger_facts), ) authored = author_target_references(contract, facts, config) + authored = _add_diagnostic_variable_ids(authored) _add_uk_membership_accounting(authored.membership_report, authored.references) resource = target_references_resource( country="uk", @@ -140,6 +171,25 @@ def main() -> None: ) +def _add_diagnostic_variable_ids( + authored: AuthoredTargetReferences, +) -> AuthoredTargetReferences: + """Assign producer-defined dashboard categories to OBR target references.""" + + references: list[dict[str, Any]] = [] + for reference in authored.references: + metadata = dict(reference["metadata"]) + target_id = str(metadata["contract_target_id"]) + variable_id = OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID.get(target_id) + if variable_id is not None: + metadata["diagnostic_variable_id"] = variable_id + references.append({**reference, "metadata": metadata}) + return AuthoredTargetReferences( + references=tuple(references), + membership_report=authored.membership_report, + ) + + def _parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser() parser.add_argument("--contract", type=Path, required=True)