diff --git a/packages/microcosm-build/src/microcosm/build/uk/target_references.json b/packages/microcosm-build/src/microcosm/build/uk/target_references.json index d7cdb07f0..114952a29 100644 --- a/packages/microcosm-build/src/microcosm/build/uk/target_references.json +++ b/packages/microcosm-build/src/microcosm/build/uk/target_references.json @@ -22,6 +22,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.income_tax", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -40,6 +41,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.ni", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -58,6 +60,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.ni_employee", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -76,6 +79,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.ni_employer", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -94,6 +98,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.ni_self_employed", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -112,6 +117,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.vat", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -130,6 +136,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.fuel_duties", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -148,6 +155,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.capital_gains_tax", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -166,6 +174,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.sdlt", + "diagnostic_variable_id": "efo_receipts", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -184,6 +193,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.attendance_allowance", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -202,6 +212,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.carers_allowance", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -220,6 +231,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.child_benefit", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -238,6 +250,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.council_tax", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -256,6 +269,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.esa", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection", @@ -275,6 +289,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.housing_benefit", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -293,6 +308,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.jobseekers_allowance", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -311,6 +327,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.pension_credit", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -329,6 +346,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.pip", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -347,6 +365,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.state_pension", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -365,6 +384,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.statutory_maternity_pay", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -383,6 +403,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.tv_licence_fee", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -401,6 +422,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.universal_credit_in_cap", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -419,6 +441,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.universal_credit_outside_cap", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" @@ -437,6 +460,7 @@ "period": 2025, "metadata": { "contract_target_id": "obr.winter_fuel_allowance", + "diagnostic_variable_id": "efo_expenditure", "measure_kind": "prepared_column" }, "assertion_policy": "allow_source_projection" diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/diagnostics.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/diagnostics.py index 7907b4d71..befc156eb 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/diagnostics.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/diagnostics.py @@ -1,12 +1,11 @@ """Standard calibration diagnostics for UK release candidates. The shared :mod:`microcosm.calibrate.diagnostics` payload is the release -contract: it carries the target surface, every target row, solver options, and -the concentration scalars used by US releases. UK needs a little more release -evidence without changing that shared schema (and therefore without changing -US output): the effective-sample-size fraction, shipped-weight concentration, -zero-weight rows split by their declared support strata, and target fit by UK -geography level. +contract: it carries the target surface, every target row, solver options, +schema-7 source/variable/dimension identity, and the concentration scalars used +by US releases. UK adds the effective-sample-size fraction, shipped-weight +concentration, zero-weight rows split by their declared support strata, and +target fit by UK geography level. This module wraps the shared payload and places those additions under a separately versioned ``uk_diagnostics`` block. The common top-level diff --git a/packages/microcosm-build/tests/test_uk_diagnostics.py b/packages/microcosm-build/tests/test_uk_diagnostics.py index 5eb8e8f01..6c7145b6f 100644 --- a/packages/microcosm-build/tests/test_uk_diagnostics.py +++ b/packages/microcosm-build/tests/test_uk_diagnostics.py @@ -577,7 +577,7 @@ def test_payload_requires_a_valid_matching_uk_registry() -> None: target_geography_levels=geography, target_registry=TargetRegistry((), country="uk"), ) - with pytest.raises(ValueError, match="exactly partition"): + with pytest.raises(ValueError, match="does not contain compiled target row"): uk_calibration_diagnostics_payload( result, frame, diff --git a/packages/microcosm-build/tests/test_uk_rowwise_candidate.py b/packages/microcosm-build/tests/test_uk_rowwise_candidate.py index 2d5a8933b..5de3268bb 100644 --- a/packages/microcosm-build/tests/test_uk_rowwise_candidate.py +++ b/packages/microcosm-build/tests/test_uk_rowwise_candidate.py @@ -399,7 +399,7 @@ def test_candidate_build_writes_calibrated_h5_and_evidence( assert diagnostics["metric"].unique().tolist() == ["households"] assert len(support) == 8 assert past_cap["n_targets"] == 4 - assert calibration_diagnostics["schema_version"] == 6 + assert calibration_diagnostics["schema_version"] == 7 uk_diagnostics = calibration_diagnostics["uk_diagnostics"] assert len(uk_diagnostics["weakest_families"]) == 1 assert len(uk_diagnostics["weakest_areas_by_fit"]["bottom_by_fit"]) == 4 diff --git a/packages/microcosm-build/tests/test_uk_target_references.py b/packages/microcosm-build/tests/test_uk_target_references.py index d9ee9d216..d86fd7fc8 100644 --- a/packages/microcosm-build/tests/test_uk_target_references.py +++ b/packages/microcosm-build/tests/test_uk_target_references.py @@ -24,9 +24,12 @@ TargetReferenceAuthoringConfig, author_target_references, ) +from microcosm.calibrate.diagnostics import diagnostics_payload from microcosm.calibrate.matrix import build_constraint_matrix +from microcosm.calibrate.score import score_targets from microcosm.frame import EntitySchema, Frame, WeightKind, Weights from tools.generate_uk_target_references import ( + OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID, POLICYENGINE_BINDING_KEYS, _annual_uc_award_band_token, _geography_pins, @@ -128,10 +131,16 @@ def test_uk_target_references_follow_contract_derivation_rules() -> None: assert reference["measure"] == expected_measure assert reference["family"] == target["family"] assert reference["period"] == 2025 - assert reference["metadata"] == { + expected_metadata = { "contract_target_id": contract_target_id, "measure_kind": "prepared_column", } + diagnostic_variable_id = OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID.get( + contract_target_id + ) + if diagnostic_variable_id is not None: + expected_metadata["diagnostic_variable_id"] = diagnostic_variable_id + assert reference["metadata"] == expected_metadata # The measure is a prepared column, so the pointed-to contract binding # must carry what the microcosm#622 materializer needs to prepare it. assert ( @@ -152,6 +161,29 @@ def test_uk_target_references_do_not_bind_known_mismatched_property_amounts() -> ] +def test_uk_obr_references_declare_receipts_and_expenditure_categories() -> None: + references = _load_uk_resource("target_references.json")["target_references"] + obr_references = [ + reference + for reference in references + if reference["metadata"]["contract_target_id"].startswith("obr.") + ] + + assert len(obr_references) == len(OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID) + assert { + reference["metadata"]["contract_target_id"]: reference["metadata"][ + "diagnostic_variable_id" + ] + for reference in obr_references + } == OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID + assert not [ + reference["name"] + for reference in references + if not reference["metadata"]["contract_target_id"].startswith("obr.") + and "diagnostic_variable_id" in reference["metadata"] + ] + + def test_uk_target_references_do_not_emit_nan_uc_payment_bands() -> None: resource = _load_uk_resource("target_references.json") @@ -418,6 +450,7 @@ def test_uk_target_references_compile_from_real_staged_feed_rows() -> None: assert income_tax.period == 2025 assert income_tax.metadata["ledger_assertion"] == "source_projection" assert income_tax.metadata["ledger_assertion_policy"] == ("allow_source_projection") + assert income_tax.metadata["diagnostic_variable_id"] == "efo_receipts" tcl_households = targets["dwp.uc.two_child_limit.households_affected"] assert tcl_households.value == 469_780 @@ -558,6 +591,33 @@ def test_uk_target_references_constrain_a_frame_with_prepared_columns() -> None: for name, target_value in zip(problem.names, problem.target_vector, strict=True): assert target_value == fact_values_by_name[name], name + diagnostics = diagnostics_payload( + score_targets(frame, registry.to_target_set()), + target_registry=registry, + ) + obr_variables = { + row["name"]: row["variable"] + for row in diagnostics["targets"] + if row["source"]["id"] == "obr" + } + assert { + row["source"]["label"] + for row in diagnostics["targets"] + if row["source"]["id"] == "obr" + } == {"Office for Budget Responsibility"} + assert obr_variables == { + "obr.esa@2025": { + "id": "efo_expenditure", + "label": "EFO expenditure", + "measure": "total", + }, + "obr.income_tax@2025": { + "id": "efo_receipts", + "label": "EFO receipts", + "measure": "total", + }, + } + def _real_uk_consumer_fact_rows() -> list[dict]: """Public rows copied from .codex-work/consumer_facts_uk.jsonl.""" diff --git a/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py b/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py index 0794037d1..10c839717 100644 --- a/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py +++ b/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py @@ -1743,20 +1743,16 @@ def selected(path, *, allow_terminal_gate_failure): ) return - loaded_frame, receipt, loaded_identity = ( - builder._load_base_pool_if_identified( - pool_h5, - allow_gate_failed_base_pool=allow_gate_failed, - ) + loaded_frame, receipt, loaded_identity = builder._load_base_pool_if_identified( + pool_h5, + allow_gate_failed_base_pool=allow_gate_failed, ) assert loaded_frame is frame assert loaded_identity is authenticated assert receipt["status"] == status assert receipt["allow_gate_failed_base_pool"] is allow_gate_failed - assert receipt["agreement_gate_reference"]["failure_count"] == len( - gate_failures - ) + assert receipt["agreement_gate_reference"]["failure_count"] == len(gate_failures) def test_builder_refuses_actual_red_base_h5_pool_sidecar_without_opt_in( @@ -1777,7 +1773,9 @@ def test_builder_refuses_actual_red_base_h5_pool_sidecar_without_opt_in( ) out = tmp_path / "out" monkeypatch.setattr(builder, "_git_dirty", lambda: False) - monkeypatch.setattr(builder, "_refuse_certified_release_dir_reuse", lambda path: None) + monkeypatch.setattr( + builder, "_refuse_certified_release_dir_reuse", lambda path: None + ) monkeypatch.setattr( builder, "_load_frame", @@ -1826,7 +1824,9 @@ def test_builder_refuses_bare_stamped_pool_h5_before_generic_load( ) out = tmp_path / "out" monkeypatch.setattr(builder, "_git_dirty", lambda: False) - monkeypatch.setattr(builder, "_refuse_certified_release_dir_reuse", lambda path: None) + monkeypatch.setattr( + builder, "_refuse_certified_release_dir_reuse", lambda path: None + ) monkeypatch.setattr( builder, "_load_frame", @@ -4160,6 +4160,13 @@ def test_release_calibration_diagnostics_writes_nan_final_loss_as_null( measure="income", value=500_000.0, source="fixture", + metadata={ + "ledger_selector_source_name": "irs_soi", + "ledger_measure_concept": "irs_soi.income", + "ledger_measure_unit": "usd", + "ledger_geography_level": "state", + "ledger_geography_id": "0400000US06", + }, ), ), country="us", @@ -4197,6 +4204,25 @@ def test_release_calibration_diagnostics_writes_nan_final_loss_as_null( ) diagnostics = json.loads((tmp_path / "calibration_diagnostics.json").read_text()) + assert diagnostics["schema_version"] == 7 + assert diagnostics["targets"][0]["source"] == { + "id": "irs_soi", + "label": "IRS Statistics of Income", + "citation": "fixture", + } + assert diagnostics["targets"][0]["variable"] == { + "id": "income", + "label": "Income", + "measure": "total", + } + assert diagnostics["targets"][0]["dimensions"] == {"geography_state": "0400000US06"} + assert diagnostics["dimensions"]["geography_state"] == { + "label": "State", + "role": "geography", + "level": "state", + "values": {"0400000US06": "CA"}, + "order": ["0400000US06"], + } assert diagnostics["final_loss"] is None assert diagnostics["build"]["default_dataset"]["final_loss"] is None @@ -9160,9 +9186,7 @@ def test_exact_k_receipt_stays_strict_even_when_base_h5_opt_in_is_present() -> N builder = _load_builder_module() with pytest.raises(RuntimeError, match="lost its passing agreement gate"): - builder._exact_k_ladder_manifest_payload( - **_gate_failed_exact_k_inputs(builder) - ) + builder._exact_k_ladder_manifest_payload(**_gate_failed_exact_k_inputs(builder)) def _gate_failed_base_pool_receipt() -> dict[str, object]: diff --git a/packages/microcosm-calibrate/README.md b/packages/microcosm-calibrate/README.md index 233f24847..3ef0919ad 100644 --- a/packages/microcosm-calibrate/README.md +++ b/packages/microcosm-calibrate/README.md @@ -76,6 +76,34 @@ frontier can be read off any run's artifact. The standalone `effective_sample_size(weights)` scores any weight vector, e.g. a published artifact's. +Registry-backed diagnostics use schema 7. Each target publishes structured +`source`, `variable`, and `dimensions` objects, and the artifact publishes a +top-level dimension dictionary. The `source.id` remains the stable provider +identifier, while `source.label` comes from separate country-owned provider +label mappings in `microcosm.calibrate.provider_labels`. The `variable.id` +defines a provider's calibration statistic category, while `variable.label` +comes from separate country- and provider-specific mappings in +`microcosm.calibrate.variable_labels`. Labels are not copied into Chronicle +facts or repeated in target-reference metadata. Ledger geography metadata +becomes one typed geography dimension per level (for example, +`geography_country` or `geography_state`), with stable geography identifiers, +producer-owned labels, and deterministic value order. Ledger filter and layout +dimensions remain separate non-geographic dimensions. This applies to every +country release that passes its `TargetRegistry`, including the UK and US +release builders. Calls without a registry retain legacy target identity fields +because they do not provide enough declared information to construct structured +identities. + +Schema 7 also separates the statistic category from its measurement. For +legacy Ledger concepts whose declared unit agrees with a trailing `_count` or +`_amount`, the suffix is represented as `variable.measure` (`count` or `total`) +instead of remaining in `variable.id`. For example, +`hmrc.spi_employment_income_count` and +`hmrc.spi_employment_income_amount` both use the variable identifier +`spi_employment_income`; their measure values remain distinct. An explicit +`diagnostic_variable_id` or `variable` metadata value always takes precedence +and is not rewritten. + ## Example ```python diff --git a/packages/microcosm-calibrate/src/microcosm/calibrate/__init__.py b/packages/microcosm-calibrate/src/microcosm/calibrate/__init__.py index 5aa9db083..1fd3c55b6 100644 --- a/packages/microcosm-calibrate/src/microcosm/calibrate/__init__.py +++ b/packages/microcosm-calibrate/src/microcosm/calibrate/__init__.py @@ -103,6 +103,12 @@ def _assert_frame_compatible(version: str, required: tuple[int, int]) -> None: from microcosm.calibrate.monetary_binding import ( # noqa: E402 - after compat gate MonetaryBindingIntegrityError, ) +from microcosm.calibrate.provider_labels import ( # noqa: E402 - after compat gate + CALIBRATION_PROVIDER_LABELS_BY_COUNTRY, + UK_CALIBRATION_PROVIDER_LABELS, + US_CALIBRATION_PROVIDER_LABELS, + calibration_provider_label, +) from microcosm.calibrate.registry import ( # noqa: E402 - after the compat gate TargetRegistry, TargetSpec, @@ -128,11 +134,19 @@ def _assert_frame_compatible(version: str, required: tuple[int, int]) -> None: Target, TargetSet, ) +from microcosm.calibrate.variable_labels import ( # noqa: E402 - after compat gate + CALIBRATION_VARIABLE_LABELS_BY_COUNTRY, + UK_CALIBRATION_VARIABLE_LABELS, + US_CALIBRATION_VARIABLE_LABELS, + calibration_variable_label, +) __version__ = "0.1.0" __all__ = [ "CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION", + "CALIBRATION_PROVIDER_LABELS_BY_COUNTRY", + "CALIBRATION_VARIABLE_LABELS_BY_COUNTRY", "CONSERVE_MASS", "FREE_MASS", "TARGET_LOSS_ATTRIBUTION_ABS_TOLERANCE", @@ -149,9 +163,15 @@ def _assert_frame_compatible(version: str, required: tuple[int, int]) -> None: "TargetRegistry", "TargetSet", "TargetSpec", + "UK_CALIBRATION_PROVIDER_LABELS", + "UK_CALIBRATION_VARIABLE_LABELS", + "US_CALIBRATION_PROVIDER_LABELS", + "US_CALIBRATION_VARIABLE_LABELS", "build_constraint_matrix", "calibrate", "calibrate_l0_refit", + "calibration_provider_label", + "calibration_variable_label", "assert_exact_k_support", "default_target_loss_scales", "effective_sample_size", diff --git a/packages/microcosm-calibrate/src/microcosm/calibrate/diagnostics.py b/packages/microcosm-calibrate/src/microcosm/calibrate/diagnostics.py index f63913d68..ed3d526de 100644 --- a/packages/microcosm-calibrate/src/microcosm/calibrate/diagnostics.py +++ b/packages/microcosm-calibrate/src/microcosm/calibrate/diagnostics.py @@ -21,6 +21,7 @@ import json import logging import math +import re from collections.abc import Mapping from pathlib import Path from typing import Any @@ -30,7 +31,9 @@ TargetLossAttributionError, assemble_target_loss_attribution, ) +from microcosm.calibrate.provider_labels import calibration_provider_label from microcosm.calibrate.solve import CalibrationResult +from microcosm.calibrate.variable_labels import calibration_variable_label __all__ = [ "CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION", @@ -49,10 +52,90 @@ #: v6 added authoritative final per-target loss attribution and an explicit #: warning-only degradation state when that supplementary attribution cannot #: be validated. -CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION = 6 +#: v7 adds producer-defined source, variable, and dimension identity for +#: registry-backed release diagnostics. Sources include country-owned display +#: labels when registered. Geography is represented as a typed dimension with +#: stable identifiers and display labels. +CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION = 7 _LOGGER = logging.getLogger(__name__) +_UK_GEOGRAPHY_LABELS = { + "K02000001": "United Kingdom", + "K03000001": "Great Britain", + "E92000001": "England", + "W92000004": "Wales", + "S92000003": "Scotland", + "N92000002": "Northern Ireland", +} + +_US_STATE_POSTAL = { + "01": "AL", + "02": "AK", + "04": "AZ", + "05": "AR", + "06": "CA", + "08": "CO", + "09": "CT", + "10": "DE", + "11": "DC", + "12": "FL", + "13": "GA", + "15": "HI", + "16": "ID", + "17": "IL", + "18": "IN", + "19": "IA", + "20": "KS", + "21": "KY", + "22": "LA", + "23": "ME", + "24": "MD", + "25": "MA", + "26": "MI", + "27": "MN", + "28": "MS", + "29": "MO", + "30": "MT", + "31": "NE", + "32": "NV", + "33": "NH", + "34": "NJ", + "35": "NM", + "36": "NY", + "37": "NC", + "38": "ND", + "39": "OH", + "40": "OK", + "41": "OR", + "42": "PA", + "44": "RI", + "45": "SC", + "46": "SD", + "47": "TN", + "48": "TX", + "49": "UT", + "50": "VT", + "51": "VA", + "53": "WA", + "54": "WV", + "55": "WI", + "56": "WY", +} + +_COUNT_UNITS = frozenset( + { + "count", + "households", + "people", + "persons", + "returns", + "claims", + } +) +_TOTAL_UNITS = frozenset({"gbp", "usd", "dollars", "pounds"}) +_MEAN_UNITS = frozenset({"percent", "percentage", "rate", "ratio"}) + def _finite(value: float) -> float | None: """JSON has no NaN/inf; a non-finite diagnostic serializes as null.""" @@ -120,6 +203,322 @@ def _registry_spec_lookup(target_registry: object | None) -> dict[str, object]: return lookup +def _metadata_string(metadata: Mapping[str, object], key: str) -> str: + """Return a stripped metadata string, or an empty string.""" + + value = metadata.get(key) + return value.strip() if isinstance(value, str) else "" + + +def _source_id(spec: object, metadata: Mapping[str, object]) -> str: + """Return the publisher identifier declared by a registry-backed target.""" + + explicit = _metadata_string(metadata, "diagnostic_source_id") + if explicit: + return explicit + selector_source = _metadata_string(metadata, "ledger_selector_source_name") + if selector_source: + return selector_source + source_record = _metadata_string(metadata, "ledger_source_record_id") + if source_record: + return source_record.split(".", 1)[0] + family = str(getattr(spec, "family", "")).strip() + if family: + return family + name = str(getattr(spec, "name", "")).strip() + return re.split(r"[./]", name, maxsplit=1)[0] or "other" + + +def _variable_id( + spec: object, + metadata: Mapping[str, object], + *, + source_id: str, +) -> str: + """Return a stable statistic identifier for a registry-backed target. + + Explicit diagnostic identifiers are already producer declarations and are + preserved verbatim. Ledger measure concepts are older compound identifiers: + some append ``_count`` or ``_amount`` even though schema 7 represents that + distinction separately in ``variable.measure``. Remove only the suffix that + agrees with the declared unit so count and amount rows remain one dashboard + category without conflating their measurements. + """ + + def without_source_prefix(value: str) -> str: + for prefix in (f"{source_id}.", f"{source_id}:"): + if value.startswith(prefix): + return value[len(prefix) :] + return value + + for key in ("diagnostic_variable_id", "variable"): + value = _metadata_string(metadata, key) + if value: + return without_source_prefix(value) + + measure = _variable_measure(metadata) + measure_suffix = {"count": "_count", "total": "_amount"}.get(measure, "") + for key in ("ledger_measure_concept", "ledger_source_concept"): + value = _metadata_string(metadata, key) + if value: + identifier = without_source_prefix(value) + if measure_suffix and identifier.endswith(measure_suffix): + identifier = identifier[: -len(measure_suffix)] + return identifier + contract_id = _metadata_string(metadata, "contract_target_id") + if contract_id: + prefix = re.split(r"[./]", contract_id, maxsplit=1)[0] + remainder = contract_id[len(prefix) :].lstrip("./") + return remainder or contract_id + measure = str(getattr(spec, "measure", "")).strip() + return measure or str(getattr(spec, "name", "")).strip() or "unknown" + + +def _variable_measure(metadata: Mapping[str, object]) -> str: + """Classify a declared Ledger unit into the dashboard's measure vocabulary.""" + + unit = _metadata_string(metadata, "ledger_measure_unit").lower() + if unit in _COUNT_UNITS: + return "count" + if unit in _TOTAL_UNITS: + return "total" + if unit in _MEAN_UNITS: + return "mean" + return "" + + +def _humanize_identifier(value: str) -> str: + """Turn a machine identifier into a concise dimension label.""" + + tail = value.rsplit("#", 1)[-1].rsplit(".", 1)[-1] + return " ".join(part.capitalize() for part in re.split(r"[_:/-]+", tail) if part) + + +def _dimension_label(dimension_id: str) -> str: + """Return the established display label for a Ledger dimension id.""" + + if dimension_id == "us:statutes/26/62#adjusted_gross_income": + return "Income Band" + if dimension_id == "census_stc.item": + return "Item" + if dimension_id == "hhs_acf_tanf.spending_category": + return "Spending Category" + if dimension_id == "income_range": + return "Income Band" + if dimension_id == "filing_status": + return "Filing Status" + if dimension_id == "eitc_child_count": + return "Qualifying Children" + return _humanize_identifier(dimension_id) + + +def _normalized_geography_level(value: str) -> str: + """Normalize producer aliases used by the existing country contracts.""" + + normalized = value.strip().lower().replace("-", "_").replace(" ", "_") + return "local_authority" if normalized == "la" else normalized + + +def _geography_label( + *, + country: str, + level: str, + geography_id: str, + metadata: Mapping[str, object], +) -> str: + """Resolve a producer-owned geography label without consumer name parsing.""" + + explicit = _metadata_string(metadata, "ledger_geography_name") + if explicit: + return explicit + if country == "uk": + return _UK_GEOGRAPHY_LABELS.get(geography_id, geography_id) + if country == "us": + if geography_id == "0100000US" or level in {"country", "national"}: + return "United States" + match = re.search(r"US(\d{2})(\d{2})$", geography_id) + if level == "congressional_district" and match: + postal = _US_STATE_POSTAL.get(match.group(1)) + if postal: + return f"{postal}-{match.group(2)}" + match = re.search(r"US(\d{2})$", geography_id) + if level == "state" and match: + return _US_STATE_POSTAL.get(match.group(1), geography_id) + return geography_id + + +def _structured_dimensions( + metadata: Mapping[str, object], + *, + country: str, +) -> tuple[dict[str, str], dict[str, dict[str, object]]]: + """Build schema-7 row dimensions and their producer definitions.""" + + values: dict[str, str] = {} + definitions: dict[str, dict[str, object]] = {} + + geography_id = _metadata_string(metadata, "ledger_geography_id") + geography_level = _normalized_geography_level( + _metadata_string(metadata, "ledger_geography_level") + ) + if geography_id and geography_level: + dimension_id = f"geography_{geography_level}" + values[dimension_id] = geography_id + definitions[dimension_id] = { + "label": _humanize_identifier(geography_level), + "role": "geography", + "level": geography_level, + "values": { + geography_id: _geography_label( + country=country, + level=geography_level, + geography_id=geography_id, + metadata=metadata, + ) + }, + "order": [geography_id], + } + + filter_dimensions = [ + (key.removeprefix("ledger_filter_"), raw_value.strip()) + for key, raw_value in metadata.items() + if key.startswith("ledger_filter_") + and isinstance(raw_value, str) + and key.removeprefix("ledger_filter_") + and raw_value.strip() + ] + layout_dimension = _metadata_string(metadata, "ledger_layout_groupby_dimension") + layout_value = _metadata_string(metadata, "ledger_layout_groupby_value_id") + layout_label = _dimension_label(layout_dimension) + duplicate_filter = any( + _dimension_label(dimension_id) == layout_label and value == layout_value + for dimension_id, value in filter_dimensions + ) + resolved_geography_label = ( + _geography_label( + country=country, + level=geography_level, + geography_id=geography_id, + metadata=metadata, + ) + if geography_id and geography_level + else "" + ) + geography_layout = layout_dimension in { + "geography", + "state", + "cms_medicaid.state_abbreviation", + } + redundant_geography = layout_value.lower() in { + geography_id.lower(), + resolved_geography_label.lower(), + } + if ( + layout_dimension + and layout_value + and not duplicate_filter + and not geography_layout + and not redundant_geography + ): + values[layout_dimension] = layout_value + definitions[layout_dimension] = { + "label": layout_label, + "values": {layout_value: _humanize_identifier(layout_value)}, + "order": [layout_value], + } + + for dimension_id, value in filter_dimensions: + values[dimension_id] = value + definitions[dimension_id] = { + "label": _dimension_label(dimension_id), + "values": {value: _humanize_identifier(value)}, + "order": [value], + } + return values, definitions + + +def _merge_dimension_definitions( + destination: dict[str, dict[str, object]], + additions: Mapping[str, Mapping[str, object]], +) -> None: + """Merge per-row dimension declarations into one deterministic dictionary.""" + + for dimension_id, addition in additions.items(): + current = destination.get(dimension_id) + if current is None: + destination[dimension_id] = { + **addition, + "values": dict(addition.get("values", {})), + "order": list(addition.get("order", [])), + } + continue + for key in ("label", "role", "level"): + incoming = addition.get(key) + if incoming is not None and current.get(key) != incoming: + raise ValueError( + f"Diagnostics dimension {dimension_id!r} has conflicting " + f"{key} declarations {current.get(key)!r} and {incoming!r}." + ) + current_values = current.setdefault("values", {}) + current_order = current.setdefault("order", []) + if not isinstance(current_values, dict) or not isinstance(current_order, list): + raise TypeError("Diagnostics dimension aggregation state is malformed.") + for raw_value, label in addition.get("values", {}).items(): + existing = current_values.get(raw_value) + if existing is not None and existing != label: + raise ValueError( + f"Diagnostics dimension {dimension_id!r} value " + f"{raw_value!r} has conflicting labels {existing!r} and {label!r}." + ) + current_values[raw_value] = label + for raw_value in addition.get("order", []): + if raw_value not in current_order: + current_order.append(raw_value) + + +def _structured_target_fields( + target: object, + spec: object, + *, + country: str, +) -> tuple[dict[str, object], dict[str, dict[str, object]]]: + """Serialize complete schema-7 identity for one registry-backed target.""" + + metadata = dict(getattr(spec, "metadata", {}) or {}) + source_id = _source_id(spec, metadata) + variable_id = _variable_id(spec, metadata, source_id=source_id) + dimensions, definitions = _structured_dimensions(metadata, country=country) + citation = str(getattr(target, "source", "")).strip() + source: dict[str, str] = {"id": source_id} + source_label = calibration_provider_label(country, source_id) + if source_label: + source["label"] = source_label + if citation: + source["citation"] = citation + source_url = next( + ( + part.strip() + for part in citation.split("|") + if part.strip().startswith(("https://", "http://")) + ), + "", + ) + if source_url: + source["url"] = source_url + variable: dict[str, str] = {"id": variable_id} + variable_label = calibration_variable_label(country, source_id, variable_id) + if variable_label: + variable["label"] = variable_label + measure = _variable_measure(metadata) + if measure: + variable["measure"] = measure + return { + "source": source, + "variable": variable, + "dimensions": dimensions, + }, definitions + + def _target_identity_rows(result: CalibrationResult) -> list[dict[str, object]]: """The target surface as structured rows suitable for hashing.""" rows: list[dict[str, object]] = [] @@ -328,17 +727,33 @@ def diagnostics_payload( floats become ``null``). """ registry_specs = _registry_spec_lookup(target_registry) - target_rows = [ - _target_row( + registry_country = str(getattr(target_registry, "country", "")).strip() + dimension_definitions: dict[str, dict[str, object]] = {} + target_rows: list[dict[str, object]] = [] + for index, (diagnostic, target) in enumerate( + zip(result.diagnostics, result.problem.targets, strict=True) + ): + spec = registry_specs.get(diagnostic.name) + if target_registry is not None and spec is None: + raise ValueError( + "The supplied target registry does not contain compiled target " + f"row {diagnostic.name!r}." + ) + row = _target_row( diagnostic, target, compiled_target=result.problem.target_vector[index], - spec=registry_specs.get(diagnostic.name), - ) - for index, (diagnostic, target) in enumerate( - zip(result.diagnostics, result.problem.targets, strict=True) + spec=spec, ) - ] + if spec is not None: + structured_fields, row_definitions = _structured_target_fields( + target, + spec, + country=registry_country, + ) + row.update(structured_fields) + _merge_dimension_definitions(dimension_definitions, row_definitions) + target_rows.append(row) payload = { "schema_version": CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION, "weight_entity": result.weight_entity, @@ -361,6 +776,8 @@ def diagnostics_payload( "diagnostic_warnings": [], "targets": target_rows, } + if target_registry is not None: + payload["dimensions"] = dimension_definitions try: attribution = assemble_target_loss_attribution(result) except TargetLossAttributionError as error: diff --git a/packages/microcosm-calibrate/src/microcosm/calibrate/provider_labels.py b/packages/microcosm-calibrate/src/microcosm/calibrate/provider_labels.py new file mode 100644 index 000000000..e6593c512 --- /dev/null +++ b/packages/microcosm-calibrate/src/microcosm/calibrate/provider_labels.py @@ -0,0 +1,84 @@ +"""Country-owned display labels for calibration data providers. + +The identifiers remain the stable values used for grouping, filtering, and +joins. These mappings provide presentation text only; they deliberately live +outside Chronicle facts and target-reference metadata so every target from the +same provider is serialized consistently. +""" + +from __future__ import annotations + +from collections.abc import Mapping +from types import MappingProxyType + +__all__ = [ + "CALIBRATION_PROVIDER_LABELS_BY_COUNTRY", + "UK_CALIBRATION_PROVIDER_LABELS", + "US_CALIBRATION_PROVIDER_LABELS", + "calibration_provider_label", +] + + +US_CALIBRATION_PROVIDER_LABELS: Mapping[str, str] = MappingProxyType( + { + "bea": "BEA", + "bea_nipa": "BEA · National Income and Product Accounts", + "cbo": "CBO", + "census_acs": "Census · American Community Survey", + "census_pep": "Census · Population Estimates Program", + "census_population": "Census population", + "census_population_projections": "Census · Population Projections", + "census_stc": "Census · State Tax Collections", + "cms_aca": "CMS · ACA marketplace", + "cms_medicaid": "CMS · Medicaid / CHIP", + "cms_medicare": "CMS · Medicare", + "cms_nhe": "CMS · National Health Expenditure Accounts", + "federal_reserve": "Federal Reserve", + "federal_reserve_z1": "Federal Reserve · Financial Accounts (Z.1)", + "hhs_acf_liheap": "HHS · LIHEAP", + "hhs_acf_tanf": "HHS · TANF", + "ici": "Investment Company Institute", + "irs_soi": "IRS Statistics of Income", + "jct": "JCT", + "kff": "KFF", + "ssa": "SSA", + "ssa_ssi_monthly": "SSA · SSI Monthly Statistics", + "ssa_supplement": "SSA · Annual Statistical Supplement", + "state_income_tax": "State income tax", + "unspecified": "Internal calibration constraint", + "usda_snap": "USDA · SNAP", + } +) + + +UK_CALIBRATION_PROVIDER_LABELS: Mapping[str, str] = MappingProxyType( + { + "dwp": "Department for Work and Pensions", + "hmrc": "HM Revenue and Customs", + "isc": "Independent Schools Council", + "obr": "Office for Budget Responsibility", + "ons": "Office for National Statistics", + "scotgov": "Scottish Government", + "slc": "Student Loans Company", + "voa": "Valuation Office Agency", + } +) + + +CALIBRATION_PROVIDER_LABELS_BY_COUNTRY: Mapping[str, Mapping[str, str]] = ( + MappingProxyType( + { + "uk": UK_CALIBRATION_PROVIDER_LABELS, + "us": US_CALIBRATION_PROVIDER_LABELS, + } + ) +) + + +def calibration_provider_label(country: str, source_id: str) -> str | None: + """Return a country-owned display label for a structured source id.""" + + labels = CALIBRATION_PROVIDER_LABELS_BY_COUNTRY.get(country.strip().lower()) + if labels is None: + return None + return labels.get(source_id.strip()) diff --git a/packages/microcosm-calibrate/src/microcosm/calibrate/variable_labels.py b/packages/microcosm-calibrate/src/microcosm/calibrate/variable_labels.py new file mode 100644 index 000000000..2c01d6f85 --- /dev/null +++ b/packages/microcosm-calibrate/src/microcosm/calibrate/variable_labels.py @@ -0,0 +1,294 @@ +"""Country-owned display labels for calibration statistic categories. + +Schema 7 calls these categories ``variable`` objects. A category may combine +several Chronicle facts or measures, so its display label belongs with +Microcosm's calibration grouping rather than with any one source fact. +""" + +from __future__ import annotations + +from collections.abc import Mapping +from types import MappingProxyType + +__all__ = [ + "CALIBRATION_VARIABLE_LABELS_BY_COUNTRY", + "UK_CALIBRATION_VARIABLE_LABELS", + "US_CALIBRATION_VARIABLE_LABELS", + "calibration_variable_label", +] + + +def _immutable_registry( + labels: dict[str, dict[str, str]], +) -> Mapping[str, Mapping[str, str]]: + return MappingProxyType( + { + source_id: MappingProxyType(dict(variable_labels)) + for source_id, variable_labels in labels.items() + } + ) + + +US_CALIBRATION_VARIABLE_LABELS: Mapping[str, Mapping[str, str]] = _immutable_registry( + { + "bea_nipa": { + "proprietors_income_with_inventory_valuation_and_capital_consumption_adjustments": ( + "Proprietors' income" + ), + "wages_and_salaries": "Wages and salaries", + }, + "cbo": { + "adjusted_gross_income_projection": "Adjusted gross income projection", + "net_business_income_projection": "Net business income projection", + "net_capital_gain_projection": "Net capital gain projection", + "qualified_dividend_income_projection": ( + "Qualified dividend income projection" + ), + "wages_and_salaries_projection": "Wages and salaries projection", + }, + "census_pep": { + "resident_population": "Resident population", + }, + "census_stc": { + "individual_income_tax_collections": ("Individual income tax collections"), + }, + "cms_aca": { + "aptc_consumers": "APTC consumers", + "marketplace_plan_selections": "Marketplace plan selections", + }, + "cms_medicaid": { + "total_chip_enrollment": "Total CHIP enrollment", + "total_medicaid_chip_enrollment": "Total Medicaid and CHIP enrollment", + "total_medicaid_enrollment": "Total Medicaid enrollment", + }, + "cms_medicare": { + "part_b_premium_income": "Part B premium income", + }, + "federal_reserve_z1": { + "federal_reserve.z1.households_nonprofits_net_worth": ( + "Households and nonprofit organizations net worth" + ), + }, + "hhs_acf_liheap": { + "households_served_by_state_programs": ( + "Households served by state programs" + ), + }, + "hhs_acf_tanf": { + "cash_assistance_expenditures": "Cash assistance expenditures", + }, + "irs_soi": { + "adjusted_gross_income": "Adjusted gross income", + "assigned_aca_ptc": "Assigned ACA PTC", + "business_net_profits": "Business net profits", + "capital_gains_gross": "Gross capital gains", + "charitable_deduction": "Charitable deduction", + "count": "Individual income tax returns", + "ctc": "CTC", + "deductible_mortgage_interest": "Deductible mortgage interest", + "eitc": "EITC", + "employment_income": "Employment income", + "estate_income": "Estate income", + "estate_losses": "Estate losses", + "income": "Income", + "income_tax": "Income tax", + "income_tax_before_credits": "Income tax before credits", + "interest_deduction": "Interest deduction", + "ira_distributions": "IRA distributions", + "itemized_taxable_income_deductions": ( + "Itemized taxable income deductions" + ), + "medical_expense_deduction": "Medical expense deduction", + "non_sch_d_capital_gains": "Non-Schedule D capital gains", + "ordinary_dividend_income": "Ordinary dividend income", + "partnership_and_s_corp_income": "Partnership and S corporation income", + "qualified_dividends": "Qualified dividends", + "real_estate_taxes": "Real estate taxes", + "refundable_ctc": "Refundable CTC", + "rent_and_royalty_net_income": "Rent and royalty net income", + "salt_deduction": "SALT deduction", + "tax_exempt_interest_income": "Tax-exempt interest income", + "taxable_income": "Taxable income", + "taxable_interest_income": "Taxable interest income", + "taxable_pension_income": "Taxable pension income", + "taxable_social_security": "Taxable Social Security", + "tip_income": "Tip income", + "unemployment_compensation": "Unemployment compensation", + }, + "jct": { + "individual_tax_expenditure_revenue_loss": ( + "Individual tax expenditure revenue loss" + ), + }, + "ssa_ssi_monthly": { + "ssa.ssi_federal_payment_recipient": "Federal SSI payment recipients", + }, + "ssa_supplement": { + "ssa.annual_oasdi_or_ssi_payment": "Annual OASDI or SSI payments", + "ssa.ssi_payment": "SSI payments", + "ssa.ssi_recipient": "SSI recipients", + }, + "unspecified": { + "selection_mass_protection.keogh_distributions": ( + "Keogh distributions selection-mass constraint" + ), + }, + "usda_snap": { + "average_monthly_households": "Average monthly households", + "total_benefits": "Total benefits", + }, + } +) + + +UK_CALIBRATION_VARIABLE_LABELS: Mapping[str, Mapping[str, str]] = _immutable_registry( + { + "dwp": { + "esa_claimants": "ESA claimants", + "jsa_claimants": "JSA claimants", + "uc_benefit_units": "Universal Credit benefit units", + "uc_capped_households_total": "Universal Credit capped households", + "uc_households": "Universal Credit households", + "uc_tcl_children_by_children_3": ( + "Universal Credit two-child limit: children in households with 3 children" + ), + "uc_tcl_children_by_children_4": ( + "Universal Credit two-child limit: children in households with 4 children" + ), + "uc_tcl_children_by_children_5": ( + "Universal Credit two-child limit: children in households with 5 children" + ), + "uc_tcl_children_by_children_6_plus": ( + "Universal Credit two-child limit: children in households with 6+ children" + ), + "uc_tcl_children_by_disability_claimant_pip": ( + "Universal Credit two-child limit: children by claimant PIP status" + ), + "uc_tcl_children_by_disability_disabled_child_element": ( + "Universal Credit two-child limit: children by disabled-child element" + ), + "uc_tcl_headline_children_affected_by_policy": ( + "Universal Credit two-child limit: children affected" + ), + "uc_tcl_headline_children_within_affected_households": ( + "Universal Credit two-child limit: children in affected households" + ), + "uc_tcl_headline_households_affected": ( + "Universal Credit two-child limit: households affected" + ), + "uc_tcl_households_by_children_3": ( + "Universal Credit two-child limit: households with 3 children" + ), + "uc_tcl_households_by_children_4": ( + "Universal Credit two-child limit: households with 4 children" + ), + "uc_tcl_households_by_children_5": ( + "Universal Credit two-child limit: households with 5 children" + ), + "uc_tcl_households_by_children_6_plus": ( + "Universal Credit two-child limit: households with 6+ children" + ), + "uc_tcl_households_by_disability_claimant_pip": ( + "Universal Credit two-child limit: households by claimant PIP status" + ), + "uc_tcl_households_by_disability_disabled_child_element": ( + "Universal Credit two-child limit: households by disabled-child element" + ), + }, + "hmrc": { + "cgt_gains_total": "Total CGT gains", + "cgt_taxpayers_total": "Total CGT taxpayers", + "salary_sacrifice_pension_users": "Salary-sacrifice pension users", + "spi_dividend_income": "SPI dividend income", + "spi_employment_income": "SPI employment income", + "spi_private_pension_income": "SPI private pension income", + "spi_property_income": "SPI property income", + "spi_self_employment_income": "SPI self-employment income", + "spi_state_pension": "SPI state pension", + }, + "isc": { + "pupils_at_member_schools": "Pupils at member schools", + }, + "obr": { + "efo_expenditure": "EFO expenditure", + "efo_receipts": "EFO receipts", + }, + "ons": { + "household_interest_resources": "Household interest resources", + "households_by_type": "Households by type", + "mid_year_population_estimate": "Mid-year population estimate", + "nbs_land_value_households": "NBS land value: households", + "nbs_land_value_nfc": "NBS land value: non-financial corporations", + "nbs_land_value_total": "NBS land value: total", + "population": "Population", + "pse_headcount_total_public_sector": "Public-sector employment headcount", + }, + "scotgov": { + "chargeable_dwellings_band_a": "Chargeable dwellings: Council Tax band A", + "chargeable_dwellings_band_b": "Chargeable dwellings: Council Tax band B", + "chargeable_dwellings_band_c": "Chargeable dwellings: Council Tax band C", + "chargeable_dwellings_band_d": "Chargeable dwellings: Council Tax band D", + "chargeable_dwellings_band_e": "Chargeable dwellings: Council Tax band E", + "chargeable_dwellings_band_f": "Chargeable dwellings: Council Tax band F", + "chargeable_dwellings_band_g": "Chargeable dwellings: Council Tax band G", + "chargeable_dwellings_band_h": "Chargeable dwellings: Council Tax band H", + "chargeable_dwellings_total": "Total chargeable dwellings", + "social_security_assistance_spending": ( + "Social security assistance spending" + ), + }, + "slc": { + "maintenance_loan_amount_paid": "Maintenance loan amount paid", + "maintenance_loan_recipients": "Maintenance loan recipients", + "student_loan_borrowers": "Student loan borrowers", + "student_loan_net_repayments_plan_1": ( + "Student loan net repayments: Plan 1" + ), + "student_loan_net_repayments_plan_2_full_time": ( + "Student loan net repayments: full-time Plan 2" + ), + "student_loan_net_repayments_total_higher_education": ( + "Student loan net repayments: higher education total" + ), + "targeted_support_amount_awarded": "Targeted support amount awarded", + "targeted_support_recipients": "Targeted support recipients", + }, + "voa": { + "ct_stock_all_properties": "Council Tax stock: all properties", + "ct_stock_band_a": "Council Tax stock: band A", + "ct_stock_band_b": "Council Tax stock: band B", + "ct_stock_band_c": "Council Tax stock: band C", + "ct_stock_band_d": "Council Tax stock: band D", + "ct_stock_band_e": "Council Tax stock: band E", + "ct_stock_band_f": "Council Tax stock: band F", + "ct_stock_band_g": "Council Tax stock: band G", + "ct_stock_band_h": "Council Tax stock: band H", + }, + } +) + + +CALIBRATION_VARIABLE_LABELS_BY_COUNTRY: Mapping[ + str, Mapping[str, Mapping[str, str]] +] = MappingProxyType( + { + "uk": UK_CALIBRATION_VARIABLE_LABELS, + "us": US_CALIBRATION_VARIABLE_LABELS, + } +) + + +def calibration_variable_label( + country: str, + source_id: str, + variable_id: str, +) -> str | None: + """Return a country-owned label for a provider's statistic category.""" + + country_labels = CALIBRATION_VARIABLE_LABELS_BY_COUNTRY.get(country.strip().lower()) + if country_labels is None: + return None + source_labels = country_labels.get(source_id.strip()) + if source_labels is None: + return None + return source_labels.get(variable_id.strip()) diff --git a/packages/microcosm-calibrate/tests/test_diagnostics.py b/packages/microcosm-calibrate/tests/test_diagnostics.py index e7e683285..c5eb14832 100644 --- a/packages/microcosm-calibrate/tests/test_diagnostics.py +++ b/packages/microcosm-calibrate/tests/test_diagnostics.py @@ -115,7 +115,7 @@ def test_payload_reports_complete_uniform_final_loss_attribution( result = _result(feasible_frame, epochs=1) payload = diagnostics_payload(result) - assert payload["schema_version"] == 6 + assert payload["schema_version"] == 7 assert payload["diagnostic_warnings"] == [] basis = payload["target_loss_basis"] assert basis["formula"] == ( @@ -437,6 +437,16 @@ def test_payload_can_carry_target_registry_identity(feasible_frame) -> None: period=2024, source="IRS SOI 2024", family="irs_soi", + metadata={ + "ledger_selector_source_name": "irs_soi", + "ledger_measure_concept": "irs_soi.adjusted_gross_income", + "ledger_measure_unit": "usd", + "ledger_geography_level": "state", + "ledger_geography_id": "0400000US06", + "ledger_layout_groupby_dimension": "filing_status", + "ledger_layout_groupby_value_id": "single", + "ledger_filter_filing_status": "single", + }, ), ), country="us", @@ -451,10 +461,184 @@ def test_payload_can_carry_target_registry_identity(feasible_frame) -> None: } income = next(row for row in payload["targets"] if row["target_name"] == "income") assert income["period"] == 2024 - assert income["source"] == "IRS SOI 2024" + assert income["source"] == { + "id": "irs_soi", + "label": "IRS Statistics of Income", + "citation": "IRS SOI 2024", + } + assert income["variable"] == { + "id": "adjusted_gross_income", + "label": "Adjusted gross income", + "measure": "total", + } + assert income["dimensions"] == { + "geography_state": "0400000US06", + "filing_status": "single", + } + assert payload["dimensions"] == { + "geography_state": { + "label": "State", + "role": "geography", + "level": "state", + "values": {"0400000US06": "CA"}, + "order": ["0400000US06"], + }, + "filing_status": { + "label": "Filing Status", + "values": {"single": "Single"}, + "order": ["single"], + }, + } assert income["registry"]["family"] == "irs_soi" +def test_registry_diagnostics_publish_uk_geography_and_all_ledger_dimensions( + feasible_frame, +) -> None: + frame, truths = feasible_frame() + geographies = { + "K02000001": "United Kingdom", + "K03000001": "Great Britain", + "E92000001": "England", + "S92000003": "Scotland", + } + registry = TargetRegistry( + tuple( + TargetSpec( + name=f"population_female_{geography_id}", + entity="household", + measure="household_count", + value=truths["population"], + period=2025, + source="ons | Population table | https://example.test/ons", + family="ons_population", + metadata={ + "ledger_selector_source_name": "ons", + "ledger_measure_concept": "ons.population", + "ledger_measure_unit": "count", + "ledger_geography_level": "country", + "ledger_geography_id": geography_id, + "ledger_layout_groupby_dimension": "age_band", + "ledger_layout_groupby_value_id": "18_64", + "ledger_filter_sex": "female", + }, + ) + for geography_id in geographies + ), + country="uk", + ) + result = score_targets(frame, registry.to_target_set()) + + payload = diagnostics_payload(result, target_registry=registry) + + assert payload["targets"][2]["source"] == { + "id": "ons", + "label": "Office for National Statistics", + "citation": "ons | Population table | https://example.test/ons", + "url": "https://example.test/ons", + } + assert payload["targets"][2]["variable"] == { + "id": "population", + "label": "Population", + "measure": "count", + } + assert payload["targets"][2]["dimensions"] == { + "geography_country": "E92000001", + "age_band": "18_64", + "sex": "female", + } + assert payload["dimensions"]["geography_country"] == { + "label": "Country", + "role": "geography", + "level": "country", + "values": geographies, + "order": list(geographies), + } + assert payload["dimensions"]["age_band"]["values"] == {"18_64": "18 64"} + assert payload["dimensions"]["sex"]["values"] == {"female": "Female"} + + +def test_registry_diagnostics_separate_measure_from_variable_category( + feasible_frame, +) -> None: + frame, truths = feasible_frame() + registry = TargetRegistry( + ( + TargetSpec( + name="employment_income_amount_band_100000", + entity="household", + measure="income", + value=truths["income"], + period=2025, + source="HMRC SPI", + family="hmrc", + metadata={ + "ledger_selector_source_name": "hmrc", + "ledger_measure_concept": "hmrc.spi_employment_income_amount", + "ledger_measure_unit": "gbp", + "ledger_filter_total_income_lower_bound": "100000", + }, + ), + TargetSpec( + name="employment_income_count_band_100000", + entity="household", + measure="household_count", + value=truths["population"], + period=2025, + source="HMRC SPI", + family="hmrc", + metadata={ + "ledger_selector_source_name": "hmrc", + "ledger_measure_concept": "hmrc.spi_employment_income_count", + "ledger_measure_unit": "count", + "ledger_filter_total_income_lower_bound": "100000", + }, + ), + ), + country="uk", + ) + result = score_targets(frame, registry.to_target_set()) + + payload = diagnostics_payload(result, target_registry=registry) + + variables = [row["variable"] for row in payload["targets"]] + assert variables == [ + { + "id": "spi_employment_income", + "label": "SPI employment income", + "measure": "total", + }, + { + "id": "spi_employment_income", + "label": "SPI employment income", + "measure": "count", + }, + ] + assert payload["targets"][0]["dimensions"] == {"total_income_lower_bound": "100000"} + assert payload["targets"][1]["dimensions"] == {"total_income_lower_bound": "100000"} + + +def test_registry_diagnostics_reject_missing_compiled_target_identity( + feasible_frame, +) -> None: + result = _result(feasible_frame, epochs=1) + registry = TargetRegistry( + ( + TargetSpec( + name="different", + entity="household", + measure="household_count", + value=1.0, + source="Fixture", + ), + ), + country="us", + ) + + with pytest.raises(ValueError, match="does not contain compiled target row"): + diagnostics_payload(result, target_registry=registry) + + def test_payload_reports_weight_concentration(feasible_frame) -> None: """The accuracy-vs-spread coordinates ship with every calibration.""" result = _result(feasible_frame) diff --git a/packages/microcosm-calibrate/tests/test_provider_labels.py b/packages/microcosm-calibrate/tests/test_provider_labels.py new file mode 100644 index 000000000..0c1d024f9 --- /dev/null +++ b/packages/microcosm-calibrate/tests/test_provider_labels.py @@ -0,0 +1,66 @@ +"""Country-owned provider labels used by schema-7 diagnostics.""" + +from __future__ import annotations + +import pytest + +from microcosm.calibrate import ( + CALIBRATION_PROVIDER_LABELS_BY_COUNTRY, + UK_CALIBRATION_PROVIDER_LABELS, + US_CALIBRATION_PROVIDER_LABELS, + calibration_provider_label, +) + + +def test_current_us_schema_7_sources_have_provider_labels() -> None: + current_source_ids = { + "bea_nipa", + "cbo", + "census_pep", + "census_stc", + "cms_aca", + "cms_medicaid", + "cms_medicare", + "federal_reserve_z1", + "hhs_acf_liheap", + "hhs_acf_tanf", + "irs_soi", + "jct", + "ssa_ssi_monthly", + "ssa_supplement", + "unspecified", + "usda_snap", + } + + assert current_source_ids <= US_CALIBRATION_PROVIDER_LABELS.keys() + + +def test_current_uk_schema_7_sources_have_provider_labels() -> None: + current_source_ids = { + "dwp", + "hmrc", + "isc", + "obr", + "ons", + "scotgov", + "slc", + "voa", + } + + assert current_source_ids == UK_CALIBRATION_PROVIDER_LABELS.keys() + + +def test_provider_label_resolution_is_country_specific() -> None: + assert calibration_provider_label("us", "irs_soi") == ("IRS Statistics of Income") + assert calibration_provider_label("UK", "obr") == ( + "Office for Budget Responsibility" + ) + assert calibration_provider_label("be", "statbel") is None + assert calibration_provider_label("uk", "irs_soi") is None + + +def test_provider_label_registries_are_immutable() -> None: + with pytest.raises(TypeError): + CALIBRATION_PROVIDER_LABELS_BY_COUNTRY["be"] = {} # type: ignore[index] + with pytest.raises(TypeError): + UK_CALIBRATION_PROVIDER_LABELS["new"] = "New" # type: ignore[index] diff --git a/packages/microcosm-calibrate/tests/test_variable_labels.py b/packages/microcosm-calibrate/tests/test_variable_labels.py new file mode 100644 index 000000000..b49dff779 --- /dev/null +++ b/packages/microcosm-calibrate/tests/test_variable_labels.py @@ -0,0 +1,182 @@ +"""Country-owned Level 2 labels used by schema-7 diagnostics.""" + +from __future__ import annotations + +import pytest + +from microcosm.calibrate import ( + CALIBRATION_VARIABLE_LABELS_BY_COUNTRY, + UK_CALIBRATION_VARIABLE_LABELS, + US_CALIBRATION_VARIABLE_LABELS, + calibration_variable_label, +) + + +def _pairs(value: str) -> set[tuple[str, str]]: + return {tuple(line.split("|", 1)) for line in value.splitlines() if line.strip()} + + +_CURRENT_US_VARIABLES = _pairs( + """bea_nipa|proprietors_income_with_inventory_valuation_and_capital_consumption_adjustments +bea_nipa|wages_and_salaries +cbo|adjusted_gross_income_projection +cbo|net_business_income_projection +cbo|net_capital_gain_projection +cbo|qualified_dividend_income_projection +cbo|wages_and_salaries_projection +census_pep|resident_population +census_stc|individual_income_tax_collections +cms_aca|aptc_consumers +cms_aca|marketplace_plan_selections +cms_medicaid|total_chip_enrollment +cms_medicaid|total_medicaid_chip_enrollment +cms_medicaid|total_medicaid_enrollment +cms_medicare|part_b_premium_income +federal_reserve_z1|federal_reserve.z1.households_nonprofits_net_worth +hhs_acf_liheap|households_served_by_state_programs +hhs_acf_tanf|cash_assistance_expenditures +irs_soi|adjusted_gross_income +irs_soi|assigned_aca_ptc +irs_soi|business_net_profits +irs_soi|capital_gains_gross +irs_soi|charitable_deduction +irs_soi|count +irs_soi|ctc +irs_soi|deductible_mortgage_interest +irs_soi|eitc +irs_soi|employment_income +irs_soi|estate_income +irs_soi|estate_losses +irs_soi|income_tax +irs_soi|income_tax_before_credits +irs_soi|interest_deduction +irs_soi|ira_distributions +irs_soi|itemized_taxable_income_deductions +irs_soi|medical_expense_deduction +irs_soi|non_sch_d_capital_gains +irs_soi|ordinary_dividend_income +irs_soi|partnership_and_s_corp_income +irs_soi|qualified_dividends +irs_soi|real_estate_taxes +irs_soi|refundable_ctc +irs_soi|rent_and_royalty_net_income +irs_soi|salt_deduction +irs_soi|tax_exempt_interest_income +irs_soi|taxable_income +irs_soi|taxable_interest_income +irs_soi|taxable_pension_income +irs_soi|taxable_social_security +irs_soi|tip_income +irs_soi|unemployment_compensation +jct|individual_tax_expenditure_revenue_loss +ssa_ssi_monthly|ssa.ssi_federal_payment_recipient +ssa_supplement|ssa.annual_oasdi_or_ssi_payment +ssa_supplement|ssa.ssi_payment +ssa_supplement|ssa.ssi_recipient +unspecified|selection_mass_protection.keogh_distributions +usda_snap|average_monthly_households +usda_snap|total_benefits""" +) + + +_CURRENT_UK_VARIABLES = _pairs( + """dwp|esa_claimants +dwp|jsa_claimants +dwp|uc_benefit_units +dwp|uc_capped_households_total +dwp|uc_households +dwp|uc_tcl_children_by_children_3 +dwp|uc_tcl_children_by_children_4 +dwp|uc_tcl_children_by_children_5 +dwp|uc_tcl_children_by_children_6_plus +dwp|uc_tcl_children_by_disability_claimant_pip +dwp|uc_tcl_children_by_disability_disabled_child_element +dwp|uc_tcl_headline_children_affected_by_policy +dwp|uc_tcl_headline_children_within_affected_households +dwp|uc_tcl_headline_households_affected +dwp|uc_tcl_households_by_children_3 +dwp|uc_tcl_households_by_children_4 +dwp|uc_tcl_households_by_children_5 +dwp|uc_tcl_households_by_children_6_plus +dwp|uc_tcl_households_by_disability_claimant_pip +dwp|uc_tcl_households_by_disability_disabled_child_element +hmrc|cgt_gains_total +hmrc|cgt_taxpayers_total +hmrc|salary_sacrifice_pension_users +hmrc|spi_dividend_income +hmrc|spi_employment_income +hmrc|spi_private_pension_income +hmrc|spi_property_income +hmrc|spi_self_employment_income +hmrc|spi_state_pension +isc|pupils_at_member_schools +obr|efo_expenditure +obr|efo_receipts +ons|household_interest_resources +ons|households_by_type +ons|mid_year_population_estimate +ons|nbs_land_value_households +ons|nbs_land_value_nfc +ons|nbs_land_value_total +ons|pse_headcount_total_public_sector +scotgov|chargeable_dwellings_band_a +scotgov|chargeable_dwellings_band_b +scotgov|chargeable_dwellings_band_c +scotgov|chargeable_dwellings_band_d +scotgov|chargeable_dwellings_band_e +scotgov|chargeable_dwellings_band_f +scotgov|chargeable_dwellings_band_g +scotgov|chargeable_dwellings_band_h +scotgov|chargeable_dwellings_total +scotgov|social_security_assistance_spending +slc|maintenance_loan_amount_paid +slc|maintenance_loan_recipients +slc|student_loan_borrowers +slc|student_loan_net_repayments_plan_1 +slc|student_loan_net_repayments_plan_2_full_time +slc|student_loan_net_repayments_total_higher_education +slc|targeted_support_amount_awarded +slc|targeted_support_recipients +voa|ct_stock_all_properties +voa|ct_stock_band_a +voa|ct_stock_band_b +voa|ct_stock_band_c +voa|ct_stock_band_d +voa|ct_stock_band_e +voa|ct_stock_band_f +voa|ct_stock_band_g +voa|ct_stock_band_h""" +) + + +@pytest.mark.parametrize( + ("country", "pairs"), + [("us", _CURRENT_US_VARIABLES), ("uk", _CURRENT_UK_VARIABLES)], +) +def test_current_schema_7_variables_have_labels( + country: str, + pairs: set[tuple[str, str]], +) -> None: + missing = { + (source_id, variable_id) + for source_id, variable_id in pairs + if calibration_variable_label(country, source_id, variable_id) is None + } + + assert missing == set() + + +def test_variable_label_resolution_is_scoped_by_country_and_provider() -> None: + assert calibration_variable_label("us", "irs_soi", "eitc") == "EITC" + assert calibration_variable_label("UK", "obr", "efo_receipts") == ("EFO receipts") + assert calibration_variable_label("uk", "hmrc", "efo_receipts") is None + assert calibration_variable_label("be", "statbel", "population") is None + + +def test_variable_label_registries_are_immutable() -> None: + with pytest.raises(TypeError): + CALIBRATION_VARIABLE_LABELS_BY_COUNTRY["be"] = {} # type: ignore[index] + with pytest.raises(TypeError): + UK_CALIBRATION_VARIABLE_LABELS["obr"]["new"] = "New" # type: ignore[index] + with pytest.raises(TypeError): + US_CALIBRATION_VARIABLE_LABELS["irs_soi"] = {} # type: ignore[index] diff --git a/packages/microcosm-data/src/microcosm/data/contract.py b/packages/microcosm-data/src/microcosm/data/contract.py index 91376f846..ebe6c32d6 100644 --- a/packages/microcosm-data/src/microcosm/data/contract.py +++ b/packages/microcosm-data/src/microcosm/data/contract.py @@ -125,11 +125,13 @@ ) # Lockstep with microcosm.calibrate.diagnostics.CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION -# (schema 6 = final per-target loss attribution plus warning-only degradation). +# (schema 7 = structured source, variable, and dimension identity for +# registry-backed release diagnostics). # microcosm-data cannot import # microcosm-calibrate (dependency direction), so the builder test suite pins the # two constants equal — see test_calibration_diagnostics_schema_lockstep. -CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION = 6 +CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION = 7 +_SUPPORTED_CALIBRATION_DIAGNOSTICS_SCHEMA_VERSIONS = frozenset({6, 7}) US_SOURCE_COVERAGE_DIAGNOSTICS_FILE = "us_source_coverage.json" SOURCE_COVERAGE_DIAGNOSTICS_SCHEMA_VERSION = 1 _SHA256_RE = re.compile(r"^[0-9a-f]{64}$") @@ -3165,14 +3167,16 @@ def _check_calibration_diagnostics( "calibration_diagnostics.json grandfathered June UK release " f"requires legacy schema version 2, got {schema_version!r}." ) - elif ( - not grandfathered_uk_june - and schema_version != CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION + elif not grandfathered_uk_june and ( + isinstance(schema_version, bool) + or not isinstance(schema_version, int) + or schema_version not in _SUPPORTED_CALIBRATION_DIAGNOSTICS_SCHEMA_VERSIONS ): failures.append( f"calibration_diagnostics.json 'schema_version' is {schema_version!r}; " - f"this library publishes version " - f"{CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION}." + "supported versions are " + f"{sorted(_SUPPORTED_CALIBRATION_DIAGNOSTICS_SCHEMA_VERSIONS)} and " + f"this library publishes version {CALIBRATION_DIAGNOSTICS_SCHEMA_VERSION}." ) expected_sections = { @@ -3205,6 +3209,15 @@ def _check_calibration_diagnostics( ) targets = diagnostics.get("targets") + dimension_definitions = diagnostics.get("dimensions") + if schema_version == 7 and not isinstance(dimension_definitions, Mapping): + failures.append( + "calibration_diagnostics.json schema 7 requires a top-level " + "'dimensions' object." + ) + dimension_definitions = {} + if schema_version == 7 and isinstance(dimension_definitions, Mapping): + _check_diagnostics_dimension_definitions(dimension_definitions, failures) if isinstance(targets, list): surface = diagnostics.get("target_surface") if isinstance(surface, Mapping) and surface.get("n_targets") != len(targets): @@ -3241,6 +3254,17 @@ def _check_calibration_diagnostics( "calibration_diagnostics.json target row " f"{index} is missing non-empty 'source'." ) + if schema_version == 7: + _check_structured_diagnostics_target( + target, + index=index, + dimension_definitions=( + dimension_definitions + if isinstance(dimension_definitions, Mapping) + else {} + ), + failures=failures, + ) if not grandfathered_uk_june: if not isinstance(target.get("measure"), Mapping): failures.append( @@ -3260,6 +3284,127 @@ def _check_calibration_diagnostics( ) +def _check_diagnostics_dimension_definitions( + dimensions: Mapping, + failures: list[str], +) -> None: + """Validate the schema-7 dimension dictionary.""" + + for dimension_id, definition in dimensions.items(): + owner = f"calibration_diagnostics.json dimension {dimension_id!r}" + if not isinstance(dimension_id, str) or not dimension_id: + failures.append( + "calibration_diagnostics.json dimension ids must be non-empty strings." + ) + continue + if not isinstance(definition, Mapping): + failures.append(f"{owner} must be an object.") + continue + if ( + not isinstance(definition.get("label"), str) + or not str(definition.get("label")).strip() + ): + failures.append(f"{owner} requires a non-empty string 'label'.") + role = definition.get("role") + if role not in {None, "geography"}: + failures.append(f"{owner} has unsupported role {role!r}.") + if role == "geography" and ( + not isinstance(definition.get("level"), str) + or not str(definition.get("level")).strip() + ): + failures.append( + f"{owner} with role 'geography' requires a non-empty string 'level'." + ) + value_labels = definition.get("values") + if value_labels is not None and not isinstance(value_labels, Mapping): + failures.append(f"{owner} 'values' must be an object when provided.") + elif isinstance(value_labels, Mapping): + for raw_value, label in value_labels.items(): + if ( + not isinstance(raw_value, str) + or not raw_value + or not isinstance(label, str) + or not label.strip() + ): + failures.append( + f"{owner} value labels must map non-empty strings to " + "non-empty strings." + ) + break + order = definition.get("order") + if order is not None and ( + not isinstance(order, list) + or any(not isinstance(value, str) or not value for value in order) + or len(set(order)) != len(order) + ): + failures.append( + f"{owner} 'order' must be a list of unique non-empty strings." + ) + + +def _check_structured_diagnostics_target( + target: Mapping, + *, + index: int, + dimension_definitions: Mapping, + failures: list[str], +) -> None: + """Validate complete structured identity on one schema-7 target row.""" + + owner = f"calibration_diagnostics.json target row {index}" + for field in ("source", "variable"): + value = target.get(field) + if ( + not isinstance(value, Mapping) + or not isinstance(value.get("id"), str) + or not str(value.get("id")).strip() + ): + failures.append( + f"{owner} schema 7 requires {field!r} to be an object with a " + "non-empty string 'id'." + ) + if ( + isinstance(value, Mapping) + and "label" in value + and ( + not isinstance(value.get("label"), str) + or not str(value.get("label")).strip() + ) + ): + failures.append( + f"{owner} schema 7 {field} 'label' must be a non-empty string " + "when provided." + ) + values = target.get("dimensions") + if not isinstance(values, Mapping): + failures.append(f"{owner} schema 7 requires a 'dimensions' object.") + return + geography_count = 0 + for dimension_id, raw_value in values.items(): + if dimension_id not in dimension_definitions: + failures.append(f"{owner} references undefined dimension {dimension_id!r}.") + continue + if not isinstance(raw_value, str) or not raw_value.strip(): + failures.append( + f"{owner} dimension {dimension_id!r} must have a non-empty " + "string value." + ) + continue + definition = dimension_definitions.get(dimension_id) + if not isinstance(definition, Mapping): + continue + if definition.get("role") == "geography": + geography_count += 1 + labels = definition.get("values") + if isinstance(labels, Mapping) and raw_value not in labels: + failures.append( + f"{owner} geography value {raw_value!r} has no label in " + f"dimension {dimension_id!r}." + ) + if geography_count > 1: + failures.append(f"{owner} may populate at most one geography-role dimension.") + + def _uk_non_negative_int( value: object, *, diff --git a/packages/microcosm-data/tests/test_contract.py b/packages/microcosm-data/tests/test_contract.py index 9c536c332..81bf85db2 100644 --- a/packages/microcosm-data/tests/test_contract.py +++ b/packages/microcosm-data/tests/test_contract.py @@ -2341,7 +2341,7 @@ def test_legacy_diagnostics_exemption_is_scoped_to_the_exact_june_id( payload=diagnostics, ) - with pytest.raises(ReleaseContractError, match="publishes version 6"): + with pytest.raises(ReleaseContractError, match="publishes version 7"): validate_release_dir(directory) @@ -3548,6 +3548,109 @@ def test_malformed_calibration_diagnostics_is_rejected( assert "targets" in failures +def test_schema_7_structured_calibration_diagnostics_are_accepted( + release_dir: Path, +) -> None: + diagnostics = _calibration_diagnostics() + diagnostics["schema_version"] = 7 + diagnostics["dimensions"] = { + "geography_country": { + "label": "Country", + "role": "geography", + "level": "country", + "values": {"0100000US": "United States"}, + "order": ["0100000US"], + } + } + for row in diagnostics["targets"]: + row["source"] = { + "id": "fixture", + "label": "Fixture provider", + "citation": row["source"], + } + row["variable"] = { + "id": row["target_name"], + "label": "Fixture statistic", + } + row["dimensions"] = {"geography_country": "0100000US"} + _write_json_and_refresh_manifest_hash( + release_dir, + filename="calibration_diagnostics.json", + artifact_key="calibration_diagnostics", + payload=diagnostics, + ) + + validate_release_dir(release_dir) + + +@pytest.mark.parametrize("field", ["source", "variable"]) +def test_schema_7_rejects_an_empty_identity_label( + release_dir: Path, + field: str, +) -> None: + diagnostics = _calibration_diagnostics() + diagnostics["schema_version"] = 7 + diagnostics["dimensions"] = {} + for row in diagnostics["targets"]: + row["source"] = {"id": "fixture"} + row["variable"] = {"id": row["target_name"]} + row[field]["label"] = " " + row["dimensions"] = {} + _write_json_and_refresh_manifest_hash( + release_dir, + filename="calibration_diagnostics.json", + artifact_key="calibration_diagnostics", + payload=diagnostics, + ) + + with pytest.raises(ReleaseContractError) as excinfo: + validate_release_dir(release_dir) + + failures = "\n".join(excinfo.value.failures) + assert f"{field} 'label' must be a non-empty string" in failures + + +def test_schema_7_rejects_partial_identity_and_multiple_geographies( + release_dir: Path, +) -> None: + diagnostics = _calibration_diagnostics() + diagnostics["schema_version"] = 7 + diagnostics["dimensions"] = { + "geography_country": { + "label": "Country", + "role": "geography", + "level": "country", + "values": {"0100000US": "United States"}, + }, + "geography_state": { + "label": "State", + "role": "geography", + "level": "state", + "values": {"0400000US06": "CA"}, + }, + } + for row in diagnostics["targets"]: + row["source"] = {"id": "fixture"} + row["variable"] = {"label": "Missing id"} + row["dimensions"] = { + "geography_country": "0100000US", + "geography_state": "0400000US06", + } + _write_json_and_refresh_manifest_hash( + release_dir, + filename="calibration_diagnostics.json", + artifact_key="calibration_diagnostics", + payload=diagnostics, + ) + + with pytest.raises(ReleaseContractError) as excinfo: + validate_release_dir(release_dir) + + failures = "\n".join(excinfo.value.failures) + assert "'variable' to be an object with a non-empty string 'id'" in failures + assert "at most one geography-role dimension" in failures + + def test_malformed_us_source_coverage_diagnostics_is_rejected( release_dir: Path, ) -> None: diff --git a/tools/assemble_uk_release_dir.py b/tools/assemble_uk_release_dir.py index 439469e39..f9eab98b5 100644 --- a/tools/assemble_uk_release_dir.py +++ b/tools/assemble_uk_release_dir.py @@ -23,6 +23,7 @@ from microcosm.build.uk_runtime.national_frame import load_uk_national_frame from microcosm.build.uk_runtime.release_identity import UK_NATIONAL_RELEASE_ID +from microcosm.calibrate import UK_CALIBRATION_PROVIDER_LABELS from microcosm.data.contract import validate_release_dir _ATTEMPT_PREFIX = "uk-frs-calibration-attempt-" @@ -463,12 +464,7 @@ def artifact(kind: str, path: str, sha256: str) -> dict[str, str]: "UK national calibration pipeline release assembled from attempt " f"{attempt_id} at immutable cut {cut_tag}." ), - "publisher_labels": { - "obr": "Office for Budget Responsibility", - "hmrc": "HM Revenue and Customs", - "ons": "Office for National Statistics", - "dwp": "Department for Work and Pensions", - }, + "publisher_labels": dict(UK_CALIBRATION_PROVIDER_LABELS), } release_manifest_path = release_dir / "release_manifest.json" _write_json(release_manifest_path, release_manifest) diff --git a/tools/generate_uk_target_references.py b/tools/generate_uk_target_references.py index 79c3998ae..81b69282b 100644 --- a/tools/generate_uk_target_references.py +++ b/tools/generate_uk_target_references.py @@ -16,6 +16,7 @@ from typing import Any from microcosm.build.target_reference_authoring import ( + AuthoredTargetReferences, TargetReferenceAuthoringConfig, author_target_references, target_references_resource, @@ -91,10 +92,39 @@ "classes and geography-pin decisions are recorded in " "uk/target_reference_membership.json. metadata.measure_kind records that " "measures are prepared columns produced from the contract binding payload " - "referenced by metadata.contract_target_id." + "referenced by metadata.contract_target_id. OBR references also declare " + "metadata.diagnostic_variable_id as efo_receipts or efo_expenditure so " + "schema-7 consumers group the forecast lines by their source table." ) NATIONAL_GEOGRAPHY_LEVELS = frozenset({"country", "region"}) +OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID = { + "obr.income_tax": "efo_receipts", + "obr.ni": "efo_receipts", + "obr.ni_employee": "efo_receipts", + "obr.ni_employer": "efo_receipts", + "obr.ni_self_employed": "efo_receipts", + "obr.vat": "efo_receipts", + "obr.fuel_duties": "efo_receipts", + "obr.capital_gains_tax": "efo_receipts", + "obr.sdlt": "efo_receipts", + "obr.attendance_allowance": "efo_expenditure", + "obr.carers_allowance": "efo_expenditure", + "obr.child_benefit": "efo_expenditure", + "obr.council_tax": "efo_expenditure", + "obr.esa": "efo_expenditure", + "obr.housing_benefit": "efo_expenditure", + "obr.jobseekers_allowance": "efo_expenditure", + "obr.pension_credit": "efo_expenditure", + "obr.pip": "efo_expenditure", + "obr.state_pension": "efo_expenditure", + "obr.statutory_maternity_pay": "efo_expenditure", + "obr.tv_licence_fee": "efo_expenditure", + "obr.universal_credit_in_cap": "efo_expenditure", + "obr.universal_credit_outside_cap": "efo_expenditure", + "obr.winter_fuel_allowance": "efo_expenditure", +} + def main() -> None: args = _parser().parse_args() @@ -124,6 +154,7 @@ def main() -> None: source_fact_feed=str(args.ledger_facts), ) authored = author_target_references(contract, facts, config) + authored = _add_diagnostic_variable_ids(authored) _add_uk_membership_accounting(authored.membership_report, authored.references) resource = target_references_resource( country="uk", @@ -140,6 +171,25 @@ def main() -> None: ) +def _add_diagnostic_variable_ids( + authored: AuthoredTargetReferences, +) -> AuthoredTargetReferences: + """Assign producer-defined dashboard categories to OBR target references.""" + + references: list[dict[str, Any]] = [] + for reference in authored.references: + metadata = dict(reference["metadata"]) + target_id = str(metadata["contract_target_id"]) + variable_id = OBR_DIAGNOSTIC_VARIABLE_BY_TARGET_ID.get(target_id) + if variable_id is not None: + metadata["diagnostic_variable_id"] = variable_id + references.append({**reference, "metadata": metadata}) + return AuthoredTargetReferences( + references=tuple(references), + membership_report=authored.membership_report, + ) + + def _parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser() parser.add_argument("--contract", type=Path, required=True)