diff --git a/changelog.d/681-uk-was-wealth.added.md b/changelog.d/681-uk-was-wealth.added.md new file mode 100644 index 000000000..b551d9e4f --- /dev/null +++ b/changelog.d/681-uk-was-wealth.added.md @@ -0,0 +1 @@ +Add the UK WAS wealth imputation and regional property-uprating source-stage layer. diff --git a/packages/microcosm-build/src/microcosm/build/source_manifest.py b/packages/microcosm-build/src/microcosm/build/source_manifest.py index 92823f9f2..6d0c04413 100644 --- a/packages/microcosm-build/src/microcosm/build/source_manifest.py +++ b/packages/microcosm-build/src/microcosm/build/source_manifest.py @@ -49,6 +49,7 @@ "assign_clipped_normal", "assign_uniform_draw", "aggregate_person_to_benunit", + "allocate_within_group_waterfall", "allocate_zero_weight_prior_mass", "annualize_periodic_amounts", "assemble_group_entities", @@ -91,6 +92,7 @@ "fit_vehicle_model", "fit_weighted_imputer", "fit_weighted_qrf", + "fit_weighted_qrf_chain", "fit_weighted_qrf_stage1", "fit_weighted_qrf_stage2", "fold_into", @@ -127,6 +129,7 @@ "strict_read_private_table", "support_clip", "uprate", + "uprate_to_regional_reference", "verify_certified_candidate", "verify_pinned_hmrc_source_pair", "zero_when_false", diff --git a/packages/microcosm-build/src/microcosm/build/uk/country_package.json b/packages/microcosm-build/src/microcosm/build/uk/country_package.json index 04ea02f8a..d5e83e02b 100644 --- a/packages/microcosm-build/src/microcosm/build/uk/country_package.json +++ b/packages/microcosm-build/src/microcosm/build/uk/country_package.json @@ -12,12 +12,14 @@ "hmrc_income_release_gate_report.json", "hmrc_income_replay_report.json", "hmrc_income_source_stages.json", + "regional_land_values.json", "source_stages.json", "take_up_contract.json", "input_mass_reviewed_exclusions.json", "national_staging_build_record.json", "qrf_tail_reviewed_exclusions.json", "release_input_coverage_manifest.json", + "was_wealth_support_bounds.json", "uk_local_target_census.json" ] } diff --git a/packages/microcosm-build/src/microcosm/build/uk/gates.json b/packages/microcosm-build/src/microcosm/build/uk/gates.json index 9e9df762f..e8f576861 100644 --- a/packages/microcosm-build/src/microcosm/build/uk/gates.json +++ b/packages/microcosm-build/src/microcosm/build/uk/gates.json @@ -109,6 +109,16 @@ "parameters": {}, "notes": "Every UK source-stage column declared in source_stages.json nonnegative_outputs must be finite and non-negative on the terminal frame; this makes the shared nonnegative column gate live for UK." }, + { + "id": "uk_support", + "gate": "support", + "phase": "terminal", + "criticality": "release_blocking", + "parameters": { + "support_bounds_resource": "was_wealth_support_bounds.json" + }, + "notes": "Every WAS-imputed wealth output must remain inside the committed disclosure-safe donor support bounds. Exact donor ranges are used in-run for clipping; this release gate uses outward-rounded reviewed bounds so unit-record WAS minima and maxima are never committed." + }, { "id": "uk_export_surface", "gate": "export_surface", @@ -119,6 +129,7 @@ "benunit.child_benefit_opts_out", "household.bus_fare_spending", "household.bus_subsidy_spending", + "household.cash_isa", "household.clone_index", "household.constituency_code_oa", "household.consumer_debt", @@ -136,6 +147,7 @@ "household.property_purchased", "household.rail_usage", "household.region_code_oa", + "household.stocks_and_shares_isa", "person.aa_category", "person.age_started_or_accepted_current_education_or_training", "person.attends_private_school_random_draw", diff --git a/packages/microcosm-build/src/microcosm/build/uk/input_mass_reviewed_exclusions.json b/packages/microcosm-build/src/microcosm/build/uk/input_mass_reviewed_exclusions.json index 95a178527..298b6b8c8 100644 --- a/packages/microcosm-build/src/microcosm/build/uk/input_mass_reviewed_exclusions.json +++ b/packages/microcosm-build/src/microcosm/build/uk/input_mass_reviewed_exclusions.json @@ -9,6 +9,13 @@ "adjudication": "microcosm#630", "approved_on": "2026-08-17", "expires_on": "2027-02-17" + }, + "owned_land": { + "reason": "Sparse heavy-tailed WAS donor column (0.7 percent weighted nonzero share) whose weighted total is dominated by a handful of large farm/estate records: the E5 stability receipt (data/ukds/acceptance/e5/owned_land_stability_receipt.json) measures a 37.7 percent national and 2.41x London swing between adjacent seeds, the same realization-variance class the archived incumbent data repo records at uk-data#448 (4.6x Wales swing across releases). Register parity at this grain is not meaningful until the whole-spine comparison; the one-month expiry enforces the end-of-workstream revisit registered on microcosm#145 (winsorised donor or separate land imputation are the candidate remedies).", + "approved_by": "juaristi22", + "adjudication": "microcosm#714", + "approved_on": "2026-08-19", + "expires_on": "2026-09-19" } } } diff --git a/packages/microcosm-build/src/microcosm/build/uk/regional_land_values.json b/packages/microcosm-build/src/microcosm/build/uk/regional_land_values.json new file mode 100644 index 000000000..a62830c92 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/uk/regional_land_values.json @@ -0,0 +1,72 @@ +{ + "version": 1, + "country": "uk", + "policy": "Public regional land-value reference for deterministic post-WAS property uprating.", + "source": { + "provenance": "Incumbent public regional_land_values.csv re-homed as a JSON country-package resource for Microcosm. Values combine MHCLG dwellings and ONS UK House Price Index average prices for December 2025.", + "citation_urls": [ + "https://www.gov.uk/government/collections/dwelling-stock-including-vacants", + "https://www.ons.gov.uk/economy/inflationandpriceindices/bulletins/housepriceindex/december2025" + ] + }, + "values": [ + { + "region": "NORTH_EAST", + "avg_house_price": 165257, + "dwellings": 1280700 + }, + { + "region": "NORTH_WEST", + "avg_house_price": 217428, + "dwellings": 3441730 + }, + { + "region": "YORKSHIRE", + "avg_house_price": 208447, + "dwellings": 2530680 + }, + { + "region": "EAST_MIDLANDS", + "avg_house_price": 243632, + "dwellings": 2213820 + }, + { + "region": "WEST_MIDLANDS", + "avg_house_price": 246141, + "dwellings": 2619300 + }, + { + "region": "EAST_OF_ENGLAND", + "avg_house_price": 338002, + "dwellings": 2841060 + }, + { + "region": "LONDON", + "avg_house_price": 551294, + "dwellings": 3813790 + }, + { + "region": "SOUTH_EAST", + "avg_house_price": 378800, + "dwellings": 4138200 + }, + { + "region": "SOUTH_WEST", + "avg_house_price": 301226, + "dwellings": 2683030 + }, + { + "region": "WALES", + "avg_house_price": 214883, + "dwellings": 1476690 + }, + { + "region": "SCOTLAND", + "avg_house_price": 190649, + "dwellings": 2602545 + } + ], + "chronicle": [ + "2026-08-18: added for microcosm#681 E5 WAS wealth and regional property uprating." + ] +} diff --git a/packages/microcosm-build/src/microcosm/build/uk/release_input_coverage_manifest.json b/packages/microcosm-build/src/microcosm/build/uk/release_input_coverage_manifest.json index 741674100..6a12fa68f 100644 --- a/packages/microcosm-build/src/microcosm/build/uk/release_input_coverage_manifest.json +++ b/packages/microcosm-build/src/microcosm/build/uk/release_input_coverage_manifest.json @@ -562,6 +562,57 @@ "spi_prior_national_household_mass_share": 0.5, "stage": "hmrc_spi_income", "status": "required_at_build" + }, + "regional_property_uprating": { + "base_candidate_sha256": "f17306ccb2aad7ff0130be3589b560afb2e2a12a943570911cd0c77f07934833", + "base_candidate_tier": "frs", + "effective_mass_requirements": {}, + "output_weight_kind": "importance", + "outputs": [], + "required_mass_change_reason": "E5 source-stage transform preserves household rows and typed household weights; total household mass is conserved.", + "rewrites": [ + "main_residence_value", + "property_wealth" + ], + "source_manifest": "source_stages.json", + "source_manifest_sha256": "c8eba47691c3e3b4abe833b2c2816acfc091c7039df5131e3768eb4fce393eb3", + "source_vintages": { + "source": "MHCLG dwellings and ONS UK House Price Index December 2025 regional average prices.", + "survey": "Public regional property reference" + }, + "stage": "regional_property_uprating", + "status": "required_at_build" + }, + "was_wealth": { + "base_candidate_sha256": "f17306ccb2aad7ff0130be3589b560afb2e2a12a943570911cd0c77f07934833", + "base_candidate_tier": "frs", + "effective_mass_requirements": {}, + "output_weight_kind": "importance", + "outputs": [ + "owned_land", + "property_wealth", + "corporate_wealth", + "gross_financial_wealth", + "net_financial_wealth", + "main_residence_value", + "other_residential_property_value", + "non_residential_property_value", + "savings", + "num_vehicles", + "cash_isa", + "stocks_and_shares_isa", + "student_loan_balance" + ], + "required_mass_change_reason": "E5 source-stage transform preserves household rows and typed household weights; total household mass is conserved.", + "rewrites": [], + "source_manifest": "source_stages.json", + "source_manifest_sha256": "c8eba47691c3e3b4abe833b2c2816acfc091c7039df5131e3768eb4fce393eb3", + "source_vintages": { + "source": "Office for National Statistics Wealth and Assets Survey, UK Data Service SN 7215, DOI 10.5255/UKDA-SN-7215-20; local licensed 2006-22 household tab.", + "survey": "Wealth and Assets Survey round 8" + }, + "stage": "was_wealth", + "status": "required_at_build" } }, "reference": { diff --git a/packages/microcosm-build/src/microcosm/build/uk/source_stages.json b/packages/microcosm-build/src/microcosm/build/uk/source_stages.json index 393def1b1..31f8d0929 100644 --- a/packages/microcosm-build/src/microcosm/build/uk/source_stages.json +++ b/packages/microcosm-build/src/microcosm/build/uk/source_stages.json @@ -803,6 +803,212 @@ ], "notes": "Identity-keyed seed 0 streams replace the incumbent BRMA seed 0 sequential generator. Household collapse uses salt brma:household_pick." }, + { + "stage": "was_wealth", + "survey": "Wealth and Assets Survey round 8", + "source": "Office for National Statistics Wealth and Assets Survey, UK Data Service SN 7215, DOI 10.5255/UKDA-SN-7215-20; local licensed 2006-22 household tab.", + "grain": "household", + "base_candidate": { + "filename": "populace_uk_2023.h5", + "revision": "populace-uk-2023-dd68c73-4aa4b14-20260619T023711Z", + "sha256": "f17306ccb2aad7ff0130be3589b560afb2e2a12a943570911cd0c77f07934833", + "tier": "frs" + }, + "artifacts": [ + { + "role": "was_qrf_donor", + "kind": "private_microdata", + "ukds_study_number": 7215, + "doi": "10.5255/UKDA-SN-7215-20", + "filename": "was_round_8_hhold_eul_may_2025_230525.tab", + "sha256": "18b3eb980c02c99f3d8a3254af859bee31682b2bdc11703877677292b3ce9374", + "size_bytes": 39073613, + "access": "private_local_input", + "locator": "caller-supplied local input", + "runtime_sha256_required": true + } + ], + "operations": [ + { + "kind": "derive", + "scope": "strict WAS donor cleaning with lower-case exact column matching and no fuzzy r/w fallback", + "fillna": 0, + "direct_maps": { + "owned_land": "DVLUKValR8_sum", + "property_wealth": "DVPropertyR8", + "gross_financial_wealth": "HFINWR8_SUM", + "net_financial_wealth": "HFINWNTR8_Sum", + "main_residence_value": "DVhvalueR8", + "other_residential_property_value": "DVHseValR8_sum", + "non_residential_property_value": "DVBlDValR8_sum", + "savings": "DVSaValR8_aggr", + "num_vehicles": "vcarnr8", + "weight": "R8xshhwgt" + }, + "derived": { + "corporate_wealth_excl_isa": "(totalpenr8_aggr - dvvaldbt_scaper8_aggr) + DVFESHARESR8_aggr + DVFShUKVR8_aggr + DVFCollVR8_aggr", + "stocks_and_shares_isa": "DVIISAVR8_aggr", + "cash_isa": "DVCISAVR8_aggr", + "student_loan_balance": "Tot_LosR8_aggr - Tot_los_exc_SLCR8_aggr", + "is_renting": "DVPriRntR8 == 1", + "region": "GORR8 via incumbent REGIONS map" + } + }, + { + "kind": "materialize_rules_engine_predictors", + "predictors": [ + "household_net_income", + "num_adults", + "num_children", + "private_pension_income", + "employment_income", + "self_employment_income", + "capital_income", + "is_renting" + ], + "consumed_only": true + }, + { + "kind": "fit_weighted_qrf_chain", + "weights": "explicit", + "seed": 0, + "n_estimators": 100, + "predictors": [ + "household_net_income", + "num_adults", + "num_children", + "private_pension_income", + "employment_income", + "self_employment_income", + "capital_income", + "num_bedrooms", + "council_tax", + "is_renting", + "region" + ], + "categorical_predictors": [ + "region", + "is_renting" + ], + "region_remap": { + "NORTHERN_IRELAND": "WALES" + }, + "integer_outputs": [ + "num_vehicles" + ], + "chain_order": [ + "owned_land", + "property_wealth", + "corporate_wealth_excl_isa", + "stocks_and_shares_isa", + "corporate_wealth", + "gross_financial_wealth", + "net_financial_wealth", + "main_residence_value", + "other_residential_property_value", + "non_residential_property_value", + "savings", + "num_vehicles", + "student_loan_balance", + "cash_isa" + ] + }, + { + "kind": "fold_into", + "output": "corporate_wealth", + "inputs": [ + "corporate_wealth_excl_isa", + "stocks_and_shares_isa" + ], + "drop_inputs": [ + "corporate_wealth_excl_isa" + ] + }, + { + "kind": "support_clip", + "range": "donor_realized" + }, + { + "kind": "allocate_within_group_waterfall", + "source": "household.student_loan_balance", + "target": "person.student_loan_balance", + "tiers": [ + "student_loan_repayments > 0", + "student_loans > 0", + "highest_education == TERTIARY", + "current_education == TERTIARY", + "18 <= age <= 55", + "everyone" + ], + "keyed_by": "entity_id" + } + ], + "outputs": [ + "owned_land", + "property_wealth", + "corporate_wealth", + "gross_financial_wealth", + "net_financial_wealth", + "main_residence_value", + "other_residential_property_value", + "non_residential_property_value", + "savings", + "num_vehicles", + "cash_isa", + "stocks_and_shares_isa", + "student_loan_balance" + ], + "nonnegative_outputs": [ + "owned_land", + "property_wealth", + "corporate_wealth", + "gross_financial_wealth", + "main_residence_value", + "other_residential_property_value", + "non_residential_property_value", + "savings", + "num_vehicles", + "cash_isa", + "stocks_and_shares_isa", + "student_loan_balance" + ], + "notes": "Ports incumbent WAS round-8 wealth imputation with signed E5 differences: exact lower-case column matching replaces the fuzzy r/w fallback; cash ISA uses DVCISAVR8_aggr and stocks-and-shares ISA uses DVIISAVR8_aggr; corporate_wealth folds stocks-and-shares ISA after drawing corporate_wealth_excl_isa; recipient Northern Ireland regions are mapped to Wales for prediction only; student_loan_balance is allocated by household id rather than the incumbent positional off-by-one. Engine predictors materialize at their native entity and person/benunit values are summed to household, reproducing the incumbent map_to=household semantics; region is one-hot encoded jointly across donor and recipient (the incumbent's dummy encoding), with unmapped donor GOR codes becoming all-zero dummy rows. The WAS and FRS predictor definitions are not fully like-for-like and are ported as-is; raw WAS missing values are blanket-filled with zero; UKDS negative sentinel codes (-9/-8/-7/-6) are recoded to zero for the nonnegative-domain columns the licensed audit found carrying them (vcarnr8: 2 rows; HBedRmR8: 95.8 percent - the bedrooms question is effectively unasked in the WAS household file, predictor-quality revisit registered on microcosm#145) - a signed difference vs the incumbent, which trains on raw sentinels; DVPriRntR8's -9 is structural not-applicable so the is_renting mapping is unchanged; genuinely negative domains are never recoded." + }, + { + "stage": "regional_property_uprating", + "survey": "Public regional property reference", + "source": "MHCLG dwellings and ONS UK House Price Index December 2025 regional average prices.", + "grain": "household", + "base_candidate": { + "filename": "populace_uk_2023.h5", + "revision": "populace-uk-2023-dd68c73-4aa4b14-20260619T023711Z", + "sha256": "f17306ccb2aad7ff0130be3589b560afb2e2a12a943570911cd0c77f07934833", + "tier": "frs" + }, + "artifacts": [ + { + "role": "regional_reference", + "resource": "regional_land_values.json", + "kind": "public_aggregate_reference", + "format": "json" + } + ], + "operations": [ + { + "kind": "uprate_to_regional_reference", + "resource": "regional_land_values.json", + "factor": "avg_house_price / unweighted_mean(main_residence_value | region, >0)", + "owner_rows": "main_residence_value > 0", + "skip_empty_or_nonpositive_regions": true + } + ], + "outputs": [], + "rewrites": [ + "main_residence_value", + "property_wealth" + ], + "notes": "Deterministically rescales owner rows so regional unweighted owner means match the public house-price reference. Northern Ireland has no reference row and is never scaled; empty and nonpositive regions are skipped. The unweighted mean follows the incumbent behavior." + }, { "stage": "frs_hmrc_spine_leaves", "survey": "Family Resources Survey 2023-24", diff --git a/packages/microcosm-build/src/microcosm/build/uk/was_wealth_support_bounds.json b/packages/microcosm-build/src/microcosm/build/uk/was_wealth_support_bounds.json new file mode 100644 index 000000000..0d9bce9bd --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/uk/was_wealth_support_bounds.json @@ -0,0 +1,69 @@ +{ + "version": 1, + "country": "uk", + "policy": "Disclosure-safe outward-rounded WAS wealth support bounds for the UK support gate, generated from the donor household tab recorded under source.tab_sha256 by tools/build_uk_was_wealth_support_bounds.py. Values are donor-realized exact ranges rounded outward to one significant figure, so unit-record minima and maxima are never committed.", + "source": { + "ukds_study_number": 7215, + "doi": "10.5255/UKDA-SN-7215-20", + "artifact": "was_round_8_hhold_eul_may_2025_230525.tab", + "tab_sha256": "18b3eb980c02c99f3d8a3254af859bee31682b2bdc11703877677292b3ce9374", + "sdc_treatment": "Exact donor min/max values are rounded outward to one significant figure before commit, so no committed number is an exact unit-record value. Caveat considered and accepted: WAS applies top-coding and a round top-code can survive outward rounding unchanged; top-codes are published survey methodology rather than unit-record disclosures. Adjudicated acceptable under the UKDS EUL disclosure rules by juaristi22 on 2026-08-19 (microcosm#714)." + }, + "bounds": { + "owned_land": [ + 0.0, + 6000000 + ], + "property_wealth": [ + 0.0, + 20000000 + ], + "corporate_wealth": [ + 0.0, + 30000000 + ], + "gross_financial_wealth": [ + 0.0, + 30000000 + ], + "net_financial_wealth": [ + -400000, + 30000000 + ], + "main_residence_value": [ + 0.0, + 8000000 + ], + "other_residential_property_value": [ + 0.0, + 6000000 + ], + "non_residential_property_value": [ + 0.0, + 2000000 + ], + "savings": [ + 0.0, + 7000000 + ], + "num_vehicles": [ + 0.0, + 10 + ], + "cash_isa": [ + 0.0, + 900000 + ], + "stocks_and_shares_isa": [ + 0.0, + 3000000 + ], + "student_loan_balance": [ + 0.0, + 200000 + ] + }, + "chronicle": [ + "Support bounds generated from the tab pinned as source.tab_sha256 (18b3eb980c02\u2026) with outward SDC rounding." + ] +} diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/battery_bindings.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/battery_bindings.py index 4fba72b41..3833da3f9 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/battery_bindings.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/battery_bindings.py @@ -28,12 +28,16 @@ from __future__ import annotations +import json from collections.abc import Callable, Mapping from dataclasses import dataclass, replace from datetime import date, datetime +from importlib.resources import files from types import MappingProxyType from typing import Any +import numpy as np + from microcosm.build.gate_battery import ( DEFAULT_REGISTRY, EvidenceContext, @@ -43,6 +47,7 @@ GateResult, enum_domain_gate, nonnegative_columns_gate, + support_gate, weights_audit_gate, ) from microcosm.build.uk_runtime.frs_take_up import uk_take_up_signal_gate @@ -260,6 +265,34 @@ def _evaluate_brma_enum_domain( ) +def _evaluate_support( + context: EvidenceContext, parameters: Mapping[str, Any] +) -> GateResult: + resource_name = parameters.get("support_bounds_resource") + if resource_name != "was_wealth_support_bounds.json": + raise ValueError( + "uk_support must declare support_bounds_resource " + "'was_wealth_support_bounds.json'." + ) + resource = json.loads( + files("microcosm.build.uk").joinpath(str(resource_name)).read_text() + ) + raw_bounds = resource.get("bounds") + if not isinstance(raw_bounds, Mapping): + raise ValueError("WAS wealth support-bounds resource is missing bounds.") + donor_ranges = { + str(column): (float(bounds[0]), float(bounds[1])) + for column, bounds in raw_bounds.items() + } + values: dict[str, np.ndarray] = {} + for entity in context.frame.entities: + table = context.frame.table(entity) + for column in donor_ranges: + if column in table.columns: + values.setdefault(column, table[column].to_numpy()) + return support_gate(values, donor_ranges) + + def _stage_names_evidence( context: EvidenceContext, parameters: Mapping[str, Any] ) -> object: @@ -662,6 +695,11 @@ def _evaluate_tail_concentration( evaluator=_evaluate_brma_enum_domain, parameter_keys=frozenset({"columns"}), ), + "support": UKGateBinding( + name="support", + evaluator=_evaluate_support, + parameter_keys=frozenset({"support_bounds_resource"}), + ), "degenerate_release_surface": UKGateBinding( name="degenerate_release_surface", evaluator=_evaluate_degenerate_release_surface, diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/national_build.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/national_build.py index c4940826f..f02eea039 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/national_build.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/national_build.py @@ -765,30 +765,50 @@ def _validate_stages(stages: tuple[PlanStage, ...]) -> None: names.add(stage.name) +#: Stage names that perform production fits and therefore owe the terminal +#: weights audit their :class:`FitWeightRecord` evidence even when a swapped +#: or hollow transform stops exposing it. +_UK_FITTING_STAGE_NAMES = frozenset({"hmrc_spi_income", "was_wealth"}) + + def _stage_fit_weight_records( stages: tuple[PlanStage, ...], ) -> tuple[object, ...] | None: - """The weights-audit evidence artifact: present iff the HMRC stage is. - - ``None`` (no HMRC stage) leaves the artifact unsupplied, so the audit is - a named ``evidence_absent`` gap. A present stage always supplies the - artifact — records that are missing, unreadable, or empty coerce to - ``()``, which the UK audit binding fails: an absent audit is not a - passing audit. + """The weights-audit evidence artifact, aggregated across fitting stages. + + A stage counts as fitting when its name is a declared fitting stage + (HMRC SPI income, WAS wealth) or its transform exposes + ``fit_weight_records``; each contributes records in stage order. + ``None`` (no fitting stage scheduled) leaves the artifact unsupplied, + so the audit is a named ``evidence_absent`` gap. A present fitting + stage always supplies the artifact — records that are missing, + unreadable, or empty coerce to ``()``, which the UK audit binding + fails: an absent audit is not a passing audit. """ - hmrc_stage = next( - (stage for stage in stages if stage.name == "hmrc_spi_income"), - None, + fitting_stages = tuple( + stage + for stage in stages + if stage.name in _UK_FITTING_STAGE_NAMES + or hasattr(stage.transform, "fit_weight_records") ) - if hmrc_stage is None: + if not fitting_stages: return None - try: - records = getattr(hmrc_stage.transform, "fit_weight_records", None) - return () if records is None else tuple(records) - except Exception: # noqa: BLE001 - unreadable records coerce to () and - # fail the audit as missing evidence rather than crashing the batch. - return () + collected: list[object] = [] + for stage in fitting_stages: + try: + records = tuple(stage.transform.fit_weight_records or ()) + except Exception: # noqa: BLE001 - unreadable records coerce to () + # and fail the audit as missing evidence rather than crashing + # the batch. + return () + if not records: + # A scheduled fitting stage with no records is missing evidence; + # it must fail the audit, not be absorbed by another stage's + # records. + return () + collected.extend(records) + return tuple(collected) def _brma_enum_domain(engine: object) -> tuple[str, ...] | None: diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/regional_uprating.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/regional_uprating.py new file mode 100644 index 000000000..5a3803015 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/regional_uprating.py @@ -0,0 +1,106 @@ +"""UK regional property uprating stage.""" + +from __future__ import annotations + +import json +from collections.abc import Mapping +from dataclasses import dataclass +from importlib.resources import files +from typing import Any + +import numpy as np +import pandas as pd + +from microcosm.build.source_manifest import SourceStageSpec +from microcosm.build.uk_runtime.national_frame import ( + uk_household_weight_kind, + uk_national_frame, + uk_time_period, + validate_uk_national_frame, +) +from microcosm.frame import Frame + +UK_REGIONAL_PROPERTY_REWRITES = ("main_residence_value", "property_wealth") + + +@dataclass(frozen=True) +class UKRegionalPropertyUpratingStageTransform: + """Whole-stage callable for regional property-value uprating.""" + + stage: SourceStageSpec + resource: Mapping[str, Any] | None = None + + def __call__(self, frame: Frame) -> Frame: + resource = ( + self.resource + if self.resource is not None + else load_regional_land_values_resource() + ) + household = uprate_household_property_by_region( + frame.table("household"), + resource, + ) + result = uk_national_frame( + person=frame.table("person").copy(), + benunit=frame.table("benunit").copy(), + household=household, + time_period=uk_time_period(frame), + weight_kind=uk_household_weight_kind(frame), + household_weights=frame.weights_for("household").values, + mass_log=frame.mass_log, + ) + validate_uk_national_frame(result) + return result + + @staticmethod + def output_columns() -> tuple[str, ...]: + return () + + +def load_regional_land_values_resource() -> Mapping[str, Any]: + return json.loads( + files("microcosm.build.uk").joinpath("regional_land_values.json").read_text() + ) + + +def uprate_household_property_by_region( + household: pd.DataFrame, + resource: Mapping[str, Any], +) -> pd.DataFrame: + """Scale owner property values to region-level public house-price means.""" + + required = {"region", "main_residence_value", "property_wealth"} + missing = sorted(required - set(household.columns)) + if missing: + raise KeyError( + f"household table is missing property-uprating columns: {missing}" + ) + values = resource.get("values") + if not isinstance(values, list): + raise ValueError("regional land values resource must contain a values list.") + hpi_prices = { + str(entry["region"]): float(entry["avg_house_price"]) + for entry in values + if isinstance(entry, Mapping) + } + result = household.copy() + for region, hpi_price in hpi_prices.items(): + region_mask = result["region"].astype(str) == region + owners_mask = region_mask & ( + pd.to_numeric(result["main_residence_value"], errors="coerce") > 0 + ) + if not owners_mask.any(): + continue + # Sort before summing so the unweighted mean is independent of row + # order: the stage is receipted bitwise under row permutation, and + # float addition is not associative. + owner_values = np.sort( + result.loc[owners_mask, "main_residence_value"].to_numpy(dtype=float) + ) + imputed_mean = float(owner_values.sum() / len(owner_values)) + if imputed_mean <= 0: + continue + factor = hpi_price / imputed_mean + result.loc[owners_mask, "main_residence_value"] *= factor + result.loc[owners_mask, "property_wealth"] *= factor + return result diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/source_runtime.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/source_runtime.py index 547c18d1e..c1b4548b8 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/source_runtime.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/source_runtime.py @@ -32,8 +32,7 @@ def _uk_nonnegative_outputs_by_stage() -> dict[str, tuple[str, ...]]: if spec.sources is None: return {} return { - stage.stage: tuple(stage.nonnegative_outputs) - for stage in spec.sources.stages + stage.stage: tuple(stage.nonnegative_outputs) for stage in spec.sources.stages } @@ -62,6 +61,8 @@ def uk_stage_implementations( frs_person_draws_transform: Callable[[Frame], Frame] | None = None, frs_household_draws_transform: Callable[[Frame], Frame] | None = None, frs_brma_transform: Callable[[Frame], Frame] | None = None, + was_wealth_transform: Callable[[Frame], Frame] | None = None, + regional_property_uprating_transform: Callable[[Frame], Frame] | None = None, frs_hmrc_spine_leaves_transform: Callable[[Frame], Frame] | None = None, spi_support_channel_transform: Callable[[Frame], Frame] | None = None, hmrc_spi_income_spine_transform: Callable[[Frame], Frame] | None = None, @@ -84,6 +85,8 @@ def uk_stage_implementations( "frs_person_draws": frs_person_draws_transform, "frs_household_draws": frs_household_draws_transform, "frs_brma": frs_brma_transform, + "was_wealth": was_wealth_transform, + "regional_property_uprating": regional_property_uprating_transform, "frs_hmrc_spine_leaves": frs_hmrc_spine_leaves_transform, "spi_support_channel": spi_support_channel_transform, "hmrc_spi_income_spine": hmrc_spi_income_spine_transform, diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/terminal_gates.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/terminal_gates.py index 55dce6ba3..85d5cddb5 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/terminal_gates.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/terminal_gates.py @@ -157,6 +157,7 @@ def __post_init__(self) -> None: "benunit.child_benefit_opts_out", "household.bus_fare_spending", "household.bus_subsidy_spending", + "household.cash_isa", "household.clone_index", "household.constituency_code_oa", "household.consumer_debt", @@ -174,6 +175,7 @@ def __post_init__(self) -> None: "household.property_purchased", "household.rail_usage", "household.region_code_oa", + "household.stocks_and_shares_isa", "person.aa_category", "person.age_started_or_accepted_current_education_or_training", "person.attends_private_school_random_draw", diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/was_wealth.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/was_wealth.py new file mode 100644 index 000000000..5a5b9c44d --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/was_wealth.py @@ -0,0 +1,591 @@ +"""UK WAS wealth imputation stage.""" + +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +import numpy as np +import pandas as pd + +from microcosm.build.gates import FitWeightRecord +from microcosm.build.source_manifest import SourceStageSpec +from microcosm.build.uk_runtime.frs_brma import _benunit_household_map +from microcosm.build.uk_runtime.frs_spine import read_pinned_tab +from microcosm.build.uk_runtime.national_frame import ( + uk_household_weight_kind, + uk_national_frame, + uk_time_period, + validate_uk_national_frame, +) +from microcosm.frame import Frame +from microcosm.frame.rules import assert_rules_engine_country + +WAS_DONOR_FILENAME = "was_round_8_hhold_eul_may_2025_230525.tab" +WAS_DONOR_SHA256 = "18b3eb980c02c99f3d8a3254af859bee31682b2bdc11703877677292b3ce9374" +WAS_DONOR_SIZE_BYTES = 39_073_613 + +UK_WAS_WEALTH_PREDICTORS = ( + "household_net_income", + "num_adults", + "num_children", + "private_pension_income", + "employment_income", + "self_employment_income", + "capital_income", + "num_bedrooms", + "council_tax", + "is_renting", + "region", +) +UK_WAS_ENGINE_PREDICTORS = ( + "household_net_income", + "num_adults", + "num_children", + "private_pension_income", + "employment_income", + "self_employment_income", + "capital_income", + "is_renting", +) +UK_WAS_WEALTH_OUTPUT_COLUMNS = ( + "owned_land", + "property_wealth", + "corporate_wealth", + "gross_financial_wealth", + "net_financial_wealth", + "main_residence_value", + "other_residential_property_value", + "non_residential_property_value", + "savings", + "num_vehicles", + "cash_isa", + "stocks_and_shares_isa", + "student_loan_balance", +) +UK_WAS_WEALTH_HOUSEHOLD_OUTPUT_COLUMNS = tuple( + column + for column in UK_WAS_WEALTH_OUTPUT_COLUMNS + if column != "student_loan_balance" +) +UK_WAS_WEALTH_NONNEGATIVE_OUTPUT_COLUMNS = tuple( + column + for column in UK_WAS_WEALTH_OUTPUT_COLUMNS + if column != "net_financial_wealth" +) +UK_WAS_WEALTH_DECLARED_SEEDS = {"was_wealth": 0} +UK_WAS_WEALTH_FIT_NAME = "uk_was_2018_20_wealth" + +REGIONS: Mapping[int, str] = { + 1: "NORTH_EAST", + 2: "NORTH_WEST", + 4: "YORKSHIRE", + 5: "EAST_MIDLANDS", + 6: "WEST_MIDLANDS", + 7: "EAST_OF_ENGLAND", + 8: "LONDON", + 9: "SOUTH_EAST", + 10: "SOUTH_WEST", + 11: "WALES", + 12: "SCOTLAND", +} +REGION_REMAP = {"NORTHERN_IRELAND": "WALES"} +UK_WAS_ENGINE_PREDICTOR_ENTITIES: Mapping[str, str] = { + "household_net_income": "household", + "num_adults": "benunit", + "num_children": "benunit", + "private_pension_income": "person", + "employment_income": "person", + "self_employment_income": "person", + "capital_income": "person", + "is_renting": "household", +} + +#: UKDS negative sentinel codes observed in the round-8 household tab. +#: Recoded to zero ONLY for columns whose domain cannot be negative and +#: which the 2026-08-19 licensed audit found carrying them: vcarnr8 (2 rows +#: of -8) and HBedRmR8 (95.8% -8 - the bedrooms question is effectively +#: unasked in the WAS household file; the predictor-quality question is +#: registered for the end-of-workstream revisit on microcosm#145). +#: DVPriRntR8's -9 is structural not-applicable (not a private renter), so +#: the is_renting == 1 mapping is already correct; genuinely negative +#: domains (net financial wealth, self-employment losses, BHC income) are +#: never recoded. Signed difference vs the incumbent, which trains on the +#: raw sentinel values. +_SENTINEL_CODES = (-9.0, -8.0, -7.0, -6.0) +_SENTINEL_RECODE_COLUMNS = ("num_vehicles", "num_bedrooms") + +_RAW_TO_CLEAN = { + "R8xshhwgt": "weight", + "DVLUKValR8_sum": "owned_land", + "DVPropertyR8": "property_wealth", + "DVFESHARESR8_aggr": "emp_shares_options", + "DVFShUKVR8_aggr": "uk_shares", + "DVIISAVR8_aggr": "stocks_and_shares_isa", + "DVCISAVR8_aggr": "cash_isa", + "DVFCollVR8_aggr": "unit_investment_trusts", + "totalpenr8_aggr": "pensions", + "dvvaldbt_scaper8_aggr": "db_pensions", + "NumAdultR8": "num_adults", + "NumCh18R8": "num_children", + "DVGIPPENR8_AGGR": "private_pension_income", + "DVGISER8_AGGR": "self_employment_income", + "DVGIINVR8_aggr": "capital_income", + "DVGIEMPR8_AGGR": "employment_income", + "HBedRmR8": "num_bedrooms", + "GORR8": "region_code", + "DVPriRntR8": "private_rent_code", + "CTAmtR8": "council_tax", + "HFINWNTR8_Sum": "net_financial_wealth", + "HFINWR8_SUM": "gross_financial_wealth", + "DVhvalueR8": "main_residence_value", + "DVHseValR8_sum": "other_residential_property_value", + "DVBlDValR8_sum": "non_residential_property_value", + "DVTotinc_bhcR8": "household_net_income", + "DVSaValR8_aggr": "savings", + "vcarnr8": "num_vehicles", + "Tot_LosR8_aggr": "total_loans", + "Tot_los_exc_SLCR8_aggr": "total_loans_exc_slc", +} + + +@dataclass +class UKWASWealthStageTransform: + """Whole-stage callable for WAS-trained UK wealth imputation. + + Not frozen: like the HMRC restoration stage, the transform carries + mutable post-run fit-weight evidence for the terminal weights audit. + """ + + stage: SourceStageSpec + engine: object + was_tab_path: str | Path | None = None + donor: pd.DataFrame | None = None + #: Fit-weight evidence from the most recent run, read by the national + #: build's weights-audit collector (the HMRC-stage precedent). + last_fit_weight_records: tuple[FitWeightRecord, ...] | None = field( + default=None, + init=False, + repr=False, + ) + + @property + def fit_weight_records(self) -> tuple[FitWeightRecord, ...]: + """Return immutable fit-weight evidence from the most recent run.""" + + if self.last_fit_weight_records is None: + return () + return tuple(self.last_fit_weight_records) + + def __call__(self, frame: Frame) -> Frame: + assert_rules_engine_country(self.engine, "uk") + donor = ( + clean_was_household_table(self.donor) + if self.donor is not None + else clean_was_household_table( + read_pinned_tab( + _require_path(self.was_tab_path), _donor_artifact(self.stage) + ) + ) + ) + household_predictors = recipient_predictors(frame, self.engine) + imputation = impute_was_wealth( + donor, + household_predictors, + seed=UK_WAS_WEALTH_DECLARED_SEEDS["was_wealth"], + n_estimators=_qrf_n_estimators(self.stage), + ) + self.last_fit_weight_records = imputation.fit_weight_records + household_draws = support_clip_to_donor(imputation.draws, donor) + household_draws["num_vehicles"] = ( + np.rint(household_draws["num_vehicles"]).clip(lower=0).astype("int64") + ) + person = frame.table("person").copy() + household = frame.table("household").copy() + for column in UK_WAS_WEALTH_HOUSEHOLD_OUTPUT_COLUMNS: + household[column] = household_draws[column].to_numpy() + person["student_loan_balance"] = allocate_student_loan_balance_to_people( + household_balances=household_draws["student_loan_balance"].clip(lower=0), + household_ids=household["household_id"], + person=person, + ) + result = uk_national_frame( + person=person, + benunit=frame.table("benunit").copy(), + household=household, + time_period=uk_time_period(frame), + weight_kind=uk_household_weight_kind(frame), + household_weights=frame.weights_for("household").values, + mass_log=frame.mass_log, + ) + validate_uk_national_frame(result) + return result + + @staticmethod + def output_columns() -> tuple[str, ...]: + return UK_WAS_WEALTH_OUTPUT_COLUMNS + + +def clean_was_household_table(raw: pd.DataFrame) -> pd.DataFrame: + """Return the WAS donor table with exact lower-case column matching.""" + + lowered = {str(column).lower(): column for column in raw.columns} + if len(lowered) != len(raw.columns): + raise ValueError("WAS donor has duplicate columns after lower-case matching.") + renamed: dict[str, str] = {} + missing: list[str] = [] + for source, target in _RAW_TO_CLEAN.items(): + actual = lowered.get(source.lower()) + if actual is None: + missing.append(source) + else: + renamed[actual] = target + if missing: + raise ValueError(f"WAS donor is missing required column(s): {missing}.") + cleaned = raw.rename(columns=renamed)[list(renamed.values())].copy() + for column in cleaned.columns: + cleaned[column] = pd.to_numeric(cleaned[column], errors="coerce") + cleaned = cleaned.fillna(0) + for column in _SENTINEL_RECODE_COLUMNS: + values = cleaned[column] + cleaned[column] = values.where(~values.isin(_SENTINEL_CODES), 0) + cleaned["is_renting"] = cleaned["private_rent_code"] == 1 + cleaned["corporate_wealth_excl_isa"] = ( + cleaned["pensions"] + - cleaned["db_pensions"] + + cleaned["emp_shares_options"] + + cleaned["uk_shares"] + + cleaned["unit_investment_trusts"] + ) + cleaned["corporate_wealth"] = ( + cleaned["corporate_wealth_excl_isa"] + cleaned["stocks_and_shares_isa"] + ) + cleaned["student_loan_balance"] = ( + cleaned["total_loans"] - cleaned["total_loans_exc_slc"] + ) + cleaned["region"] = cleaned["region_code"].map(REGIONS) + return cleaned[ + [ + *UK_WAS_WEALTH_PREDICTORS, + "weight", + "corporate_wealth_excl_isa", + *UK_WAS_WEALTH_HOUSEHOLD_OUTPUT_COLUMNS, + "student_loan_balance", + ] + ] + + +def recipient_predictors(frame: Frame, engine: object) -> pd.DataFrame: + """Materialize the WAS predictor surface on recipient households. + + Engine variables live at their native entity; person- and benunit-level + predictors are summed to household grain, reproducing the incumbent's + ``map_to="household"`` semantics. + """ + + materialized = engine.materialize( + frame, UK_WAS_ENGINE_PREDICTORS, uk_time_period(frame) + ) + household = frame.table("household") + person = frame.table("person") + benunit = frame.table("benunit") + household_ids = pd.Index(household["household_id"]) + result = pd.DataFrame(index=household.index) + for predictor in UK_WAS_ENGINE_PREDICTORS: + entity = UK_WAS_ENGINE_PREDICTOR_ENTITIES[predictor] + declared = str(engine.variable_metadata(predictor).entity) + if declared != entity: + raise ValueError( + f"engine declares {predictor!r} at entity {declared!r}; " + f"the WAS wealth stage expects {entity!r}." + ) + values = np.asarray(materialized[predictor]) + if entity == "household": + if values.shape != (len(household),): + raise ValueError( + f"materialized {predictor!r} has shape {values.shape}; " + f"expected ({len(household)},)." + ) + result[predictor] = values + elif entity == "benunit": + if values.shape != (len(benunit),): + raise ValueError( + f"materialized {predictor!r} has shape {values.shape}; " + f"expected ({len(benunit)},)." + ) + groups = benunit["benunit_id"].map(_benunit_household_map(person)) + summed = pd.Series(values.astype(float)).groupby(groups.to_numpy()).sum() + result[predictor] = summed.reindex(household_ids).fillna(0.0).to_numpy() + else: + if values.shape != (len(person),): + raise ValueError( + f"materialized {predictor!r} has shape {values.shape}; " + f"expected ({len(person)},)." + ) + summed = ( + pd.Series(values.astype(float)) + .groupby(person["person_household_id"].to_numpy()) + .sum() + ) + result[predictor] = summed.reindex(household_ids).fillna(0.0).to_numpy() + for predictor in ("num_bedrooms", "council_tax", "region"): + if predictor not in household.columns: + raise KeyError(f"recipient household table is missing {predictor!r}.") + result[predictor] = household[predictor].to_numpy() + result["region"] = result["region"].map(_enum_name).replace(REGION_REMAP) + result["is_renting"] = result["is_renting"].astype(bool) + return result.loc[:, UK_WAS_WEALTH_PREDICTORS] + + +@dataclass(frozen=True) +class UKWASWealthImputationResult: + """WAS wealth draws plus the auditable fit-weight evidence.""" + + draws: pd.DataFrame + fit_weight_records: tuple[FitWeightRecord, ...] + + +def impute_was_wealth( + donor: pd.DataFrame, + recipient_predictor_frame: pd.DataFrame, + *, + seed: int, + n_estimators: int, +) -> UKWASWealthImputationResult: + """Fit segmented checkpointed QRF chains and draw WAS wealth outputs.""" + + from microcosm.fit import RegimeGatedQRF + + donor_encoded, recipient_encoded, encoded_predictors = encode_qrf_predictor_pair( + donor, recipient_predictor_frame + ) + model = RegimeGatedQRF(n_estimators=n_estimators, seed=seed) + raw = pd.DataFrame(index=recipient_encoded.index) + fit_records: list[FitWeightRecord] = [] + + def run_segment(base_predictors: Sequence[str], targets: Sequence[str]) -> None: + state = model.start_chain( + donor_encoded, + list(base_predictors), + list(targets), + weights="weight", + ) + segment_raw = pd.DataFrame(index=recipient_encoded.index) + recipient_base = pd.concat( + [recipient_encoded.loc[:, list(base_predictors)]], + axis=1, + ) + for target in targets: + result = model.fit_draw_next( + donor_encoded, + recipient_base, + segment_raw, + state=state, + weights="weight", + ) + fit_records.append( + FitWeightRecord( + f"{UK_WAS_WEALTH_FIT_NAME}:{target}", result.weight_kind + ) + ) + segment_raw[target] = result.raw_draw + raw[target] = result.raw_draw + state = result.state + + base = encoded_predictors + run_segment(base, ("owned_land", "property_wealth")) + donor_encoded["corporate_wealth"] = donor_encoded["corporate_wealth"].astype(float) + recipient_encoded["owned_land"] = raw["owned_land"] + recipient_encoded["property_wealth"] = raw["property_wealth"] + run_segment( + (*base, "owned_land", "property_wealth"), + ("corporate_wealth_excl_isa", "stocks_and_shares_isa"), + ) + raw["corporate_wealth"] = ( + raw["corporate_wealth_excl_isa"] + raw["stocks_and_shares_isa"] + ) + recipient_encoded["corporate_wealth"] = raw["corporate_wealth"] + run_segment( + (*base, "owned_land", "property_wealth", "corporate_wealth"), + ( + "gross_financial_wealth", + "net_financial_wealth", + "main_residence_value", + "other_residential_property_value", + "non_residential_property_value", + "savings", + "num_vehicles", + "student_loan_balance", + "cash_isa", + ), + ) + return UKWASWealthImputationResult( + draws=raw.loc[:, UK_WAS_WEALTH_OUTPUT_COLUMNS], + fit_weight_records=tuple(fit_records), + ) + + +def encode_qrf_predictor_pair( + donor: pd.DataFrame, recipient: pd.DataFrame +) -> tuple[pd.DataFrame, pd.DataFrame, tuple[str, ...]]: + """One-hot the region predictor jointly across donor and recipient. + + Mirrors the SPI stage's paired dummy encoding and the incumbent's + dummy-encoded region. Donor rows with an unmapped region code (the + incumbent's absent GOR code 3) become all-zero dummy rows. + """ + + numeric_predictors = tuple( + predictor for predictor in UK_WAS_WEALTH_PREDICTORS if predictor != "region" + ) + combined_region = pd.concat( + [ + donor["region"].reset_index(drop=True), + recipient["region"].reset_index(drop=True), + ], + ignore_index=True, + ) + dummies = pd.get_dummies(combined_region, prefix="region", dtype=float) + dummies = dummies.reindex(sorted(dummies.columns), axis=1) + + def _encode(table: pd.DataFrame, block: pd.DataFrame) -> pd.DataFrame: + encoded = table.drop(columns=["region"]).copy() + if "is_renting" in encoded.columns: + encoded["is_renting"] = encoded["is_renting"].astype(bool).astype(float) + for column in encoded.columns: + if column == "weight": + continue + encoded[column] = pd.to_numeric(encoded[column], errors="coerce").fillna( + 0.0 + ) + block = block.copy() + block.index = encoded.index + return pd.concat([encoded, block], axis=1) + + donor_encoded = _encode(donor, dummies.iloc[: len(donor)]) + recipient_encoded = _encode(recipient, dummies.iloc[len(donor) :]) + return ( + donor_encoded, + recipient_encoded, + (*numeric_predictors, *tuple(dummies.columns)), + ) + + +def support_clip_to_donor(draws: pd.DataFrame, donor: pd.DataFrame) -> pd.DataFrame: + """Clip output draws to donor-realized support.""" + + clipped = draws.copy() + for column in UK_WAS_WEALTH_OUTPUT_COLUMNS: + if column not in clipped or column not in donor: + continue + values = pd.to_numeric(donor[column], errors="coerce") + finite = values[np.isfinite(values)] + if finite.empty: + continue + clipped[column] = clipped[column].clip( + lower=float(finite.min()), + upper=float(finite.max()), + ) + return clipped + + +def allocate_student_loan_balance_to_people( + *, + household_balances: pd.Series, + household_ids: Sequence[object], + person: pd.DataFrame, +) -> np.ndarray: + """Allocate household student-loan balances to plausible holders by id.""" + + balances_by_household = pd.Series( + np.asarray(household_balances, dtype=float), + index=pd.Index(household_ids), + ) + allocated = np.zeros(len(person), dtype=float) + if len(person) == 0: + return allocated + group_indices = person.groupby("person_household_id", sort=False).indices + age = _numeric_person(person, "age", 0.0) + repayments = _numeric_person(person, "student_loan_repayments", 0.0) + student_loans = _numeric_person(person, "student_loans", 0.0) + highest_education = _string_person(person, "highest_education", "UPPER_SECONDARY") + current_education = _string_person(person, "current_education", "NOT_IN_EDUCATION") + for household_id, household_balance in balances_by_household.items(): + if household_balance <= 0 or household_id not in group_indices: + continue + idx = np.asarray(group_indices[household_id], dtype=int) + tier_masks = ( + repayments[idx] > 0, + student_loans[idx] > 0, + highest_education[idx] == "TERTIARY", + current_education[idx] == "TERTIARY", + (age[idx] >= 18) & (age[idx] <= 55), + np.ones(len(idx), dtype=bool), + ) + selected_mask = next(mask for mask in tier_masks if mask.any()) + selected = idx[selected_mask] + if tier_masks[0].any() and repayments[idx][tier_masks[0]].sum() > 0: + repayers = idx[tier_masks[0]] + weights = repayments[repayers] + allocated[repayers] += household_balance * weights / weights.sum() + else: + allocated[selected] += household_balance / len(selected) + return allocated + + +def donor_realized_ranges(donor: pd.DataFrame) -> dict[str, tuple[float, float]]: + """Return exact donor min/max ranges for synthetic receipts and tests.""" + + ranges: dict[str, tuple[float, float]] = {} + for column in UK_WAS_WEALTH_OUTPUT_COLUMNS: + values = pd.to_numeric(donor[column], errors="coerce") + finite = values[np.isfinite(values)] + if not finite.empty: + ranges[column] = (float(finite.min()), float(finite.max())) + return ranges + + +def _numeric_person(person: pd.DataFrame, column: str, default: float) -> np.ndarray: + values = ( + person[column] if column in person else pd.Series(default, index=person.index) + ) + return pd.to_numeric(values, errors="coerce").fillna(default).to_numpy(dtype=float) + + +def _string_person(person: pd.DataFrame, column: str, default: str) -> np.ndarray: + values = ( + person[column] if column in person else pd.Series(default, index=person.index) + ) + return values.fillna(default).map(_enum_name).astype(str).to_numpy() + + +def _donor_artifact(stage: SourceStageSpec) -> Mapping[str, Any]: + for artifact in stage.artifacts: + if artifact.get("role") == "was_qrf_donor": + return artifact + raise ValueError( + "was_wealth stage declares no was_qrf_donor artifact; refusing to read " + "an unpinned WAS tab." + ) + + +def _qrf_n_estimators(stage: SourceStageSpec) -> int: + for operation in stage.operations: + if operation.kind == "fit_weighted_qrf_chain": + value = operation.parameters.get("n_estimators", 100) + if isinstance(value, int) and value > 0: + return value + return 100 + + +def _require_path(path: str | Path | None) -> Path: + if path is None: + raise ValueError("WAS wealth stage requires a caller-supplied WAS tab path.") + return Path(path).expanduser().resolve() + + +def _enum_name(value: object) -> str: + name = getattr(value, "name", None) + return str(name if name is not None else value) diff --git a/packages/microcosm-build/tests/fixtures/uk/regional_land_values.csv b/packages/microcosm-build/tests/fixtures/uk/regional_land_values.csv new file mode 100644 index 000000000..1d446a8db --- /dev/null +++ b/packages/microcosm-build/tests/fixtures/uk/regional_land_values.csv @@ -0,0 +1,12 @@ +region,avg_house_price,dwellings +NORTH_EAST,165257,1280700 +NORTH_WEST,217428,3441730 +YORKSHIRE,208447,2530680 +EAST_MIDLANDS,243632,2213820 +WEST_MIDLANDS,246141,2619300 +EAST_OF_ENGLAND,338002,2841060 +LONDON,551294,3813790 +SOUTH_EAST,378800,4138200 +SOUTH_WEST,301226,2683030 +WALES,214883,1476690 +SCOTLAND,190649,2602545 diff --git a/packages/microcosm-build/tests/test_country_spec.py b/packages/microcosm-build/tests/test_country_spec.py index c39e58e2a..b533aeeca 100644 --- a/packages/microcosm-build/tests/test_country_spec.py +++ b/packages/microcosm-build/tests/test_country_spec.py @@ -273,20 +273,22 @@ def test_spi_spine_adds_no_country_package_resources(self) -> None: "hmrc_income_release_gate_report.json", "hmrc_income_replay_report.json", "hmrc_income_source_stages.json", + "regional_land_values.json", "source_stages.json", "take_up_contract.json", "input_mass_reviewed_exclusions.json", "national_staging_build_record.json", "qrf_tail_reviewed_exclusions.json", "release_input_coverage_manifest.json", + "was_wealth_support_bounds.json", "uk_local_target_census.json", ) - def test_uk_source_manifest_loads_sixteen_stages(self) -> None: + def test_uk_source_manifest_loads_eighteen_stages(self) -> None: spec = load_country_spec("uk") assert spec.sources is not None - assert len(spec.sources.stages) == 16 + assert len(spec.sources.stages) == 18 class TestExistingPackagesGeneralize: @@ -315,12 +317,14 @@ def test_uk_package_loads(self) -> None: "hmrc_income_release_gate_report.json", "hmrc_income_replay_report.json", "hmrc_income_source_stages.json", + "regional_land_values.json", "source_stages.json", "take_up_contract.json", "input_mass_reviewed_exclusions.json", "national_staging_build_record.json", "qrf_tail_reviewed_exclusions.json", "release_input_coverage_manifest.json", + "was_wealth_support_bounds.json", "uk_local_target_census.json", ) @@ -353,6 +357,7 @@ def test_declares_the_full_june_battery(self, manifest) -> None: "uk_weight_ratio", "uk_weights_audit", "uk_nonnegative_columns", + "uk_support", "uk_export_surface", "uk_take_up_signal", "uk_brma_enum_domain", diff --git a/packages/microcosm-build/tests/test_uk_battery_bindings.py b/packages/microcosm-build/tests/test_uk_battery_bindings.py index 54522313a..042cbfbf5 100644 --- a/packages/microcosm-build/tests/test_uk_battery_bindings.py +++ b/packages/microcosm-build/tests/test_uk_battery_bindings.py @@ -282,6 +282,46 @@ def test_nonnegative_binding_does_not_demand_unscheduled_stages(self) -> None: assert result.passed is True + def test_support_binding_passes_in_range_was_outputs(self) -> None: + person, benunit, household = _tables(n=2) + household["owned_land"] = [0.0, 100.0] + household["cash_isa"] = [0.0, 1000.0] + person["student_loan_balance"] = [0.0, 100.0] + frame = uk_national_frame( + person=person, + benunit=benunit, + household=household, + time_period="2023", + ) + binding = UK_GATE_REGISTRY["support"] + + result = binding.evaluate( + EvidenceContext(frame=frame, artifacts={}), + {"support_bounds_resource": "was_wealth_support_bounds.json"}, + ) + + assert result.passed is True + assert result.details["columns_checked"] == 3 + + def test_support_binding_fails_out_of_range_was_outputs(self) -> None: + person, benunit, household = _tables(n=1) + household["cash_isa"] = [99_999_999.0] + frame = uk_national_frame( + person=person, + benunit=benunit, + household=household, + time_period="2023", + ) + binding = UK_GATE_REGISTRY["support"] + + result = binding.evaluate( + EvidenceContext(frame=frame, artifacts={}), + {"support_bounds_resource": "was_wealth_support_bounds.json"}, + ) + + assert result.passed is False + assert "cash_isa" in result.failures[0] + class TestUKCompatibility: """The BE plumbing test, run over the UK spec: an empty evidence context @@ -336,9 +376,9 @@ def test_fully_armed_battery_evaluates_gate_for_gate(self) -> None: entry_id for entry_id, o in by_id.items() if o.status is GateStatus.PASSED ] # 11 as on main (uk_nonnegative_columns passes with zero required - # columns — the scheduled stages declare none) plus the two E4 - # stochastic gates; their evaluators have direct tests of their own. - assert len(passed) == 13 + # columns — the scheduled stages declare none), the two E4 stochastic + # gates, and the E5 support gate; their evaluators have direct tests. + assert len(passed) == 14 qrf = by_id["uk_qrf_tail_concentration"] assert qrf.status is GateStatus.FAILED assert "declared QRF output is absent" in qrf.result.failures[0] diff --git a/packages/microcosm-build/tests/test_uk_cgt_source_manifest.py b/packages/microcosm-build/tests/test_uk_cgt_source_manifest.py index a886f54e6..1b9a86a15 100644 --- a/packages/microcosm-build/tests/test_uk_cgt_source_manifest.py +++ b/packages/microcosm-build/tests/test_uk_cgt_source_manifest.py @@ -116,6 +116,9 @@ def test_the_shipped_family_contracts_pass_the_terminal_gate_shape() -> None: spi_reason = str( manifest.family_coverage["hmrc_spi_income"]["required_mass_change_reason"] ) + e5_reason = str( + manifest.family_coverage["was_wealth"]["required_mass_change_reason"] + ) compliant = SimpleNamespace( household_weight_kind=WeightKind.IMPORTANCE, time_period="2023", @@ -134,6 +137,13 @@ def test_the_shipped_family_contracts_pass_the_terminal_gate_shape() -> None: declared_factor=1.0, reason=UK_CGT_MASS_CONSERVATION_REASON, ), + MassChangeRecord( + entity="household", + old_total=100.0, + new_total=100.0, + declared_factor=1.0, + reason=e5_reason, + ), ), ) diff --git a/packages/microcosm-build/tests/test_uk_e5_identity.py b/packages/microcosm-build/tests/test_uk_e5_identity.py new file mode 100644 index 000000000..d1269b883 --- /dev/null +++ b/packages/microcosm-build/tests/test_uk_e5_identity.py @@ -0,0 +1,72 @@ +from __future__ import annotations + +import importlib.util +from pathlib import Path + +import pandas as pd + +from microcosm.build.uk_runtime.national_frame import uk_national_frame + +ROOT = Path(__file__).resolve().parents[3] + + +def _identity_module(): + path = ROOT / "tools/verify_uk_identity_stability.py" + spec = importlib.util.spec_from_file_location("verify_uk_identity_stability", path) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +def test_e5_identity_receipt_is_stable_under_row_permutation() -> None: + frame = uk_national_frame( + person=pd.DataFrame( + { + "person_id": [101, 102, 201], + "person_benunit_id": [10, 10, 20], + "person_household_id": [1, 1, 2], + "age": [25, 40, 20], + "student_loan_repayments": [10.0, 30.0, 0.0], + "student_loans": [0.0, 0.0, 1.0], + } + ), + benunit=pd.DataFrame({"benunit_id": [10, 20]}), + household=pd.DataFrame( + { + "household_id": [1, 2], + "household_weight": [1.0, 1.0], + "region": ["LONDON", "SCOTLAND"], + "main_residence_value": [100.0, 200.0], + "property_wealth": [150.0, 300.0], + "corporate_wealth_excl_isa": [10.0, 20.0], + "stocks_and_shares_isa": [1.0, 2.0], + "student_loan_balance": [400.0, 100.0], + } + ), + time_period="2023", + ) + resource = { + "values": [ + {"region": "LONDON", "avg_house_price": 200.0}, + {"region": "SCOTLAND", "avg_house_price": 400.0}, + ] + } + + receipt = _identity_module().e5_identity_receipt( + frame, + regional_resource=resource, + permutation_seed=42, + ) + + assert receipt["check"] == "uk_e5_identity_stability" + assert receipt["identical_under_permutation"] is True + assert receipt["permutation_mismatches"] == {} + assert receipt["columns_by_entity"] == { + "household": [ + "corporate_wealth", + "main_residence_value", + "property_wealth", + ], + "person": ["student_loan_balance"], + } diff --git a/packages/microcosm-build/tests/test_uk_frs_spine.py b/packages/microcosm-build/tests/test_uk_frs_spine.py index 98a36e274..b4a6e0b38 100644 --- a/packages/microcosm-build/tests/test_uk_frs_spine.py +++ b/packages/microcosm-build/tests/test_uk_frs_spine.py @@ -1143,7 +1143,11 @@ def test_driver_writes_spine_h5_sidecars_and_logbook( sidecar = json.loads(output.with_suffix(".build.json").read_text()) assert sidecar["pipeline"] == "uk-frs-spine" assert sidecar["schema_version"] == 2 - assert sidecar["stages"] == list(tool._STAGE_NAMES) + assert sidecar["stages"] == [ + name + for name in tool._STAGE_NAMES + if name in _synthetic_spec(stage).sources.stage_map() + ] assert sidecar["entity_row_counts"] == { "person": 3, "benunit": 2, @@ -1421,15 +1425,15 @@ def test_input_artifact_pins_bind_spi_donor_and_ods() -> None: pins = tool._input_artifact_pins(stages) - assert set(pins) == {"qrf_donor", "published_fact_surface"} + assert set(pins) == {"qrf_donor", "was_qrf_donor", "published_fact_surface"} for pin in pins.values(): assert len(str(pin["sha256"])) == 64 assert int(pin["size_bytes"]) > 0 assert str(pin["filename"]) - income_stage = stage_map["hmrc_spi_income_spine"] declared = { str(artifact["role"]): str(artifact["sha256"]) - for artifact in income_stage.artifacts + for stage_name in ("hmrc_spi_income_spine", "was_wealth") + for artifact in stage_map[stage_name].artifacts if "table" not in artifact and "resource" not in artifact } assert {role: pin["sha256"] for role, pin in pins.items()} == declared diff --git a/packages/microcosm-build/tests/test_uk_national_build.py b/packages/microcosm-build/tests/test_uk_national_build.py index 00057249e..9a932bc6c 100644 --- a/packages/microcosm-build/tests/test_uk_national_build.py +++ b/packages/microcosm-build/tests/test_uk_national_build.py @@ -381,6 +381,78 @@ def __call__(self, frame: Frame) -> Frame: return frame +class _WASRecordedFitStage: + fit_weight_records = ( + FitWeightRecord("uk_was_2018_20_wealth:owned_land", "explicit"), + FitWeightRecord("uk_was_2018_20_wealth:cash_isa", "explicit"), + ) + + def __call__(self, frame: Frame) -> Frame: + return frame + + +def test_stage_fit_weight_records_aggregates_every_fitting_stage() -> None: + from types import SimpleNamespace + + from microcosm.build.uk_runtime.national_build import _stage_fit_weight_records + + plain = SimpleNamespace(name="frs_take_up", transform=lambda frame: frame) + hmrc = SimpleNamespace(name="hmrc_spi_income", transform=_RecordedFitStage()) + was = SimpleNamespace(name="was_wealth", transform=_WASRecordedFitStage()) + + assert _stage_fit_weight_records((plain,)) is None + # A declared fitting stage with a hollow transform owes evidence: the + # failing empty artifact, not a named absence. + assert ( + _stage_fit_weight_records( + (SimpleNamespace(name="was_wealth", transform=lambda frame: frame),) + ) + == () + ) + records = _stage_fit_weight_records((plain, hmrc, was)) + assert [record.fit_name for record in records] == [ + "uk_spi_2022_23_income", + "uk_frs_only_spi_fill", + "uk_was_2018_20_wealth:owned_land", + "uk_was_2018_20_wealth:cash_isa", + ] + + class _EmptyFitStage: + fit_weight_records = () + + def __call__(self, frame: Frame) -> Frame: + return frame + + # A scheduled fitting stage with no records is missing evidence: it must + # force the failing empty artifact, never be absorbed by another stage's + # records (the audit-bypass the adversarial review flagged). + assert ( + _stage_fit_weight_records( + (hmrc, SimpleNamespace(name="was_wealth", transform=_EmptyFitStage())) + ) + == () + ) + + +def test_weights_audit_details_carry_the_was_fit_records() -> None: + from microcosm.build.gate_battery import EvidenceContext + from microcosm.build.uk_runtime.battery_bindings import UK_GATE_REGISTRY + + binding = UK_GATE_REGISTRY["weights_audit"] + combined = ( + *_RecordedFitStage.fit_weight_records, + *_WASRecordedFitStage.fit_weight_records, + ) + result = binding.evaluate( + EvidenceContext(artifacts={"fit_weight_records": combined}), + {}, + ) + assert result.passed + resolved = result.details["resolved_weight_kinds"] + assert resolved["uk_was_2018_20_wealth:owned_land"] == "explicit" + assert resolved["uk_was_2018_20_wealth:cash_isa"] == "explicit" + + def test_national_build_runs_preflight_stages_gate_then_staging_write( monkeypatch, tmp_path ) -> None: @@ -930,6 +1002,7 @@ def test_national_build_real_terminal_batch_blocks_incomplete_qrf_before_staging "uk_weight_ratio": "passed", "uk_weights_audit": "passed", "uk_nonnegative_columns": "passed", + "uk_support": "passed", "uk_take_up_signal": "passed", "uk_brma_enum_domain": "passed", # The legacy report omitted unevidenced gates; the battery names diff --git a/packages/microcosm-build/tests/test_uk_regional_uprating.py b/packages/microcosm-build/tests/test_uk_regional_uprating.py new file mode 100644 index 000000000..e263490ec --- /dev/null +++ b/packages/microcosm-build/tests/test_uk_regional_uprating.py @@ -0,0 +1,40 @@ +from __future__ import annotations + +import pandas as pd +import pytest + +from microcosm.build.uk_runtime.regional_uprating import ( + uprate_household_property_by_region, +) + + +def test_regional_property_uprating_scales_owners_only_and_skips_missing_regions(): + household = pd.DataFrame( + { + "household_id": [1, 2, 3, 4, 5], + "region": ["LONDON", "LONDON", "SCOTLAND", "SCOTLAND", "NORTHERN_IRELAND"], + "main_residence_value": [100.0, 0.0, 200.0, 400.0, 500.0], + "property_wealth": [150.0, 10.0, 300.0, 600.0, 700.0], + } + ) + resource = { + "values": [ + {"region": "LONDON", "avg_house_price": 200.0, "dwellings": 1}, + {"region": "SCOTLAND", "avg_house_price": 600.0, "dwellings": 1}, + ] + } + + uprated = uprate_household_property_by_region(household, resource) + + assert uprated.loc[0, "main_residence_value"] == pytest.approx(200.0) + assert uprated.loc[0, "property_wealth"] == pytest.approx(300.0) + assert uprated.loc[1, "main_residence_value"] == 0.0 + assert uprated.loc[1, "property_wealth"] == 10.0 + assert uprated.loc[2, "main_residence_value"] == pytest.approx(400.0) + assert uprated.loc[3, "main_residence_value"] == pytest.approx(800.0) + assert uprated.loc[4, "main_residence_value"] == 500.0 + + +def test_regional_property_uprating_requires_columns() -> None: + with pytest.raises(KeyError, match="property-uprating"): + uprate_household_property_by_region(pd.DataFrame({"region": ["LONDON"]}), {}) diff --git a/packages/microcosm-build/tests/test_uk_release_input_coverage.py b/packages/microcosm-build/tests/test_uk_release_input_coverage.py index 18bc69e33..a92e3b431 100644 --- a/packages/microcosm-build/tests/test_uk_release_input_coverage.py +++ b/packages/microcosm-build/tests/test_uk_release_input_coverage.py @@ -602,7 +602,12 @@ def test_shipped_manifest_is_current(self) -> None: assert manifest.reviewed_exclusions == {} assert load_efrs_parity_known_gaps() == () assert manifest.required_build_stages == frozenset( - {"hmrc_spi_income", "hmrc_cgt_gains"} + { + "hmrc_spi_income", + "hmrc_cgt_gains", + "was_wealth", + "regional_property_uprating", + } ) assert RESTORED_REFERENCE_EFRS_REQUIRED_INPUTS == frozenset( {"charitable_investment_gifts", "gift_aid"} diff --git a/packages/microcosm-build/tests/test_uk_source_runtime.py b/packages/microcosm-build/tests/test_uk_source_runtime.py index 25bb2eb47..b73dabe56 100644 --- a/packages/microcosm-build/tests/test_uk_source_runtime.py +++ b/packages/microcosm-build/tests/test_uk_source_runtime.py @@ -114,9 +114,13 @@ def hmrc(frame: Frame) -> Frame: assert uk_stage_implementations( retained_leaves_transform=retained, hmrc_income_transform=hmrc, + was_wealth_transform=retained, + regional_property_uprating_transform=hmrc, ) == { "frs_hmrc_retained_leaves": retained, "hmrc_spi_income": hmrc, + "was_wealth": retained, + "regional_property_uprating": hmrc, } diff --git a/packages/microcosm-build/tests/test_uk_source_stages.py b/packages/microcosm-build/tests/test_uk_source_stages.py index d7ca43bfe..5f9c5a33d 100644 --- a/packages/microcosm-build/tests/test_uk_source_stages.py +++ b/packages/microcosm-build/tests/test_uk_source_stages.py @@ -32,6 +32,10 @@ "frs_household_draws", "frs_brma", ] +E5_STAGE_NAMES = [ + "was_wealth", + "regional_property_uprating", +] E7_STAGE_NAMES = [ "frs_hmrc_spine_leaves", "spi_support_channel", @@ -41,6 +45,7 @@ "frs_spine", *E3_STAGE_NAMES, *E4_STAGE_NAMES, + *E5_STAGE_NAMES, *E7_STAGE_NAMES, "frs_hmrc_retained_leaves", "hmrc_spi_income", @@ -201,6 +206,8 @@ def test_country_stage_plan_assembles_fourteen_stage_spine_plan(self) -> None: "frs_person_draws": _identity, "frs_household_draws": _identity, "frs_brma": _identity, + "was_wealth": _identity, + "regional_property_uprating": _identity, "frs_hmrc_spine_leaves": _identity, "spi_support_channel": _identity, "hmrc_spi_income_spine": _identity, @@ -277,6 +284,13 @@ def test_e3_outputs_are_backed_by_runtime_written_columns(self) -> None: FRS_TAKE_UP_NONNEGATIVE_OUTPUT_COLUMNS, FRS_TAKE_UP_OUTPUT_COLUMNS, ) + from microcosm.build.uk_runtime.regional_uprating import ( + UK_REGIONAL_PROPERTY_REWRITES, + ) + from microcosm.build.uk_runtime.was_wealth import ( + UK_WAS_WEALTH_NONNEGATIVE_OUTPUT_COLUMNS, + UK_WAS_WEALTH_OUTPUT_COLUMNS, + ) spec = load_country_spec("uk") stages = {stage.stage: stage for stage in spec.sources.stages} @@ -307,6 +321,16 @@ def test_e3_outputs_are_backed_by_runtime_written_columns(self) -> None: stages["frs_household_draws"].outputs == FRS_HOUSEHOLD_DRAW_OUTPUT_COLUMNS ) assert stages["frs_brma"].outputs == FRS_BRMA_OUTPUT_COLUMNS + assert stages["was_wealth"].outputs == UK_WAS_WEALTH_OUTPUT_COLUMNS + assert ( + stages["was_wealth"].nonnegative_outputs + == UK_WAS_WEALTH_NONNEGATIVE_OUTPUT_COLUMNS + ) + assert stages["regional_property_uprating"].outputs == () + assert ( + stages["regional_property_uprating"].rewrites + == UK_REGIONAL_PROPERTY_REWRITES + ) def test_e7_outputs_and_rewrites_are_backed_by_runtime_constants(self) -> None: from microcosm.build.uk_runtime.spi_spine import ( @@ -415,6 +439,17 @@ def test_e3_operation_kinds_are_declared_in_order(self) -> None: "materialize_rules_engine_predictors", "sample_categorical_from_count_table", ] + assert [op.kind for op in stages["was_wealth"].operations] == [ + "derive", + "materialize_rules_engine_predictors", + "fit_weighted_qrf_chain", + "fold_into", + "support_clip", + "allocate_within_group_waterfall", + ] + assert [op.kind for op in stages["regional_property_uprating"].operations] == [ + "uprate_to_regional_reference", + ] assert [op.kind for op in stages["frs_hmrc_spine_leaves"].operations] == [ "retain_adjudicated_frs_hmrc_leaves", "derive", @@ -447,6 +482,10 @@ def test_engine_predictor_and_rewrite_constants_match_manifest(self) -> None: from microcosm.build.uk_runtime.frs_take_up import ( UK_TAKE_UP_ANCHOR_AGGREGATES, ) + from microcosm.build.uk_runtime.was_wealth import ( + UK_WAS_ENGINE_PREDICTORS, + UK_WAS_WEALTH_PREDICTORS, + ) spec = load_country_spec("uk") stages = {stage.stage: stage for stage in spec.sources.stages} @@ -470,6 +509,14 @@ def test_engine_predictor_and_rewrite_constants_match_manifest(self) -> None: tuple(stages["frs_brma"].operations[0].parameters["predictors"]) == UK_BRMA_PREDICTORS ) + assert ( + tuple(stages["was_wealth"].operations[1].parameters["predictors"]) + == UK_WAS_ENGINE_PREDICTORS + ) + assert ( + tuple(stages["was_wealth"].operations[2].parameters["predictors"]) + == UK_WAS_WEALTH_PREDICTORS + ) rate_keys = [ op.parameters["rate_key"] for stage_name in ( @@ -516,6 +563,15 @@ def test_every_e4_stochastic_operation_declares_integer_seed(self) -> None: }: assert isinstance(operation.parameters.get("seed"), int) + def test_e5_qrf_operation_declares_integer_seed(self) -> None: + spec = load_country_spec("uk") + stages = {stage.stage: stage for stage in spec.sources.stages} + + qrf = stages["was_wealth"].operations[2] + + assert qrf.kind == "fit_weighted_qrf_chain" + assert qrf.parameters["seed"] == 0 + def test_e7_declared_seed_lockstep(self) -> None: spec = load_country_spec("uk") stages = {stage.stage: stage for stage in spec.sources.stages} @@ -530,7 +586,7 @@ def test_e7_declared_seed_lockstep(self) -> None: stages["hmrc_spi_income_spine"].operations[3].parameters["seed"] == 43 ) - def test_full_uk_source_stage_plan_compiles_with_e7_stages(self) -> None: + def test_full_uk_source_stage_plan_compiles_with_e4_stages(self) -> None: spec = load_country_spec("uk") implementations = {name: _identity for name in UK_SOURCE_STAGE_NAMES} diff --git a/packages/microcosm-build/tests/test_uk_was_wealth.py b/packages/microcosm-build/tests/test_uk_was_wealth.py new file mode 100644 index 000000000..ebd41ccf6 --- /dev/null +++ b/packages/microcosm-build/tests/test_uk_was_wealth.py @@ -0,0 +1,421 @@ +from __future__ import annotations + +from types import SimpleNamespace + +import numpy as np +import pandas as pd +import pytest + +from microcosm.build.source_manifest import SourceStageSpec +from microcosm.build.uk_runtime.national_frame import uk_national_frame +from microcosm.build.uk_runtime.was_wealth import ( + REGIONS, + UK_WAS_ENGINE_PREDICTOR_ENTITIES, + UK_WAS_WEALTH_OUTPUT_COLUMNS, + UKWASWealthStageTransform, + allocate_student_loan_balance_to_people, + clean_was_household_table, + recipient_predictors, + support_clip_to_donor, +) + + +class _FakeEngine: + """Returns engine variables at their native entity, like the real adapter.""" + + country = "uk" + + def variable_metadata(self, name): + return SimpleNamespace(entity=UK_WAS_ENGINE_PREDICTOR_ENTITIES[name]) + + def materialize(self, frame, variables, period): + assert period == "2023" + tables = { + entity: frame.table(entity) for entity in ("person", "benunit", "household") + } + return { + variable: tables[UK_WAS_ENGINE_PREDICTOR_ENTITIES[variable]][ + variable + ].to_numpy() + for variable in variables + } + + +def _stage() -> SourceStageSpec: + return SourceStageSpec.from_mapping( + { + "stage": "was_wealth", + "survey": "test", + "source": "test", + "grain": "household", + "artifacts": [], + "operations": [ + {"kind": "fit_weighted_qrf_chain", "seed": 0, "n_estimators": 2} + ], + "outputs": list(UK_WAS_WEALTH_OUTPUT_COLUMNS), + "nonnegative_outputs": [ + name + for name in UK_WAS_WEALTH_OUTPUT_COLUMNS + if name != "net_financial_wealth" + ], + } + ) + + +def _raw_was() -> pd.DataFrame: + return pd.DataFrame( + { + "R8xshhwgt": [2.0, 3.0], + "DVLUKValR8_sum": [10.0, 100.0], + "DVPropertyR8": [20.0, 200.0], + "DVFESHARESR8_aggr": [1.0, 2.0], + "DVFShUKVR8_aggr": [3.0, 4.0], + "DVIISAVR8_aggR": [5.0, 6.0], + "DVCISAVR8_aggr": [7.0, 8.0], + "DVFCollVR8_aggr": [9.0, 10.0], + "totalpenr8_aggr": [100.0, 200.0], + "dvvaldbt_scaper8_aggr": [40.0, 50.0], + "NumAdultR8": [2, 1], + "NumCh18R8": [1, 0], + "DVGIPPENR8_AGGR": [11.0, 12.0], + "DVGISER8_AGGR": [13.0, 14.0], + "DVGIINVR8_aggr": [15.0, 16.0], + "DVGIEMPR8_AGGR": [17.0, 18.0], + "HBedRmR8": [3, 4], + "GORR8": [8, 12], + "DVPriRntR8": [1, 2], + "CTAmtR8": [1000.0, 1200.0], + "HFINWNTR8_Sum": [-5.0, 50.0], + "HFINWR8_SUM": [30.0, 40.0], + "DVhvalueR8": [100000.0, 200000.0], + "DVHseValR8_sum": [1000.0, 2000.0], + "DVBlDValR8_sum": [3000.0, 4000.0], + "DVTotinc_bhcR8": [50000.0, 60000.0], + "DVSaValR8_aggr": [500.0, 600.0], + "vcarnr8": [1.2, 2.8], + "Tot_LosR8_aggr": [9000.0, 5000.0], + "Tot_los_exc_SLCR8_aggr": [4000.0, 3000.0], + } + ) + + +def _frame() -> object: + person = pd.DataFrame( + { + "person_id": [101, 102, 201, 202], + "person_benunit_id": [10, 10, 20, 20], + "person_household_id": [1, 1, 2, 2], + "age": [22, 45, 19, 70], + "student_loan_repayments": [20.0, 10.0, 0.0, 0.0], + "student_loans": [0.0, 0.0, 1.0, 0.0], + "highest_education": ["UPPER_SECONDARY", "TERTIARY", "TERTIARY", "LOW"], + "current_education": [ + "NOT_IN_EDUCATION", + "NOT_IN_EDUCATION", + "TERTIARY", + "NOT_IN_EDUCATION", + ], + "private_pension_income": [0.5, 0.5, 2.0, 0.0], + "employment_income": [4.0, 6.0, 12.0, 8.0], + "self_employment_income": [0.0, 0.0, 1.0, 0.0], + "capital_income": [1.0, 2.0, 4.0, 0.0], + } + ) + benunit = pd.DataFrame( + { + "benunit_id": [10, 20], + "num_adults": [2, 2], + "num_children": [0, 0], + } + ) + household = pd.DataFrame( + { + "household_id": [1, 2], + "household_weight": [2.0, 3.0], + "region": ["NORTHERN_IRELAND", "SCOTLAND"], + "num_bedrooms": [3, 4], + "council_tax": [1000.0, 1200.0], + "household_net_income": [50000.0, 60000.0], + "is_renting": [False, True], + } + ) + return uk_national_frame( + person=person, + benunit=benunit, + household=household, + time_period="2023", + ) + + +def test_was_donor_cleaning_arithmetic_and_exact_case_insensitive_columns() -> None: + donor = clean_was_household_table(_raw_was()) + + assert donor["stocks_and_shares_isa"].tolist() == [5.0, 6.0] + assert donor["cash_isa"].tolist() == [7.0, 8.0] + assert donor["corporate_wealth_excl_isa"].tolist() == [73.0, 166.0] + assert donor["corporate_wealth"].tolist() == [78.0, 172.0] + assert donor["student_loan_balance"].tolist() == [5000.0, 2000.0] + assert donor["region"].tolist() == ["LONDON", "SCOTLAND"] + assert donor["is_renting"].tolist() == [True, False] + assert 3 not in REGIONS + + +def test_was_donor_sentinel_codes_recode_to_zero_for_nonnegative_domains() -> None: + raw = _raw_was() + raw.loc[0, "vcarnr8"] = -8 + raw.loc[1, "HBedRmR8"] = -8 + raw.loc[0, "HFINWNTR8_Sum"] = -8.0 + + donor = clean_was_household_table(raw) + + assert donor["num_vehicles"].tolist()[0] == 0 + assert donor["num_bedrooms"].tolist()[1] == 0 + # Genuinely negative domains are never recoded: -8 is a legal net + # financial wealth value. + assert donor["net_financial_wealth"].tolist()[0] == -8.0 + + +def test_was_donor_missing_cash_isa_fails_closed() -> None: + raw = _raw_was().drop(columns=["DVCISAVR8_aggr"]) + + with pytest.raises(ValueError, match="DVCISAVR8_aggr"): + clean_was_household_table(raw) + + +def test_recipient_predictors_remap_ni_to_wales_for_prediction_only() -> None: + predictors = recipient_predictors(_frame(), _FakeEngine()) + + assert predictors["region"].tolist() == ["WALES", "SCOTLAND"] + assert _frame().table("household")["region"].tolist()[0] == "NORTHERN_IRELAND" + + +def test_recipient_predictors_sum_person_and_benunit_variables_to_household() -> None: + predictors = recipient_predictors(_frame(), _FakeEngine()) + + assert predictors["employment_income"].tolist() == [10.0, 20.0] + assert predictors["private_pension_income"].tolist() == [1.0, 2.0] + assert predictors["self_employment_income"].tolist() == [0.0, 1.0] + assert predictors["capital_income"].tolist() == [3.0, 4.0] + assert predictors["num_adults"].tolist() == [2.0, 2.0] + assert predictors["num_children"].tolist() == [0.0, 0.0] + assert predictors["household_net_income"].tolist() == [50000.0, 60000.0] + assert predictors["is_renting"].tolist() == [False, True] + + +def test_recipient_predictors_fail_loud_on_entity_mismatch() -> None: + class _WrongEntityEngine(_FakeEngine): + def variable_metadata(self, name): + if name == "employment_income": + return SimpleNamespace(entity="household") + return super().variable_metadata(name) + + with pytest.raises(ValueError, match="employment_income"): + recipient_predictors(_frame(), _WrongEntityEngine()) + + +def test_student_loan_waterfall_uses_household_ids_not_positions() -> None: + person = _frame().table("person").copy() + balances = pd.Series([900.0, 300.0], index=[1, 2]) + + allocated = allocate_student_loan_balance_to_people( + household_balances=balances, + household_ids=[1, 2], + person=person, + ) + + assert allocated.tolist() == [600.0, 300.0, 300.0, 0.0] + + +def test_student_loan_waterfall_exercises_equal_split_tiers() -> None: + person = pd.DataFrame( + { + "person_household_id": [10, 10, 20, 20, 30, 30], + "age": [30, 40, 16, 17, 60, 70], + "student_loan_repayments": [0, 0, 0, 0, 0, 0], + "student_loans": [0, 0, 0, 0, 0, 0], + "highest_education": ["TERTIARY", "LOW", "LOW", "LOW", "LOW", "LOW"], + "current_education": ["NONE", "NONE", "TERTIARY", "LOW", "LOW", "LOW"], + } + ) + + allocated = allocate_student_loan_balance_to_people( + household_balances=pd.Series([100.0, 80.0, 60.0], index=[10, 20, 30]), + household_ids=[10, 20, 30], + person=person, + ) + + assert allocated.tolist() == [100.0, 0.0, 80.0, 0.0, 30.0, 30.0] + + +def test_support_clip_and_integer_vehicle_output( + monkeypatch: pytest.MonkeyPatch, +) -> None: + donor = clean_was_household_table(_raw_was()) + draws = pd.DataFrame( + {column: [999999.0, -999999.0] for column in UK_WAS_WEALTH_OUTPUT_COLUMNS} + ) + + clipped = support_clip_to_donor(draws, donor) + + assert clipped["owned_land"].tolist() == [100.0, 10.0] + assert clipped["net_financial_wealth"].tolist() == [50.0, -5.0] + + import microcosm.build.uk_runtime.was_wealth as module + + def fake_impute(*args, **kwargs): + result = donor.loc[:, UK_WAS_WEALTH_OUTPUT_COLUMNS].reset_index(drop=True) + result = result.copy() + result["num_vehicles"] = [1.2, 2.8] + return module.UKWASWealthImputationResult( + draws=result, + fit_weight_records=( + module.FitWeightRecord("uk_was_2018_20_wealth:test", "explicit"), + ), + ) + + monkeypatch.setattr(module, "impute_was_wealth", fake_impute) + transform = UKWASWealthStageTransform( + stage=_stage(), + engine=_FakeEngine(), + donor=_raw_was(), + ) + assert transform.fit_weight_records == () + transformed = transform(_frame()) + + assert transform.fit_weight_records == ( + module.FitWeightRecord("uk_was_2018_20_wealth:test", "explicit"), + ) + assert transformed.table("household")["num_vehicles"].tolist() == [1, 3] + assert "student_loan_balance" not in transformed.table("household").columns + assert transformed.table("person")["student_loan_balance"].sum() == pytest.approx( + 7000.0 + ) + + +def test_stage_transform_is_deterministic_with_fast_synthetic_imputer( + monkeypatch: pytest.MonkeyPatch, +) -> None: + import microcosm.build.uk_runtime.was_wealth as module + + donor = clean_was_household_table(_raw_was()) + monkeypatch.setattr( + module, + "impute_was_wealth", + lambda *args, **kwargs: module.UKWASWealthImputationResult( + draws=donor.loc[:, UK_WAS_WEALTH_OUTPUT_COLUMNS].reset_index(drop=True), + fit_weight_records=(), + ), + ) + transform = UKWASWealthStageTransform( + stage=_stage(), + engine=_FakeEngine(), + donor=_raw_was(), + ) + + a = transform(_frame()) + b = transform(_frame()) + + pd.testing.assert_frame_equal(a.table("household"), b.table("household")) + pd.testing.assert_frame_equal(a.table("person"), b.table("person")) + + +def test_stage_transform_requires_a_tab_path_or_donor() -> None: + transform = UKWASWealthStageTransform(stage=_stage(), engine=_FakeEngine()) + + with pytest.raises(ValueError, match="caller-supplied WAS tab path"): + transform(_frame()) + + +def test_stage_transform_refuses_sha_mismatched_tab(tmp_path) -> None: + tab = tmp_path / "was.tab" + tab.write_text("\t".join(_raw_was().columns) + "\n") + stage = SourceStageSpec.from_mapping( + { + "stage": "was_wealth", + "survey": "test", + "source": "test", + "grain": "household", + "artifacts": [ + { + "role": "was_qrf_donor", + "kind": "private_microdata", + "filename": "was.tab", + "sha256": "0" * 64, + "size_bytes": tab.stat().st_size, + "runtime_sha256_required": True, + } + ], + "operations": [ + {"kind": "fit_weighted_qrf_chain", "seed": 0, "n_estimators": 2} + ], + "outputs": list(UK_WAS_WEALTH_OUTPUT_COLUMNS), + } + ) + transform = UKWASWealthStageTransform( + stage=stage, + engine=_FakeEngine(), + was_tab_path=tab, + ) + + with pytest.raises(ValueError, match="hashes to"): + transform(_frame()) + + +def test_was_imputer_uses_checkpointed_chain_segments( + monkeypatch: pytest.MonkeyPatch, +) -> None: + import microcosm.build.uk_runtime.was_wealth as module + import microcosm.fit + + calls = [] + + class FakeQRF: + def __init__(self, *, n_estimators, seed): + assert n_estimators == 7 + assert seed == 0 + + def start_chain(self, donor, predictors, targets, *, weights): + assert weights == "weight" + calls.append((tuple(predictors), tuple(targets))) + return SimpleNamespace(targets=tuple(targets), position=0) + + def fit_draw_next( + self, + donor, + recipient_predictors, + raw_prior_draws, + *, + state, + weights, + ): + assert weights == "weight" + target = state.targets[state.position] + if state.position: + previous = state.targets[state.position - 1] + assert previous in raw_prior_draws + return SimpleNamespace( + target=target, + raw_draw=np.full(len(recipient_predictors), float(state.position + 1)), + weight_kind="explicit", + state=SimpleNamespace( + targets=state.targets, + position=state.position + 1, + ), + ) + + monkeypatch.setattr(microcosm.fit, "RegimeGatedQRF", FakeQRF) + donor = clean_was_household_table(_raw_was()) + recipient = recipient_predictors(_frame(), _FakeEngine()) + + result = module.impute_was_wealth(donor, recipient, seed=0, n_estimators=7) + + assert result.draws.columns.tolist() == list(UK_WAS_WEALTH_OUTPUT_COLUMNS) + assert calls[0][1] == ("owned_land", "property_wealth") + assert calls[1][1] == ("corporate_wealth_excl_isa", "stocks_and_shares_isa") + assert "corporate_wealth" in calls[2][0] + assert calls[2][1][-1] == "cash_isa" + fitted_targets = [name for _, targets in calls for name in targets] + assert [record.fit_name for record in result.fit_weight_records] == [ + f"uk_was_2018_20_wealth:{target}" for target in fitted_targets + ] + assert {record.weight_kind for record in result.fit_weight_records} == {"explicit"} diff --git a/packages/microcosm-build/tests/test_uk_wealth_resources.py b/packages/microcosm-build/tests/test_uk_wealth_resources.py new file mode 100644 index 000000000..376b423bd --- /dev/null +++ b/packages/microcosm-build/tests/test_uk_wealth_resources.py @@ -0,0 +1,143 @@ +from __future__ import annotations + +import importlib.util +import json +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[3] +CSV = ROOT / "packages/microcosm-build/tests/fixtures/uk/regional_land_values.csv" +JSON_RESOURCE = ( + ROOT / "packages/microcosm-build/src/microcosm/build/uk/regional_land_values.json" +) +SUPPORT_BOUNDS = ( + ROOT + / "packages/microcosm-build/src/microcosm/build/uk/was_wealth_support_bounds.json" +) + + +def _build_resource(csv_path: Path): + path = ROOT / "tools/build_uk_regional_land_values.py" + spec = importlib.util.spec_from_file_location("build_uk_regional_land_values", path) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module.build_resource(csv_path) + + +def _support_tool(): + path = ROOT / "tools/build_uk_was_wealth_support_bounds.py" + spec = importlib.util.spec_from_file_location( + "build_uk_was_wealth_support_bounds", path + ) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +def test_regional_land_values_regenerator_round_trips() -> None: + assert _build_resource(CSV) == json.loads(JSON_RESOURCE.read_text()) + + +def test_regional_land_values_resource_shape_and_provenance() -> None: + payload = json.loads(JSON_RESOURCE.read_text()) + + assert payload["version"] == 1 + assert payload["country"] == "uk" + assert len(payload["values"]) == 11 + assert payload["values"][0]["region"] == "NORTH_EAST" + assert payload["source"]["citation_urls"] + + +def test_was_support_bounds_resource_shape() -> None: + payload = json.loads(SUPPORT_BOUNDS.read_text()) + + assert payload["version"] == 1 + assert payload["source"]["sdc_treatment"] + assert "net_financial_wealth" in payload["bounds"] + assert payload["bounds"]["net_financial_wealth"][0] < 0 + + +def test_was_support_bounds_are_generated_from_the_pinned_tab() -> None: + """The release-blocking uk_support gate must never run on placeholder + bounds: the committed resource has to record derivation from the exact + pinned licensed donor tab and carry no placeholder wording (the + adversarial-review blocker).""" + + from microcosm.build.uk_runtime.was_wealth import ( + UK_WAS_WEALTH_OUTPUT_COLUMNS, + WAS_DONOR_SHA256, + ) + + payload = json.loads(SUPPORT_BOUNDS.read_text()) + + assert payload["source"]["tab_sha256"] == WAS_DONOR_SHA256 + assert "placeholder" not in SUPPORT_BOUNDS.read_text().lower() + assert set(payload["bounds"]) == set(UK_WAS_WEALTH_OUTPUT_COLUMNS) + + +def test_was_support_bounds_round_trip_against_licensed_tab() -> None: + """Licensed-only staleness check: regenerate from the local pinned tab + and require byte-identity with the committed resource. Skipped where the + licensed tab is absent (CI is secrets-free by design).""" + + import hashlib + import os + + tab = os.environ.get("POPULACE_UK_WAS_TAB") + if not tab or not Path(tab).is_file(): + pytest.skip("licensed WAS tab not available (set POPULACE_UK_WAS_TAB)") + from microcosm.build.uk_runtime.was_wealth import WAS_DONOR_SHA256 + + assert hashlib.sha256(Path(tab).read_bytes()).hexdigest() == WAS_DONOR_SHA256, ( + "POPULACE_UK_WAS_TAB does not match the pinned donor tab." + ) + payload = _support_tool().build_support_bounds(Path(tab)) + rendered = json.dumps(payload, indent=2, sort_keys=False) + "\n" + assert rendered == SUPPORT_BOUNDS.read_text(encoding="utf-8") + + +def test_support_bounds_tool_rounds_synthetic_donor_outward(tmp_path: Path) -> None: + tab = tmp_path / "was.tab" + rows = { + "R8xshhwgt": [1, 1], + "DVLUKValR8_sum": [10, 100], + "DVPropertyR8": [20, 200], + "DVFESHARESR8_aggr": [1, 2], + "DVFShUKVR8_aggr": [3, 4], + "DVIISAVR8_aggr": [5, 6], + "DVCISAVR8_aggr": [7, 8], + "DVFCollVR8_aggr": [9, 10], + "totalpenr8_aggr": [100, 200], + "dvvaldbt_scaper8_aggr": [40, 50], + "NumAdultR8": [2, 1], + "NumCh18R8": [1, 0], + "DVGIPPENR8_AGGR": [11, 12], + "DVGISER8_AGGR": [13, 14], + "DVGIINVR8_aggr": [15, 16], + "DVGIEMPR8_AGGR": [17, 18], + "HBedRmR8": [3, 4], + "GORR8": [8, 12], + "DVPriRntR8": [1, 2], + "CTAmtR8": [1000, 1200], + "HFINWNTR8_Sum": [-5, 50], + "HFINWR8_SUM": [30, 40], + "DVhvalueR8": [100000, 200000], + "DVHseValR8_sum": [1000, 2000], + "DVBlDValR8_sum": [3000, 4000], + "DVTotinc_bhcR8": [50000, 60000], + "DVSaValR8_aggr": [500, 600], + "vcarnr8": [1, 2], + "Tot_LosR8_aggr": [9000, 5000], + "Tot_los_exc_SLCR8_aggr": [4000, 3000], + } + import pandas as pd + + pd.DataFrame(rows).to_csv(tab, sep="\t", index=False) + + payload = _support_tool().build_support_bounds(tab) + + assert payload["bounds"]["owned_land"] == [10.0, 100.0] + assert payload["bounds"]["net_financial_wealth"] == [-5.0, 50.0] diff --git a/packages/microcosm-build/tests/test_uk_weighted_integrity.py b/packages/microcosm-build/tests/test_uk_weighted_integrity.py index 938fdfead..31b8fab61 100644 --- a/packages/microcosm-build/tests/test_uk_weighted_integrity.py +++ b/packages/microcosm-build/tests/test_uk_weighted_integrity.py @@ -659,7 +659,8 @@ def test_committed_exclusion_registers_load() -> None: assert set(input_mass) == {"efrs-post-calibration"} assert set(input_mass["efrs-post-calibration"]) == { - "charitable_investment_gifts" + "charitable_investment_gifts", + "owned_land", } assert qrf_tail == {} assert uk_default_input_mass_reviewed_exclusions() is ( diff --git a/packages/microcosm-build/tests/test_uk_weighted_integrity_measurement.py b/packages/microcosm-build/tests/test_uk_weighted_integrity_measurement.py index 179ed5f23..9c1671e48 100644 --- a/packages/microcosm-build/tests/test_uk_weighted_integrity_measurement.py +++ b/packages/microcosm-build/tests/test_uk_weighted_integrity_measurement.py @@ -14,8 +14,11 @@ from pathlib import Path import numpy as np +import pandas as pd import pytest +from microcosm.build.uk_runtime.national_frame import uk_national_frame + ROOT = Path(__file__).resolve().parents[3] @@ -133,3 +136,31 @@ def test_minimum_count_below_the_standard_floor_is_refused( def test_default_minimum_count_follows_the_secondary_disclosure_advice() -> None: assert MEASUREMENT.DEFAULT_SDC_MINIMUM_COUNT == 10 assert all(k >= 10 for k in MEASUREMENT.DEFAULT_TOP_K_GRID) + + +def test_household_grain_was_tail_columns_use_household_weights() -> None: + frame = uk_national_frame( + person=pd.DataFrame( + { + "person_id": [1, 2], + "person_benunit_id": [1, 2], + "person_household_id": [10, 20], + } + ), + benunit=pd.DataFrame({"benunit_id": [1, 2]}), + household=pd.DataFrame( + { + "household_id": [10, 20], + "household_weight": [2.0, 3.0], + "owned_land": [100.0, 200.0], + "cash_isa": [10.0, 20.0], + } + ), + time_period="2023", + ) + + values, weights = MEASUREMENT.household_tail_concentration_columns(frame) + + assert values["owned_land"].tolist() == [100.0, 200.0] + assert weights["owned_land"].tolist() == [2.0, 3.0] + assert values["cash_isa"].tolist() == [10.0, 20.0] diff --git a/packages/microcosm-data/src/microcosm/data/contract.py b/packages/microcosm-data/src/microcosm/data/contract.py index 5d285be67..92273382c 100644 --- a/packages/microcosm-data/src/microcosm/data/contract.py +++ b/packages/microcosm-data/src/microcosm/data/contract.py @@ -315,6 +315,7 @@ "surface", } ), + "support": frozenset({"columns_checked"}), } _UK_TARGET_GEOGRAPHY_LEVELS = frozenset( {"national", "region", "country", "local_authority", "constituency"} @@ -342,13 +343,13 @@ # fingerprint derives from the manifest digest. Editing the spec moves all # three here in the same reviewed change. _UK_GATE_BATTERY_POLICY_SHA256 = ( - "609075af473c64fe7dbcb035b9254121d6c9c000c28fc67a14417bb02657d08f" + "5ddd3da9a52b0dc19ba1c97315f0e4f8acdedf2b74ea29bff512cbdb57de1cab" ) _UK_GATE_BATTERY_GATES_MANIFEST_SHA256 = ( - "610512a5bbddeba355cf57de52579eca5f36cf43788be239307cc5c025e76783" + "8d58fffe5e6542a7f10578076bbcc943cf587f9f22017ccc630450007b5b6166" ) _UK_GATE_BATTERY_SPEC_FINGERPRINT = ( - "cb25537c8a99aa6c44911b098df10dfb7ba143dc3210400b61f847c3d0c9b12d" + "b7b645deee1b15750403f98a4c7dac09d6e08440a878b8d1ff83a15e9195b809" ) #: Spec entry id -> the legacy gate name whose observable detail checks #: apply unchanged (the battery re-keys the report by entry id; the gate @@ -361,6 +362,7 @@ "uk_weight_ratio": "weight_ratio", "uk_weights_audit": "weights_audit", "uk_nonnegative_columns": "nonnegative_columns", + "uk_support": "support", "uk_export_surface": "export_surface", "uk_take_up_signal": "take_up_signal", "uk_brma_enum_domain": "enum_domain", @@ -386,6 +388,7 @@ "uk_weight_ratio": ("weight_ratio", "terminal"), "uk_weights_audit": ("weights_audit", "terminal"), "uk_nonnegative_columns": ("nonnegative_columns", "terminal"), + "uk_support": ("support", "terminal"), "uk_export_surface": ("export_surface", "terminal"), "uk_take_up_signal": ("take_up_signal", "terminal"), "uk_brma_enum_domain": ("enum_domain", "terminal"), @@ -410,7 +413,7 @@ # canonical hash; this pins the wrapped digest so the entry's evidence line # still binds the enhanced-FRS incumbent totals. _UK_GATE_BATTERY_INPUT_MASS_EVIDENCE_SHA256 = ( - "a77111e4acecd3945d69f77f58209bf6a58eb72d39a47de9cb01e5b73d2592f4" + "df74556909990345bc9032f6f5db7f817d9c273f75522a85ab6e332dd8dc7355" ) # The degenerate binding's evidence payload digests the resolved exclusion # records; for a release that must be the committed register, so its digest diff --git a/packages/microcosm-data/tests/test_contract.py b/packages/microcosm-data/tests/test_contract.py index 457507bb5..1c92fd75c 100644 --- a/packages/microcosm-data/tests/test_contract.py +++ b/packages/microcosm-data/tests/test_contract.py @@ -92,7 +92,27 @@ "adjudication": "microcosm#630", "approved_on": "2026-08-17", "expires_on": "2027-02-17", - } + }, + "owned_land": { + "reason": ( + "Sparse heavy-tailed WAS donor column (0.7 percent weighted " + "nonzero share) whose weighted total is dominated by a handful " + "of large farm/estate records: the E5 stability receipt " + "(data/ukds/acceptance/e5/owned_land_stability_receipt.json) " + "measures a 37.7 percent national and 2.41x London swing " + "between adjacent seeds, the same realization-variance class " + "the archived incumbent data repo records at uk-data#448 (4.6x " + "Wales swing across releases). Register parity at this grain " + "is not meaningful until the whole-spine comparison; the " + "one-month expiry enforces the end-of-workstream revisit " + "registered on microcosm#145 (winsorised donor or separate " + "land imputation are the candidate remedies)." + ), + "approved_by": "juaristi22", + "adjudication": "microcosm#714", + "approved_on": "2026-08-19", + "expires_on": "2026-09-19", + }, } GIT_COMMIT = "5fa48f07436a806ad75ff76fd22cfb8613bddbe0" DATASET_SHA = "d" * 64 @@ -123,19 +143,19 @@ def _trusted_terminal_gate_signing_key(monkeypatch) -> None: UK_GATE_BATTERY_PRODUCER = "microcosm.build.gate_battery" UK_GATE_BATTERY_SIGNING_KEY_ENV = "MICROCOSM_UK_TERMINAL_GATE_SIGNING_KEY" UK_GATE_BATTERY_POLICY_SHA256 = ( - "609075af473c64fe7dbcb035b9254121d6c9c000c28fc67a14417bb02657d08f" + "5ddd3da9a52b0dc19ba1c97315f0e4f8acdedf2b74ea29bff512cbdb57de1cab" ) UK_GATE_BATTERY_GATES_MANIFEST_SHA256 = ( - "610512a5bbddeba355cf57de52579eca5f36cf43788be239307cc5c025e76783" + "8d58fffe5e6542a7f10578076bbcc943cf587f9f22017ccc630450007b5b6166" ) UK_GATE_BATTERY_SPEC_FINGERPRINT = ( - "cb25537c8a99aa6c44911b098df10dfb7ba143dc3210400b61f847c3d0c9b12d" + "b7b645deee1b15750403f98a4c7dac09d6e08440a878b8d1ff83a15e9195b809" ) UK_GATE_BATTERY_DEGENERATE_EVIDENCE_SHA256 = ( "d0d024043132fa07c378c393dbe2b24fe99bf19e876bcc39997d2c80cc9bd4f6" ) UK_GATE_BATTERY_INPUT_MASS_EVIDENCE_SHA256 = ( - "a77111e4acecd3945d69f77f58209bf6a58eb72d39a47de9cb01e5b73d2592f4" + "df74556909990345bc9032f6f5db7f817d9c273f75522a85ab6e332dd8dc7355" ) #: Spec entry id -> (neutral gate name, phase, legacy detail-schema name). UK_GATE_BATTERY_ENTRIES = { @@ -164,6 +184,7 @@ def _trusted_terminal_gate_signing_key(monkeypatch) -> None: "terminal", "nonnegative_columns", ), + "uk_support": ("support", "terminal", "support"), "uk_export_surface": ("export_surface", "terminal", "export_surface"), "uk_take_up_signal": ("take_up_signal", "terminal", "take_up_signal"), "uk_brma_enum_domain": ("enum_domain", "terminal", "enum_domain"), @@ -748,6 +769,8 @@ def _terminal_gate_details(name: str) -> dict: "atol": 0.0, "chunk_size": 1_000_000, } + if name == "support": + return {"columns_checked": 13} if name == "export_surface": return { "candidate_columns": 1, diff --git a/packages/microcosm-fit/src/microcosm/fit/__init__.py b/packages/microcosm-fit/src/microcosm/fit/__init__.py index bd7173587..7b987f8f6 100644 --- a/packages/microcosm-fit/src/microcosm/fit/__init__.py +++ b/packages/microcosm-fit/src/microcosm/fit/__init__.py @@ -67,6 +67,12 @@ def _assert_frame_compatible(version: str, required: tuple[int, int]) -> None: _assert_frame_compatible(_frame_version, _REQUIRED_FRAME_SERIES) +from microcosm.fit.cache import ( # noqa: E402 - after the compatibility gate + QRFCacheKey, + cache_path, + load_fitted_qrf_cache, + save_fitted_qrf_cache, +) from microcosm.fit.model import ( # noqa: E402 - after the compatibility gate DESIGN_WEIGHTS, EXPLICIT_WEIGHTS, @@ -141,6 +147,10 @@ def fit( "QRFChainState", "QRFChainStepResult", "Regime", + "QRFCacheKey", + "cache_path", + "load_fitted_qrf_cache", + "save_fitted_qrf_cache", "DEFAULT_N_ESTIMATORS", "DEFAULT_ZERO_ATOL", "fit", diff --git a/packages/microcosm-fit/src/microcosm/fit/cache.py b/packages/microcosm-fit/src/microcosm/fit/cache.py new file mode 100644 index 000000000..ea64f2892 --- /dev/null +++ b/packages/microcosm-fit/src/microcosm/fit/cache.py @@ -0,0 +1,97 @@ +"""Default-off cache helpers for fitted conditional models.""" + +from __future__ import annotations + +import hashlib +import json +import pickle +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +SCHEMA_VERSION = 1 + + +@dataclass(frozen=True) +class QRFCacheKey: + """Stable metadata identity for a fitted QRF model.""" + + donor_sha256: str + predictors: tuple[str, ...] + targets: tuple[str, ...] + seed: int + weight_kind: str + package_versions: dict[str, str] + extra: dict[str, Any] | None = None + + def payload(self) -> dict[str, Any]: + return { + "schema_version": SCHEMA_VERSION, + "donor_sha256": self.donor_sha256, + "predictors": list(self.predictors), + "targets": list(self.targets), + "seed": int(self.seed), + "weight_kind": self.weight_kind, + "package_versions": dict(sorted(self.package_versions.items())), + "extra": {} if self.extra is None else self.extra, + } + + def digest(self) -> str: + return _digest_payload(self.payload()) + + +def cache_path(cache_dir: str | Path, key: QRFCacheKey) -> Path: + """Return the model path for ``key`` under an explicit cache directory.""" + + return Path(cache_dir).expanduser().resolve() / f"{key.digest()}.pkl" + + +def save_fitted_qrf_cache( + model: object, + cache_dir: str | Path, + key: QRFCacheKey, +) -> Path: + """Save a fitted model under ``key`` and return the written path.""" + + path = cache_path(cache_dir, key) + path.parent.mkdir(parents=True, exist_ok=True) + payload = { + "schema_version": SCHEMA_VERSION, + "metadata": key.payload(), + "metadata_digest": key.digest(), + "model": model, + } + tmp = path.with_suffix(path.suffix + ".tmp") + with tmp.open("wb") as file: + pickle.dump(payload, file, protocol=pickle.HIGHEST_PROTOCOL) + tmp.replace(path) + return path + + +def load_fitted_qrf_cache(cache_dir: str | Path, key: QRFCacheKey) -> object | None: + """Load the cached model for ``key``, returning ``None`` on a miss.""" + + path = cache_path(cache_dir, key) + if not path.exists(): + return None + with path.open("rb") as file: + payload = pickle.load(file) + if not isinstance(payload, dict): + raise ValueError(f"QRF cache file {path} does not contain a payload object.") + if payload.get("schema_version") != SCHEMA_VERSION: + raise ValueError(f"QRF cache file {path} has unsupported schema version.") + if payload.get("metadata_digest") != key.digest(): + return None + if payload.get("metadata") != key.payload(): + return None + return payload.get("model") + + +def _digest_payload(payload: dict[str, Any]) -> str: + data = json.dumps( + payload, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode("utf-8") + return hashlib.sha256(data).hexdigest() diff --git a/packages/microcosm-fit/tests/test_cache.py b/packages/microcosm-fit/tests/test_cache.py new file mode 100644 index 000000000..7cac908db --- /dev/null +++ b/packages/microcosm-fit/tests/test_cache.py @@ -0,0 +1,27 @@ +from __future__ import annotations + +from microcosm.fit.cache import ( + QRFCacheKey, + load_fitted_qrf_cache, + save_fitted_qrf_cache, +) + + +def _key(seed: int = 0) -> QRFCacheKey: + return QRFCacheKey( + donor_sha256="a" * 64, + predictors=("x",), + targets=("y",), + seed=seed, + weight_kind="explicit", + package_versions={"microcosm-fit": "test"}, + ) + + +def test_qrf_cache_save_load_and_miss(tmp_path) -> None: + model = {"fitted": True} + path = save_fitted_qrf_cache(model, tmp_path, _key()) + + assert path.exists() + assert load_fitted_qrf_cache(tmp_path, _key()) == model + assert load_fitted_qrf_cache(tmp_path, _key(seed=1)) is None diff --git a/tools/build_uk_frs_spine.py b/tools/build_uk_frs_spine.py index b81960bb4..d0ce32677 100644 --- a/tools/build_uk_frs_spine.py +++ b/tools/build_uk_frs_spine.py @@ -52,12 +52,16 @@ from microcosm.build.uk_runtime.hmrc_replay import write_hmrc_replay_report from microcosm.build.uk_runtime.national_build import write_uk_national_frame from microcosm.build.uk_runtime.national_frame import uk_household_weight_kind +from microcosm.build.uk_runtime.regional_uprating import ( + UKRegionalPropertyUpratingStageTransform, +) from microcosm.build.uk_runtime.spi_spine import ( UKFRSHMRCSpineLeavesStageTransform, UKSPIIncomeSpineStageTransform, UKSPISupportChannelStageTransform, ) from microcosm.build.uk_runtime.take_up_contract import load_uk_take_up_contract +from microcosm.build.uk_runtime.was_wealth import UKWASWealthStageTransform from microcosm.frame.adapters.policyengine_uk import PolicyEngineUKEngine _PIPELINE = "uk-frs-spine" @@ -75,6 +79,8 @@ "frs_person_draws", "frs_household_draws", "frs_brma", + "was_wealth", + "regional_property_uprating", "frs_hmrc_spine_leaves", "spi_support_channel", "hmrc_spi_income_spine", @@ -118,6 +124,11 @@ def _parse_args(argv: list[str] | None = None) -> argparse.Namespace: type=Path, help="Optional directory for a copy of the completed spine checkpoint.", ) + parser.add_argument( + "--was-tab", + type=Path, + help="Caller-supplied private WAS round-8 household tab for was_wealth.", + ) parser.add_argument( "--emit-nonzero-shares", type=Path, @@ -166,31 +177,32 @@ def _artifact_pins(stages) -> dict[str, dict[str, object]]: pins = {} for stage in stages: for artifact in stage.artifacts: - if "table" not in artifact: + key = artifact.get("table", artifact.get("filename")) + if key is None: continue - table = str(artifact["table"]) + key = str(key) pin = { "locator": str(artifact["locator"]), "sha256": str(artifact["sha256"]), "size_bytes": int(artifact["size_bytes"]), } - if table in pins and pins[table] != pin: + if key in pins and pins[key] != pin: raise ValueError( - f"FRS tab {table!r} has inconsistent artifact pins across stages." + f"UK source artifact {key!r} has inconsistent pins across stages." ) - pins[table] = pin + pins[key] = pin return dict(sorted(pins.items())) def _stage_artifact_pins(stage) -> dict[str, dict[str, object]]: return { - str(artifact["table"]): { + str(artifact.get("table", artifact.get("filename"))): { "locator": str(artifact["locator"]), "sha256": str(artifact["sha256"]), "size_bytes": int(artifact["size_bytes"]), } for artifact in stage.artifacts - if "table" in artifact + if "table" in artifact or "filename" in artifact } @@ -309,6 +321,8 @@ def _declared_seeds(stages) -> dict[str, dict[str, int]]: stage_seeds["stage1"] = seed elif operation.kind == "fit_weighted_qrf_stage2": stage_seeds["stage2"] = seed + elif operation.kind == "fit_weighted_qrf_chain": + stage_seeds[stage.stage] = seed if stage_seeds: declared[stage.stage] = stage_seeds return declared @@ -446,7 +460,12 @@ def main(argv: list[str] | None = None) -> int: if spec.sources is None: raise ValueError("UK country spec has no source stages.") stages_by_name = spec.sources.stage_map() - stages = [stages_by_name[name] for name in _STAGE_NAMES] + stage_names = tuple(name for name in _STAGE_NAMES if name in stages_by_name) + if "was_wealth" in stage_names and args.was_tab is None: + raise ValueError( + "--was-tab is required when the was_wealth stage is scheduled." + ) + stages = [stages_by_name[name] for name in stage_names] artifact_pins = _artifact_pins(stages) resource_pins = _resource_pins(stages, spec) input_artifact_pins = _input_artifact_pins(stages) @@ -461,7 +480,7 @@ def main(argv: list[str] | None = None) -> int: ) run_config = { "pipeline": _PIPELINE, - "stages": list(_STAGE_NAMES), + "stages": list(stage_names), "artifact_pins_digest": state.input_pins_digest, "spine_h5": str(args.spine_h5), } @@ -476,65 +495,78 @@ def main(argv: list[str] | None = None) -> int: args.hmrc_ods, stage=stages_by_name["hmrc_spi_income_spine"], ) + implementations = { + "frs_spine": UKFRSSpineStageTransform( + args.frs_raw_dir, + stage=stages_by_name["frs_spine"], + ), + "frs_employment": UKFRSEmploymentStageTransform( + args.frs_raw_dir, + stage=stages_by_name["frs_employment"], + ), + "frs_council_tax": UKFRSCouncilTaxStageTransform( + args.frs_raw_dir, + stage=stages_by_name["frs_council_tax"], + ), + "frs_disability": UKFRSDisabilityStageTransform( + stage=stages_by_name["frs_disability"], + ), + "frs_education": UKFRSEducationStageTransform( + args.frs_raw_dir, + stage=stages_by_name["frs_education"], + ), + "frs_legacy_proxies": UKFRSLegacyProxiesStageTransform( + args.frs_raw_dir, + stage=stages_by_name["frs_legacy_proxies"], + engine=engine, + ), + "frs_education_grant_split": ( + UKFRSEducationGrantSplitStageTransform( + stage=stages_by_name["frs_education_grant_split"], + engine=engine, + ) + ), + "frs_take_up": UKFRSTakeUpStageTransform( + contract=stochastic_contract, + stage=stages_by_name["frs_take_up"], + ), + "frs_person_draws": UKFRSPersonDrawsStageTransform( + contract=stochastic_contract, + stage=stages_by_name["frs_person_draws"], + ), + "frs_household_draws": UKFRSHouseholdDrawsStageTransform( + contract=stochastic_contract, + stage=stages_by_name["frs_household_draws"], + ), + "frs_brma": UKFRSBRMAStageTransform( + stage=stages_by_name["frs_brma"], + engine=engine, + ), + } + if "was_wealth" in stage_names: + implementations["was_wealth"] = UKWASWealthStageTransform( + stage=stages_by_name["was_wealth"], + engine=engine, + was_tab_path=args.was_tab, + ) + if "regional_property_uprating" in stage_names: + implementations["regional_property_uprating"] = ( + UKRegionalPropertyUpratingStageTransform( + stage=stages_by_name["regional_property_uprating"], + ) + ) + implementations["frs_hmrc_spine_leaves"] = UKFRSHMRCSpineLeavesStageTransform( + args.frs_raw_dir, + stage=stages_by_name["frs_hmrc_spine_leaves"], + ) + implementations["spi_support_channel"] = UKSPISupportChannelStageTransform( + stage=stages_by_name["spi_support_channel"], + ) + implementations["hmrc_spi_income_spine"] = hmrc_spine_transform plan = country_stage_plan( spec, - { - "frs_spine": UKFRSSpineStageTransform( - args.frs_raw_dir, - stage=stages_by_name["frs_spine"], - ), - "frs_employment": UKFRSEmploymentStageTransform( - args.frs_raw_dir, - stage=stages_by_name["frs_employment"], - ), - "frs_council_tax": UKFRSCouncilTaxStageTransform( - args.frs_raw_dir, - stage=stages_by_name["frs_council_tax"], - ), - "frs_disability": UKFRSDisabilityStageTransform( - stage=stages_by_name["frs_disability"], - ), - "frs_education": UKFRSEducationStageTransform( - args.frs_raw_dir, - stage=stages_by_name["frs_education"], - ), - "frs_legacy_proxies": UKFRSLegacyProxiesStageTransform( - args.frs_raw_dir, - stage=stages_by_name["frs_legacy_proxies"], - engine=engine, - ), - "frs_education_grant_split": ( - UKFRSEducationGrantSplitStageTransform( - stage=stages_by_name["frs_education_grant_split"], - engine=engine, - ) - ), - "frs_take_up": UKFRSTakeUpStageTransform( - contract=stochastic_contract, - stage=stages_by_name["frs_take_up"], - ), - "frs_person_draws": UKFRSPersonDrawsStageTransform( - contract=stochastic_contract, - stage=stages_by_name["frs_person_draws"], - ), - "frs_household_draws": UKFRSHouseholdDrawsStageTransform( - contract=stochastic_contract, - stage=stages_by_name["frs_household_draws"], - ), - "frs_brma": UKFRSBRMAStageTransform( - stage=stages_by_name["frs_brma"], - engine=engine, - ), - "frs_hmrc_spine_leaves": UKFRSHMRCSpineLeavesStageTransform( - args.frs_raw_dir, - stage=stages_by_name["frs_hmrc_spine_leaves"], - ), - "spi_support_channel": UKSPISupportChannelStageTransform( - stage=stages_by_name["spi_support_channel"], - ), - "hmrc_spi_income_spine": hmrc_spine_transform, - }, - stage_names=_STAGE_NAMES, + implementations, + stage_names=stage_names, ) frame, records = plan.run(uk_frs_spine_seed_frame()) append_phase(state, "spine_built") diff --git a/tools/build_uk_regional_land_values.py b/tools/build_uk_regional_land_values.py new file mode 100644 index 000000000..20c06dfa9 --- /dev/null +++ b/tools/build_uk_regional_land_values.py @@ -0,0 +1,95 @@ +"""Build the UK regional land-values JSON resource from the public CSV.""" + +from __future__ import annotations + +import argparse +import csv +import hashlib +import json +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_INPUT = ( + REPO_ROOT / "packages/microcosm-build/tests/fixtures/uk/regional_land_values.csv" +) +DEFAULT_OUTPUT = ( + REPO_ROOT + / "packages/microcosm-build/src/microcosm/build/uk/regional_land_values.json" +) +EXPECTED_SHA256 = "088b777d73e71c07890ac9e1add9b683aa962da69ca79c25dd0b54d7e325cfb6" +EXPECTED_ROWS = 11 + + +def build_resource(csv_path: Path) -> dict[str, object]: + if _sha256(csv_path) != EXPECTED_SHA256: + raise ValueError(f"{csv_path} does not match the reviewed CSV sha256 pin.") + with csv_path.open(newline="", encoding="utf-8") as stream: + rows = list(csv.DictReader(stream)) + if len(rows) != EXPECTED_ROWS: + raise ValueError(f"{csv_path} must contain {EXPECTED_ROWS} rows.") + values = [ + { + "region": row["region"], + "avg_house_price": int(row["avg_house_price"]), + "dwellings": int(row["dwellings"]), + } + for row in rows + ] + return { + "version": 1, + "country": "uk", + "policy": ( + "Public regional land-value reference for deterministic post-WAS " + "property uprating." + ), + "source": { + "provenance": ( + "Incumbent public regional_land_values.csv re-homed as a JSON " + "country-package resource for Microcosm. Values combine MHCLG " + "dwellings and ONS UK House Price Index average prices for " + "December 2025." + ), + "citation_urls": [ + "https://www.gov.uk/government/collections/dwelling-stock-including-vacants", + "https://www.ons.gov.uk/economy/inflationandpriceindices/bulletins/housepriceindex/december2025", + ], + }, + "values": values, + "chronicle": [ + "2026-08-18: added for microcosm#681 E5 WAS wealth and regional property uprating." + ], + } + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--input-csv", type=Path, default=DEFAULT_INPUT) + parser.add_argument("--output-json", type=Path, default=DEFAULT_OUTPUT) + parser.add_argument("--check", action="store_true") + args = parser.parse_args(argv) + payload = build_resource(args.input_csv) + rendered = json.dumps( + payload, + indent=2, + sort_keys=False, + ensure_ascii=True, + ) + rendered += "\n" + if args.check: + if args.output_json.read_text(encoding="utf-8") != rendered: + raise SystemExit(f"{args.output_json} is stale.") + return 0 + args.output_json.write_text(rendered, encoding="utf-8") + return 0 + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/build_uk_release_input_coverage_manifest.py b/tools/build_uk_release_input_coverage_manifest.py index 21e243931..b037486ed 100644 --- a/tools/build_uk_release_input_coverage_manifest.py +++ b/tools/build_uk_release_input_coverage_manifest.py @@ -45,6 +45,7 @@ MANIFEST_PATH = UK_PACKAGE_DIR / "release_input_coverage_manifest.json" HMRC_SOURCE_STAGES_PATH = UK_PACKAGE_DIR / "hmrc_income_source_stages.json" CGT_SOURCE_STAGES_PATH = UK_PACKAGE_DIR / "cgt_source_stages.json" +SOURCE_STAGES_PATH = UK_PACKAGE_DIR / "source_stages.json" CANDIDATE_REPO_ID = "policyengine/populace-uk-private" CANDIDATE_REPO_TYPE = "dataset" @@ -795,6 +796,14 @@ def build_manifest( "hmrc_spi_income": _hmrc_family_coverage_contract( candidate_source=candidate_source ), + "was_wealth": _source_stage_family_coverage_contract( + stage_name="was_wealth", + candidate_source=candidate_source, + ), + "regional_property_uprating": _source_stage_family_coverage_contract( + stage_name="regional_property_uprating", + candidate_source=candidate_source, + ), }, "derivation": ( "Surface = efrs_parity_reference.json populated effective loader " @@ -919,6 +928,47 @@ def _cgt_family_coverage_contract( } +def _source_stage_family_coverage_contract( + *, + stage_name: str, + candidate_source: dict[str, Any], +) -> dict[str, Any]: + payload = _load(SOURCE_STAGES_PATH) + stages = payload.get("stages") + if not isinstance(stages, list): + raise ValueError(f"{SOURCE_STAGES_PATH}: expected source stages list.") + matches = [ + stage + for stage in stages + if isinstance(stage, dict) and stage.get("stage") == stage_name + ] + if len(matches) != 1: + raise ValueError( + f"{SOURCE_STAGES_PATH}: expected exactly one {stage_name!r} stage." + ) + stage = matches[0] + return { + "status": "required_at_build", + "stage": stage_name, + "source_manifest": SOURCE_STAGES_PATH.name, + "source_manifest_sha256": _sha256(SOURCE_STAGES_PATH), + "base_candidate_sha256": str(candidate_source["sha256"]), + "base_candidate_tier": validate_uk_release_tier(candidate_source["tier"]), + "source_vintages": { + "survey": str(stage.get("survey", "")), + "source": str(stage.get("source", "")), + }, + "output_weight_kind": "importance", + "required_mass_change_reason": ( + "E5 source-stage transform preserves household rows and typed " + "household weights; total household mass is conserved." + ), + "outputs": list(stage.get("outputs", [])), + "rewrites": list(stage.get("rewrites", [])), + "effective_mass_requirements": {}, + } + + def _hmrc_family_coverage_contract( *, candidate_source: dict[str, Any], diff --git a/tools/build_uk_was_wealth_support_bounds.py b/tools/build_uk_was_wealth_support_bounds.py new file mode 100644 index 000000000..886963f44 --- /dev/null +++ b/tools/build_uk_was_wealth_support_bounds.py @@ -0,0 +1,109 @@ +"""Build disclosure-safe WAS wealth support bounds from a donor-like CSV.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +from pathlib import Path + +import pandas as pd + +from microcosm.build.uk_runtime.was_wealth import ( + UK_WAS_WEALTH_OUTPUT_COLUMNS, + clean_was_household_table, + donor_realized_ranges, +) + +REPO_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_OUTPUT = ( + REPO_ROOT + / "packages/microcosm-build/src/microcosm/build/uk/was_wealth_support_bounds.json" +) + + +def build_support_bounds(tab_path: Path) -> dict[str, object]: + tab_bytes = tab_path.read_bytes() + tab_sha256 = hashlib.sha256(tab_bytes).hexdigest() + raw = pd.read_csv(tab_path, sep="\t") + donor = clean_was_household_table(raw) + exact = donor_realized_ranges(donor) + return { + "version": 1, + "country": "uk", + "policy": ( + "Disclosure-safe outward-rounded WAS wealth support bounds for " + "the UK support gate, generated from the donor household tab " + "recorded under source.tab_sha256 by " + "tools/build_uk_was_wealth_support_bounds.py. Values are " + "donor-realized exact ranges rounded outward to one significant " + "figure, so unit-record minima and maxima are never committed." + ), + "source": { + "ukds_study_number": 7215, + "doi": "10.5255/UKDA-SN-7215-20", + "artifact": "was_round_8_hhold_eul_may_2025_230525.tab", + "tab_sha256": tab_sha256, + "sdc_treatment": ( + "Exact donor min/max values are rounded outward to one " + "significant figure before commit, so no committed number " + "is an exact unit-record value. Caveat considered and " + "accepted: WAS applies top-coding and a round top-code can " + "survive outward rounding unchanged; top-codes are " + "published survey methodology rather than unit-record " + "disclosures. Adjudicated acceptable under the UKDS EUL " + "disclosure rules by juaristi22 on 2026-08-19 " + "(microcosm#714)." + ), + }, + "bounds": { + column: list(_outward_round(bounds)) + for column, bounds in exact.items() + if column in UK_WAS_WEALTH_OUTPUT_COLUMNS + }, + "chronicle": [ + "Support bounds generated from the tab pinned as " + f"source.tab_sha256 ({tab_sha256[:12]}…) with outward SDC " + "rounding." + ], + } + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--was-tab", type=Path, required=True) + parser.add_argument("--output-json", type=Path, default=DEFAULT_OUTPUT) + parser.add_argument("--check", action="store_true") + args = parser.parse_args(argv) + payload = build_support_bounds(args.was_tab) + rendered = json.dumps(payload, indent=2, sort_keys=False) + "\n" + if args.check: + if args.output_json.read_text(encoding="utf-8") != rendered: + raise SystemExit(f"{args.output_json} is stale.") + return 0 + args.output_json.write_text(rendered, encoding="utf-8") + return 0 + + +def _outward_round(bounds: tuple[float, float]) -> tuple[float, float]: + lo, hi = bounds + return (_round_down(lo), _round_up(hi)) + + +def _round_down(value: float) -> float: + if value == 0: + return 0.0 + magnitude = 10 ** math.floor(math.log10(abs(value))) + return math.floor(value / magnitude) * magnitude + + +def _round_up(value: float) -> float: + if value == 0: + return 0.0 + magnitude = 10 ** math.floor(math.log10(abs(value))) + return math.ceil(value / magnitude) * magnitude + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/measure_uk_weighted_integrity_baselines.py b/tools/measure_uk_weighted_integrity_baselines.py index d9a2e4be4..b147f9239 100644 --- a/tools/measure_uk_weighted_integrity_baselines.py +++ b/tools/measure_uk_weighted_integrity_baselines.py @@ -48,6 +48,7 @@ uk_hmrc_weighted_qrf_output_columns, ) from microcosm.build.uk_runtime.national_build import load_uk_national_frame +from microcosm.build.uk_runtime.was_wealth import UK_WAS_WEALTH_HOUSEHOLD_OUTPUT_COLUMNS from microcosm.build.uk_runtime.weighted_integrity import ( uk_input_mass_totals, uk_qrf_tail_concentration_columns, @@ -60,6 +61,7 @@ # 30; raise this with --sdc-minimum-count when a study's Special Conditions # (EUL clause 3) require it. DEFAULT_SDC_MINIMUM_COUNT = 10 +UK_WAS_HOUSEHOLD_WEIGHTED_INTEGRITY_COLUMNS = UK_WAS_WEALTH_HOUSEHOLD_OUTPUT_COLUMNS def _parse_args() -> argparse.Namespace: @@ -202,6 +204,16 @@ def _measure( ) for column in sorted(values) } + household_values, household_weights = household_tail_concentration_columns(frame) + household_tail = { + column: _tail_measurements( + household_values[column], + household_weights[column], + top_k_grid, + minimum_count=minimum_count, + ) + for column in sorted(household_values) + } return { "path": str(path.resolve()), "filename": path.name, @@ -215,6 +227,8 @@ def _measure( "input_mass_totals": dict(sorted(totals.items())), "qrf_surface": surface, "qrf_tail": qrf_tail, + "household_qrf_surface": "was_wealth", + "household_qrf_tail": household_tail, "input_mass_reference_template": { "schema_version": 1, "identity": { @@ -228,6 +242,25 @@ def _measure( } +def household_tail_concentration_columns( + frame, + output_columns=UK_WAS_HOUSEHOLD_WEIGHTED_INTEGRITY_COLUMNS, +) -> tuple[dict[str, np.ndarray], dict[str, np.ndarray]]: + household = frame.table("household") + weights = frame.weights_for("household").values + values: dict[str, np.ndarray] = {} + aligned_weights: dict[str, np.ndarray] = {} + for column in output_columns: + if column not in household.columns: + continue + values[column] = np.asarray( + household[column], + dtype=np.float64, + ) + aligned_weights[column] = np.asarray(weights, dtype=np.float64) + return values, aligned_weights + + def main() -> int: args = _parse_args() try: diff --git a/tools/verify_uk_identity_stability.py b/tools/verify_uk_identity_stability.py index 7d78134d4..477f87070 100644 --- a/tools/verify_uk_identity_stability.py +++ b/tools/verify_uk_identity_stability.py @@ -33,6 +33,13 @@ ) from microcosm.build.uk_runtime.national_build import load_uk_national_frame from microcosm.build.uk_runtime.national_frame import uk_time_period +from microcosm.build.uk_runtime.regional_uprating import ( + load_regional_land_values_resource, + uprate_household_property_by_region, +) +from microcosm.build.uk_runtime.was_wealth import ( + allocate_student_loan_balance_to_people, +) def identity_stability_receipt( @@ -166,33 +173,198 @@ def recompute(person_t, benunit_t, household_t) -> dict[str, pd.DataFrame]: } +def e5_identity_receipt( + frame, + *, + regional_resource: Mapping[str, object] | None = None, + permutation_seed: int, +) -> dict[str, object]: + """Receipt E5 deterministic layers under row permutation by entity id.""" + + resource = regional_resource or load_regional_land_values_resource() + + def recompute(person_t, benunit_t, household_t) -> dict[str, pd.DataFrame]: + del benunit_t + household = household_t.copy() + person = pd.DataFrame(index=person_t["person_id"].to_numpy()) + household_out = pd.DataFrame(index=household_t["household_id"].to_numpy()) + if {"corporate_wealth_excl_isa", "stocks_and_shares_isa"} <= set( + household.columns + ): + household_out["corporate_wealth"] = household[ + "corporate_wealth_excl_isa" + ].to_numpy(dtype=float) + household["stocks_and_shares_isa"].to_numpy( + dtype=float + ) + household["corporate_wealth"] = household_out["corporate_wealth"].to_numpy() + if {"region", "main_residence_value", "property_wealth"} <= set( + household.columns + ): + uprated = uprate_household_property_by_region(household, resource) + household_out["main_residence_value"] = uprated[ + "main_residence_value" + ].to_numpy() + household_out["property_wealth"] = uprated["property_wealth"].to_numpy() + if "student_loan_balance" in household.columns: + person["student_loan_balance"] = allocate_student_loan_balance_to_people( + household_balances=household["student_loan_balance"], + household_ids=household["household_id"], + person=person_t, + ) + elif {"student_loan_balance", "person_household_id"} <= set(person_t.columns): + # The stage consumed the household-level balance; the allocation + # conserves mass, so the household balance is reconstructible as + # the per-household sum of the person column, and the waterfall + # can be re-run from it. Sums and proportional splits are float + # arithmetic, so this layer is compared at fp tolerance rather + # than bitwise (the math, not the summation order, is the + # contract). + canonical = person_t.sort_values("person_id") + reconstructed = ( + pd.Series( + canonical["student_loan_balance"].to_numpy(dtype=float), + index=canonical["person_household_id"].to_numpy(), + ) + .groupby(level=0) + .sum() + ) + balances = ( + reconstructed.reindex(household_t["household_id"].to_numpy()) + .fillna(0.0) + .to_numpy() + ) + person["student_loan_balance"] = allocate_student_loan_balance_to_people( + household_balances=pd.Series(balances), + household_ids=household_t["household_id"], + person=person_t, + ) + return {"household": household_out, "person": person} + + person = frame.table("person") + benunit = frame.table("benunit") + household = frame.table("household") + # Scope to the FRS spine rows: the SPI channel stages stack synthetic + # households AFTER the E5 stages ran (with their own channel-imputed + # property values), so the deterministic-layer identity claims apply + # only to the population the wealth and uprating stages actually saw. + # The stacked rows are E7's receipt surface, not E5's. + if "household_is_spi_synthetic" in household.columns: + spine_mask = ~household["household_is_spi_synthetic"].astype(bool) + household = household.loc[spine_mask].reset_index(drop=True) + spine_household_ids = set(household["household_id"].tolist()) + person = person.loc[ + person["person_household_id"].isin(spine_household_ids) + ].reset_index(drop=True) + original = recompute(person, benunit, household) + rng = np.random.default_rng(permutation_seed) + permuted = recompute( + person.iloc[rng.permutation(len(person))].reset_index(drop=True), + benunit.iloc[rng.permutation(len(benunit))].reset_index(drop=True), + household.iloc[rng.permutation(len(household))].reset_index(drop=True), + ) + # Permutation comparison is bitwise for the uprating layer (its factor + # is computed order-independently). The stored-column cross-check re- + # applies uprating to already-uprated values: the fixed point holds + # exactly only in exact arithmetic, so one rounding generation of + # tolerance applies there, as it does to the waterfall re-allocation + # (sums and proportional splits are float arithmetic). + tolerances = {("person", "student_loan_balance"): (1e-12, 1e-6)} + stored_tolerances = { + ("household", "main_residence_value"): (1e-12, 1e-6), + ("household", "property_wealth"): (1e-12, 1e-6), + ("person", "student_loan_balance"): (1e-12, 1e-6), + } + mismatches: dict[str, list[str]] = {} + stored_mismatches: dict[str, list[str]] = {} + stored_tables = {"household": household, "person": person} + for entity, values in original.items(): + for column in values.columns: + rtol, atol = tolerances.get((entity, column), (0.0, 0.0)) + left = values[column] + right = permuted[entity][column].reindex(left.index) + if not np.allclose( + left.to_numpy(dtype=float), + right.to_numpy(dtype=float), + rtol=rtol, + atol=atol, + ): + mismatches.setdefault(entity, []).append(column) + stored_table = stored_tables[entity] + if column in stored_table.columns: + stored_rtol, stored_atol = stored_tolerances.get( + (entity, column), (rtol, atol) + ) + if not np.allclose( + left.to_numpy(dtype=float), + stored_table[column].to_numpy(dtype=float), + rtol=stored_rtol, + atol=stored_atol, + ): + stored_mismatches.setdefault(entity, []).append(column) + return { + "check": "uk_e5_identity_stability", + "permutation_seed": permutation_seed, + "identical_under_permutation": not mismatches, + "permutation_mismatches": mismatches, + "matches_stored_columns": not stored_mismatches, + "stored_column_mismatches": stored_mismatches, + "tolerance_policy": ( + "permutation: bitwise for the regional-uprating rewrite " + "(order-independent factor), rtol 1e-12 / atol 1e-6 GBP for the " + "waterfall re-allocation; stored-column cross-check: rtol 1e-12 " + "/ atol 1e-6 GBP for both layers (re-applying uprating to the " + "uprated fixed point and re-summing allocations each cost one " + "float rounding generation); corporate_wealth fold not " + "reconstructible from the artifact (components consumed) - " + "unit-tested only" + ), + "columns_by_entity": { + entity: list(values.columns) for entity, values in original.items() + }, + "qrf_draw_columns_scope": ( + "excluded: seeded-stream QRF draws are covered by twin-build " + "determinism rather than identity-keyed row permutation" + ), + } + + def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--input-h5", type=Path, required=True) parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--check", choices=("e4", "e5"), default="e4") parser.add_argument("--permutation-seed", type=int, default=123) args = parser.parse_args() - from microcosm.build.uk_runtime.take_up_contract import load_uk_take_up_contract - from microcosm.frame.adapters.policyengine_uk import PolicyEngineUKEngine - frame, _provenance = load_uk_national_frame(args.input_h5) - engine = PolicyEngineUKEngine() - lha_category = engine.materialize(frame, ("LHA_category",), uk_time_period(frame))[ - "LHA_category" - ] - receipt = e4_identity_receipt( - frame, - contract=load_uk_take_up_contract(), - count_resource=load_brma_count_resource(), - lha_category=lha_category, - permutation_seed=args.permutation_seed, - ) + if args.check == "e4": + from microcosm.build.uk_runtime.take_up_contract import load_uk_take_up_contract + from microcosm.frame.adapters.policyengine_uk import PolicyEngineUKEngine + + engine = PolicyEngineUKEngine() + lha_category = engine.materialize( + frame, ("LHA_category",), uk_time_period(frame) + )["LHA_category"] + receipt = e4_identity_receipt( + frame, + contract=load_uk_take_up_contract(), + count_resource=load_brma_count_resource(), + lha_category=lha_category, + permutation_seed=args.permutation_seed, + ) + ok = bool( + receipt["identical_under_permutation"] and receipt["matches_stored_columns"] + ) + else: + receipt = e5_identity_receipt( + frame, + permutation_seed=args.permutation_seed, + ) + ok = bool( + receipt["identical_under_permutation"] and receipt["matches_stored_columns"] + ) receipt["input_h5"] = str(args.input_h5) args.output.write_text(json.dumps(receipt, indent=2, sort_keys=True) + "\n") - ok = bool( - receipt["identical_under_permutation"] and receipt["matches_stored_columns"] - ) print( "identity stability:", "PASS" if ok else f"FAIL ({args.output})",