diff --git a/chronicle/source_package.py b/chronicle/source_package.py index 815f7d16..605122a8 100644 --- a/chronicle/source_package.py +++ b/chronicle/source_package.py @@ -86,7 +86,9 @@ "hmrc-spi-income-bands-2023-24": Path("hmrc/spi_income_bands_2023_24"), "hmrc-cgt-statistics-2025": Path("hmrc/cgt_statistics_2025"), "hmrc-spi-income-by-area-2023-24": Path("hmrc/spi_income_by_area_2023_24"), - "hmrc-salary-sacrifice-relief-2024-25": Path("hmrc/salary_sacrifice_relief_2024_25"), + "hmrc-salary-sacrifice-relief-2024-25": Path( + "hmrc/salary_sacrifice_relief_2024_25" + ), "hmrc-salary-sacrifice-reform-2029-headcounts": Path( "hmrc/salary_sacrifice_reform_2029_headcounts" ), @@ -100,9 +102,7 @@ "mhclg/ehs_weekly_housing_costs_2023_24" ), "statbel-population-structure-2026": Path("statbel/population_structure_2026"), - "statbel-fiscal-income-2023-nis-2025": Path( - "statbel/fiscal_income_2023_nis_2025" - ), + "statbel-fiscal-income-2023-nis-2025": Path("statbel/fiscal_income_2023_nis_2025"), "spf-finances-pit-2023": Path("spf_finances/pit_2023"), "onss-contributions-2024": Path("onss/contributions_2024"), "onem-rva-unemployment-2024": Path("onem_rva/unemployment_2024"), @@ -112,18 +112,17 @@ "jrc-euromod-be-baseline-statistics-2025": Path( "jrc/euromod_be_baseline_statistics_2025" ), - "sfpd-legal-pension-caseload-2025": Path( - "sfpd/legal_pension_caseload_2025" - ), - "opgroeien-groeipakket-caseload-2025": Path( - "opgroeien/groeipakket_caseload_2025" - ), + "sfpd-legal-pension-caseload-2025": Path("sfpd/legal_pension_caseload_2025"), + "opgroeien-groeipakket-caseload-2025": Path("opgroeien/groeipakket_caseload_2025"), "bfp-economic-outlook-2026-06": Path("bfp/economic_outlook_2026_06"), "dft-nts-vehicle-ownership-2024": Path("dft/nts_vehicle_ownership_2024"), "dwp-benefit-cap-november-2025": Path("dwp/benefit_cap_november_2025"), "dwp-benefit-statistics-february-2026": Path( "dwp/benefit_statistics_february_2026" ), + "dwp-uc-deductions-march-2025-february-2026": Path( + "dwp/uc_deductions_march_2025_february_2026" + ), "dwp-pip-daily-living-foi-2025": Path("dwp/pip_daily_living_foi_2025"), "dwp-uc-households-by-constituency-may-2025": Path( "dwp/uc_households_by_constituency_may_2025" @@ -258,9 +257,7 @@ "kff-marketplace-effectuated-enrollment": Path( "kff/marketplace_effectuated_enrollment" ), - "ons-census2021-ts041-households-lad": Path( - "ons/census2021_ts041_households_lad" - ), + "ons-census2021-ts041-households-lad": Path("ons/census2021_ts041_households_lad"), "ons-census2021-ts041-households-pcon24": Path( "ons/census2021_ts041_households_pcon24" ), @@ -269,28 +266,20 @@ "ons-mye-2024-uk": Path("ons/mye_2024_uk"), "ons-lad-population-by-age-2024": Path("ons/lad_population_by_age_2024"), "ons-pcon24-population-by-age-2024": Path("ons/pcon24_population_by_age_2024"), - "ons-small-area-income-msoa-fye2023": Path( - "ons/small_area_income_msoa_fye2023" - ), + "ons-small-area-income-msoa-fye2023": Path("ons/small_area_income_msoa_fye2023"), "ons-census2021-ts054-tenure-lad": Path("ons/census2021_ts054_tenure_lad"), "ons-subnational-dwellings-by-tenure-2024": Path( "ons/subnational_dwellings_by_tenure_2024" ), "ons-pipr-rents-by-area-june-2026": Path("ons/pipr_rents_by_area_june_2026"), - "nrs-census2022-households-ukpc24": Path( - "nrs/census2022_households_ukpc24" - ), + "nrs-census2022-households-ukpc24": Path("nrs/census2022_households_ukpc24"), "nrs-pcon24-population-by-age-2024": Path("nrs/pcon24_population_by_age_2024"), "nrs-census2022-uv404-tenure-council-area": Path( "nrs/census2022_uv404_tenure_council_area" ), "nisra-census2021-households-lgd": Path("nisra/census2021_households_lgd"), - "nisra-census2021-households-pcon24": Path( - "nisra/census2021_households_pcon24" - ), - "nisra-pcon24-population-by-age-2024": Path( - "nisra/pcon24_population_by_age_2024" - ), + "nisra-census2021-households-pcon24": Path("nisra/census2021_households_pcon24"), + "nisra-pcon24-population-by-age-2024": Path("nisra/pcon24_population_by_age_2024"), "nisra-census2021-tenure-lgd": Path("nisra/census2021_tenure_lgd"), "ons-uk-population-projections-2024": Path("ons/npp_2024_uk"), "scotgov-band-d-council-tax-rates-2026-27": Path( @@ -301,13 +290,9 @@ "scotgov-scottish-budget-social-security-assistance-2026": Path( "scotgov/scottish_budget_social_security_assistance_2026" ), - "welshgov-council-tax-levels-2026-27": Path( - "welshgov/council_tax_levels_2026_27" - ), + "welshgov-council-tax-levels-2026-27": Path("welshgov/council_tax_levels_2026_27"), "voa-council-tax-bands-2025": Path("voa/council_tax_bands_2025"), - "voa-council-tax-stock-by-lad-2025": Path( - "voa/council_tax_stock_by_lad_2025" - ), + "voa-council-tax-stock-by-lad-2025": Path("voa/council_tax_stock_by_lad_2025"), "obr-efo-receipts-march-2026": Path("obr/efo_receipts_march_2026"), "obr-efo-expenditure-march-2026": Path("obr/efo_expenditure_march_2026"), "ons-national-balance-sheet-land-2025": Path( @@ -334,7 +319,9 @@ SOURCE_PACKAGE_FILENAME = "source_package.yaml" # Repository ``packages/`` directory: the source-package authoring surface a # complete bundle must cover (see PolicyEngine/chronicle#78). -SOURCE_PACKAGE_ROOT = Path(__file__).resolve().parents[1] / SOURCE_PACKAGE_RESOURCE_PACKAGE +SOURCE_PACKAGE_ROOT = ( + Path(__file__).resolve().parents[1] / SOURCE_PACKAGE_RESOURCE_PACKAGE +) EXCEL_COLUMN_RE = re.compile(r"^[A-Z]+$") @@ -1825,6 +1812,12 @@ def _measure_from_mapping( unit=_required(payload, "unit", "measure"), aggregation=_required(payload, "aggregation", "measure"), value_scale=_render_value(payload.get("value_scale", 1), year=year), + divisor_column=payload.get("divisor_column"), + round_to=( + _render_value(payload["round_to"], year=year) + if "round_to" in payload + else None + ), source_column_id=( str(_render_value(payload["source_column_id"], year=year)) if payload.get("source_column_id") is not None @@ -2041,9 +2034,7 @@ def _provenance_fields_from_mapping( survey_instrument = payload.get("survey_instrument") if provenance_class == "survey_aggregate": if not has_survey_instrument: - raise KeyError( - "Missing required record_set field: survey_instrument" - ) + raise KeyError("Missing required record_set field: survey_instrument") if type(survey_instrument) is not str or not survey_instrument.strip(): raise TypeError( "record_set survey_instrument must be a non-empty string for " @@ -2095,9 +2086,7 @@ def _period_coverage_from_mapping( } unknown = sorted(set(payload) - allowed_keys) if unknown: - raise ValueError( - f"record_set period_coverage has unknown keys: {unknown}." - ) + raise ValueError(f"record_set period_coverage has unknown keys: {unknown}.") return PeriodCoverage( **{ key: _optional_rendered_string(payload.get(key), year=year) diff --git a/chronicle/sources/specs.py b/chronicle/sources/specs.py index c0d75ca4..41e2382c 100644 --- a/chronicle/sources/specs.py +++ b/chronicle/sources/specs.py @@ -92,6 +92,8 @@ class SourceRecordSpec: filters: dict[str, Scalar] = field(default_factory=dict) constraints: tuple[AggregateConstraint, ...] = () value_scale: int | float = 1 + divisor_selector: CellSelectorSpec | None = None + round_to: int | float | None = None source_concept: str | None = None concept_relation: str | None = None concept_authority: str | None = None @@ -162,6 +164,8 @@ class SourceRecordSetMeasure: unit: str aggregation: str value_scale: int | float = 1 + divisor_column: str | None = None + round_to: int | float | None = None source_column_id: str | None = None expected_cell_type: str = "number" expected_column_header_row: int | None = None @@ -298,6 +302,17 @@ def compile_source_record_set_specs( provenance_class=spec.provenance_class, survey_instrument=spec.survey_instrument, value_scale=measure.value_scale * row.value_scale, + divisor_selector=( + CellSelectorSpec( + selector_id=f"{source_record_id}.divisor_selector", + sheet_name=spec.sheet_name, + address=f"{measure.divisor_column}{row.row_number}", + expected_cell_type="number", + ) + if measure.divisor_column is not None + else None + ), + round_to=measure.round_to, assertion=spec.assertion, period_coverage=spec.period_coverage, source_concept=measure.source_concept, @@ -343,6 +358,11 @@ def source_regions_from_record_set_spec( columns = [ 1, *(_excel_column_number(measure.column) for measure in spec.measures), + *( + _excel_column_number(measure.divisor_column) + for measure in spec.measures + if measure.divisor_column is not None + ), *( _excel_column_number(row.column) for row in spec.rows @@ -435,8 +455,25 @@ def resolve_source_record( cells_by_sheet_address=cells_by_sheet_address, ) value = _scale_value(_sum_cell_values(value_cells), spec.value_scale) + divisor_cells: list[SourceCell] = [] + if spec.divisor_selector is not None: + divisor_cells = _resolve_value_cells( + cells, + spec.divisor_selector, + cells_by_sheet_address=cells_by_sheet_address, + ) + divisor = _sum_cell_values(divisor_cells) + if not isinstance(divisor, (int, float)) or divisor == 0: + raise ValueError( + f"Source record {spec.source_record_id!r} has invalid divisor " + f"{divisor!r}." + ) + value = value / divisor + if spec.round_to is not None: + value = round(value / spec.round_to) * spec.round_to lineage_cells = [ *value_cells, + *divisor_cells, *_selector_lineage_guard_cells( cells, spec.selector, @@ -805,6 +842,10 @@ def _record_set_spec_hash(spec: SourceRecordSetSpec) -> str: if row.get(key) is None: row.pop(key, None) for measure in payload["measures"]: + if measure.get("divisor_column") is None: + measure.pop("divisor_column", None) + if measure.get("round_to") is None: + measure.pop("round_to", None) if measure.get("expected_column_header") is None: measure.pop("expected_column_header", None) if measure.get("expected_column_header_row") is None: diff --git a/db/data/dwp/uc_deductions_march_2025_february_2026/manifest.yaml b/db/data/dwp/uc_deductions_march_2025_february_2026/manifest.yaml new file mode 100644 index 00000000..e199c1c5 --- /dev/null +++ b/db/data/dwp/uc_deductions_march_2025_february_2026/manifest.yaml @@ -0,0 +1,18 @@ +source_id: dwp +package_id: dwp-uc-deductions-march-2025-february-2026 +dataset: dwp_uc_deductions_march_2025_february_2026 +source_page: https://www.gov.uk/government/statistics/universal-credit-quarterly-statistics-29-april-2013-to-12-february-2026 +table: 'Universal Credit deductions statistics March 2025 to February 2026, supplementary data tables' +files: + 2026: + filename: universal-credit-deductions-march-2025-to-february-2026.ods + source_url: https://assets.publishing.service.gov.uk/media/69fb3ea22a6137e93226b7ce/universal-credit-deductions-march-2025-to-february-2026.ods + sha256: 307ec8fa49a1f1e23db3c59f1282e16609de03444e501824e204b4151e2e5c9b + size_bytes: 148671 + fetched_at: '2026-08-19T00:00:00+00:00' + storage: + r2: + provider: r2 + bucket: ledger-raw + key: raw/dwp/dwp-uc-deductions-march-2025-february-2026/2026/307ec8fa49a1f1e23db3c59f1282e16609de03444e501824e204b4151e2e5c9b/universal-credit-deductions-march-2025-to-february-2026.ods + uri: r2://ledger-raw/raw/dwp/dwp-uc-deductions-march-2025-february-2026/2026/307ec8fa49a1f1e23db3c59f1282e16609de03444e501824e204b4151e2e5c9b/universal-credit-deductions-march-2025-to-february-2026.ods diff --git a/db/data/dwp/uc_deductions_march_2025_february_2026/universal-credit-deductions-march-2025-to-february-2026.ods b/db/data/dwp/uc_deductions_march_2025_february_2026/universal-credit-deductions-march-2025-to-february-2026.ods new file mode 100644 index 00000000..d8357700 Binary files /dev/null and b/db/data/dwp/uc_deductions_march_2025_february_2026/universal-credit-deductions-march-2025-to-february-2026.ods differ diff --git a/packages/dwp/uc_deductions_march_2025_february_2026/source_package.yaml b/packages/dwp/uc_deductions_march_2025_february_2026/source_package.yaml new file mode 100644 index 00000000..7038da0e --- /dev/null +++ b/packages/dwp/uc_deductions_march_2025_february_2026/source_package.yaml @@ -0,0 +1,240 @@ +# DWP calls the Universal Credit unit of assessment a household. It is a UC +# benefit unit, not an ONS household; see policyengine-uk-data#457. Table 1 +# publishes rounded counts with a deduction and rounded shares of all UC units. +# The total-caseload facts divide those two publisher cells and round the result +# to 10,000. This is a deterministic within-row derivation, not a cross-source +# estimate. Calendar-year and plateau summaries remain consumer aggregations. +schema_version: ledger.source_package.v1 +package_id: dwp-uc-deductions-march-2025-february-2026 +label: 'DWP Universal Credit benefit units and deductions, March 2025 to February 2026' +artifact: + source_name: dwp + source_table: 'Universal Credit deductions statistics March 2025 to February 2026, Table 1' + resource_package: db + resource_directory: data/dwp/uc_deductions_march_2025_february_2026 + manifest: manifest.yaml + vintage: uc_deductions_2026_05_12 + extracted_at: '2026-08-19' + extraction_method: ODS cell parse with within-row count/share derivation + parser: ods_used_range + artifact_year: 2026 +record_sets: +- &published + record_set_id: dwp.uc_deductions.month2025_03.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_03.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_03.published + sheet_name: Table_1 + period_type: month + period: 2025-03 + geography_id: K03000001 + geography_level: country + geography_name: Great Britain + geography_vintage: current + entity: benefit_unit + entity_role: universal_credit_unit_of_assessment + domain: universal_credit + groupby_dimension: dwp.uc_deductions_month + rows: [&march {value_id: month, label: March 2025, ordinal: 0, row_number: 6, expected_row_header_column: B, expected_row_header: March 2025, table_record_kind: total}] + measures: &published_measures + - measure_id: units_with_deduction + label: UC benefit units with a deduction + ordinal: 0 + column: C + expected_column_header_row: 5 + expected_column_header: Number of Households with a Deduction + concept: dwp.uc_benefit_units_with_deduction + source_concept: dwp.uc_households_with_deduction + concept_relation: source_label + concept_authority: dwp + concept_evidence_url: https://www.gov.uk/government/statistics/universal-credit-quarterly-statistics-29-april-2013-to-12-february-2026/universal-credit-deductions-statistics-march-2025-to-february-2026 + concept_evidence_notes: DWP household means the Universal Credit unit of assessment, not an ONS household (policyengine-uk-data#457). + unit: count + aggregation: sum + expected_cell_type: number + - measure_id: deduction_share + label: Share of UC benefit units with a deduction + ordinal: 1 + column: D + expected_column_header_row: 5 + expected_column_header: Percentage of Households with a Deduction + concept: dwp.uc_benefit_units_with_deduction_share + source_concept: dwp.uc_households_with_deduction_share + concept_relation: source_label + concept_authority: dwp + concept_evidence_url: https://www.gov.uk/government/statistics/universal-credit-quarterly-statistics-29-april-2013-to-12-february-2026/universal-credit-deductions-statistics-march-2025-to-february-2026 + concept_evidence_notes: Published decimal share; DWP household means the Universal Credit unit of assessment. + unit: share + aggregation: share + expected_cell_type: number +- <<: *published + record_set_id: dwp.uc_deductions.month2025_04.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_04.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_04.published + period: 2025-04 + rows: [&april {value_id: month, label: April 2025, ordinal: 0, row_number: 7, expected_row_header_column: B, expected_row_header: April 2025, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2025_05.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_05.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_05.published + period: 2025-05 + rows: [&may {value_id: month, label: May 2025, ordinal: 0, row_number: 8, expected_row_header_column: B, expected_row_header: May 2025, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2025_06.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_06.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_06.published + period: 2025-06 + rows: [&june {value_id: month, label: June 2025, ordinal: 0, row_number: 9, expected_row_header_column: B, expected_row_header: June 2025, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2025_07.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_07.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_07.published + period: 2025-07 + rows: [&july {value_id: month, label: July 2025, ordinal: 0, row_number: 10, expected_row_header_column: B, expected_row_header: July 2025, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2025_08.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_08.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_08.published + period: 2025-08 + rows: [&august {value_id: month, label: August 2025, ordinal: 0, row_number: 11, expected_row_header_column: B, expected_row_header: August 2025, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2025_09.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_09.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_09.published + period: 2025-09 + rows: [&september {value_id: month, label: September 2025, ordinal: 0, row_number: 12, expected_row_header_column: B, expected_row_header: September 2025, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2025_10.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_10.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_10.published + period: 2025-10 + rows: [&october {value_id: month, label: October 2025, ordinal: 0, row_number: 13, expected_row_header_column: B, expected_row_header: October 2025, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2025_11.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_11.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_11.published + period: 2025-11 + rows: [&november {value_id: month, label: November 2025, ordinal: 0, row_number: 14, expected_row_header_column: B, expected_row_header: November 2025, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2025_12.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_12.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_12.published + period: 2025-12 + rows: [&december {value_id: month, label: December 2025, ordinal: 0, row_number: 15, expected_row_header_column: B, expected_row_header: December 2025, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2026_01.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2026_01.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2026_01.published + period: 2026-01 + rows: [&january {value_id: month, label: January 2026, ordinal: 0, row_number: 16, expected_row_header_column: B, expected_row_header: January 2026, table_record_kind: total}] +- <<: *published + record_set_id: dwp.uc_deductions.month2026_02.published + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2026_02.published.v1 + source_record_id_prefix: dwp.uc_deductions.month2026_02.published + period: 2026-02 + rows: [&february {value_id: month, label: February 2026, ordinal: 0, row_number: 17, expected_row_header_column: B, expected_row_header: February 2026, table_record_kind: total}] +- &derived + <<: *published + record_set_id: dwp.uc_deductions.month2025_04.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_04.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_04.total_units + period: 2025-04 + rows: [*april] + measures: &derived_measures + - measure_id: total_units + label: Total UC benefit units + ordinal: 0 + column: C + divisor_column: D + round_to: 10000 + concept: dwp.uc_benefit_units + source_concept: dwp.uc_households + concept_relation: source_label + concept_authority: dwp + concept_evidence_url: https://www.gov.uk/government/statistics/universal-credit-quarterly-statistics-29-april-2013-to-12-february-2026/universal-credit-deductions-statistics-march-2025-to-february-2026 + concept_evidence_notes: 'Derived within Table 1 as households with a deduction divided by the published share, rounded to 10,000. DWP household means a UC benefit unit, not an ONS household; see policyengine-uk-data#457.' + unit: count + aggregation: sum + expected_cell_type: number +- <<: *derived + record_set_id: dwp.uc_deductions.month2025_05.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_05.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_05.total_units + period: 2025-05 + rows: [*may] +- <<: *derived + record_set_id: dwp.uc_deductions.month2025_06.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_06.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_06.total_units + period: 2025-06 + rows: [*june] +- <<: *derived + record_set_id: dwp.uc_deductions.month2025_07.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_07.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_07.total_units + period: 2025-07 + rows: [*july] +- <<: *derived + record_set_id: dwp.uc_deductions.month2025_08.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_08.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_08.total_units + period: 2025-08 + rows: [*august] +- <<: *derived + record_set_id: dwp.uc_deductions.month2025_09.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_09.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_09.total_units + period: 2025-09 + rows: [*september] +- <<: *derived + record_set_id: dwp.uc_deductions.month2025_10.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_10.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_10.total_units + period: 2025-10 + rows: [*october] +- <<: *derived + record_set_id: dwp.uc_deductions.month2025_11.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_11.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_11.total_units + period: 2025-11 + rows: [*november] +- <<: *derived + record_set_id: dwp.uc_deductions.month2025_12.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2025_12.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2025_12.total_units + period: 2025-12 + rows: [*december] +- <<: *derived + record_set_id: dwp.uc_deductions.month2026_01.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2026_01.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2026_01.total_units + period: 2026-01 + rows: [*january] +- <<: *derived + record_set_id: dwp.uc_deductions.month2026_02.total_units + provenance_class: administrative + record_set_spec_id: dwp.uc_deductions.month2026_02.total_units.v1 + source_record_id_prefix: dwp.uc_deductions.month2026_02.total_units + period: 2026-02 + rows: [*february] diff --git a/tests/test_chronicle_bundle.py b/tests/test_chronicle_bundle.py index 375678fb..34626840 100644 --- a/tests/test_chronicle_bundle.py +++ b/tests/test_chronicle_bundle.py @@ -25,18 +25,18 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): assert summary["valid"] assert summary["counts"] == { "aggregate_duplicate_key_count": 0, - "entity_count": 10, + "entity_count": 11, "error_count": 0, - "fact_count": 155396, + "fact_count": 155431, "geography_count": 12536, - "period_count": 147, + "period_count": 151, "semantic_duplicate_key_count": 12, "skipped_source_count": 10, "source_count": 41, - "source_package_count": 127, + "source_package_count": 128, "warning_count": 1, } - assert len(rows) == 155396 + assert len(rows) == 155431 assert {row["provenance_class"] for row in rows} <= { "administrative", "census", @@ -54,7 +54,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): ) assert rows[0]["aggregate_fact_key"].startswith("ledger.aggregate_fact.v2:") assert rows[0]["semantic_fact_key"].startswith("ledger.semantic_fact.v2:") - assert source_packages["source_package_count"] == 127 + assert source_packages["source_package_count"] == 128 assert source_packages["skipped_source_count"] == 10 assert sorted(item["source"] for item in source_packages["skipped_sources"]) == [ "census-acs-s0101-congressional-district-age-2024", @@ -68,7 +68,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "jct-obbba-revenue-estimates-2025", "jct-tax-expenditures-2024", ] - assert coverage["fact_count"] == 155396 + assert coverage["fact_count"] == 155431 assert coverage["counts"]["by_source"] == { "bea": 445, "bfp_economic_outlook": 5, @@ -81,7 +81,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "cms_medicare": 1, "cms_nhe": 3, "dft": 81, - "dwp": 6327, + "dwp": 6362, "eurostat": 108, "federal_reserve": 1, "hhs_acf_liheap": 2, @@ -113,11 +113,17 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "welshgov": 198, } table_counts = coverage["counts"]["by_source_table"] - assert len(table_counts) == 122 + assert len(table_counts) == 123 assert table_counts["usda_snap:SNAP FY2025 Monthly State Participation"] == 636 assert table_counts["irs_soi:Congressional District Data 2022"] == 26880 assert table_counts["irs_soi:IRS SOI County Data 2022"] == 6286 assert table_counts["census_pep:Vintage 2024 County Population Totals"] == 3144 + assert ( + table_counts[ + "dwp:Universal Credit deductions statistics March 2025 to February 2026, Table 1" + ] + == 35 + ) assert ( table_counts[ "cms_nhe:Employer-Sponsored Private Health Insurance: " @@ -365,14 +371,18 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "month:2024-12": 376, "month:2025-01": 108, "month:2025-02": 106, - "month:2025-03": 226, - "month:2025-04": 92, - "month:2025-05": 6214, - "month:2025-08": 4, - "month:2025-09": 9, - "month:2025-11": 15, - "month:2025-12": 256, - "month:2026-02": 4, + "month:2025-03": 228, + "month:2025-04": 95, + "month:2025-05": 6217, + "month:2025-06": 3, + "month:2025-07": 3, + "month:2025-08": 7, + "month:2025-09": 12, + "month:2025-10": 3, + "month:2025-11": 18, + "month:2025-12": 259, + "month:2026-01": 3, + "month:2026-02": 7, "month:2026-06": 348, "tax_year:1987": 9, "tax_year:1988": 9, @@ -426,9 +436,10 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): coverage["counts"]["by_geography"]["congressional_district:5001700US0601"] == 56 ) assert coverage["counts"]["by_geography"]["country:K02000001"] == 4189 - assert coverage["counts"]["by_geography"]["country:K03000001"] == 277 + assert coverage["counts"]["by_geography"]["country:K03000001"] == 312 assert len(coverage["counts"]["by_geography"]) == 12536 assert coverage["counts"]["by_entity"] == { + "benefit_unit": 35, "dwelling": 12708, "family": 107, "firm": 1439, diff --git a/tests/test_chronicle_source_package.py b/tests/test_chronicle_source_package.py index 01379730..7fec3999 100644 --- a/tests/test_chronicle_source_package.py +++ b/tests/test_chronicle_source_package.py @@ -208,12 +208,77 @@ def test_every_source_package_record_set_declares_provenance_class(): assert len(load_source_package(path).record_sets) == len(payload["record_sets"]) assert not missing, "Record sets missing provenance_class:\n" + "\n".join(missing) - assert not malformed, "Malformed provenance declarations:\n" + "\n".join( - malformed - ) + assert not malformed, "Malformed provenance declarations:\n" + "\n".join(malformed) assert record_set_line_count == provenance_line_count == record_set_count +def test_dwp_uc_deductions_package_preserves_rows_and_derives_uc_units(): + package = load_source_package("dwp-uc-deductions-march-2025-february-2026") + facts = package.build_facts(2026) + + published_counts = [ + fact.value + for fact in facts + if fact.measure.concept == "dwp.uc_benefit_units_with_deduction" + ] + published_shares = [ + fact.value + for fact in facts + if fact.measure.concept == "dwp.uc_benefit_units_with_deduction_share" + ] + derived_units = { + fact.period.value: fact.value + for fact in facts + if fact.measure.concept == "dwp.uc_benefit_units" + } + + assert published_counts == [ + 3_000_000, + 3_000_000, + 3_100_000, + 3_100_000, + 3_100_000, + 3_100_000, + 3_200_000, + 3_200_000, + 3_200_000, + 3_300_000, + 3_300_000, + 3_300_000, + ] + assert published_shares == pytest.approx([0.47] * 6 + [0.46] * 6) + assert derived_units == { + "2025-04": 6_380_000, + "2025-05": 6_600_000, + "2025-06": 6_600_000, + "2025-07": 6_600_000, + "2025-08": 6_600_000, + "2025-09": 6_960_000, + "2025-10": 6_960_000, + "2025-11": 6_960_000, + "2025-12": 7_170_000, + "2026-01": 7_170_000, + "2026-02": 7_170_000, + } + assert all(fact.entity.name == "benefit_unit" for fact in facts) + assert all( + len(fact.source_cell_keys) == 2 + for fact in facts + if fact.measure.concept == "dwp.uc_benefit_units" + ) + derived_regions = { + region.record_set_id: region + for region in package.build_source_regions(2026) + if region.record_set_id is not None + and region.record_set_id.endswith(".total_units") + } + assert derived_regions[ + "dwp.uc_deductions.month2025_04.total_units" + ].right_column == 4 + assert validate_consumer_fact_contract(facts).valid + assert len(consumer_fact_rows(facts)) == len(facts) + + def test_source_package_alias_compiles_soi_table_1_1_specs(): package = load_source_package("soi-table-1-1") record_set = package.build_source_record_set_specs(2023)[0] @@ -1051,8 +1116,7 @@ def test_cms_nhe_table_24_package_builds_esi_employer_contribution_facts(): assert total_2023.source.raw_r2_uri assert not total_2023.constraints assert [ - (item.variable, item.operator, item.value) - for item in private_2023.constraints + (item.variable, item.operator, item.value) for item in private_2023.constraints ] == [("esi_employer_sector", "==", "private")] @@ -1378,8 +1442,7 @@ def test_jct_obbba_package_builds_no_tax_provision_projection_facts(): assert fact.measure.legal_vintage == f"fiscal_year_{year}" assert fact.filters["provision"] == provision assert [ - (item.variable, item.operator, item.value) - for item in fact.constraints + (item.variable, item.operator, item.value) for item in fact.constraints ] == [("provision", "==", provision)] @@ -1693,8 +1756,7 @@ def test_ssa_ssi_monthly_2024_12_package_builds_federal_payment_age_facts(): assert aged_18_to_64.value == 3_905_779 assert aged_65_or_older.value == 2_382_142 assert ( - under_18.value + aged_18_to_64.value + aged_65_or_older.value - == all_ages.value + under_18.value + aged_18_to_64.value + aged_65_or_older.value == all_ages.value ) assert all_ages.period.type == "month" @@ -1706,12 +1768,10 @@ def test_ssa_ssi_monthly_2024_12_package_builds_federal_payment_age_facts(): assert all_ages.layout.table_record_kind == "total" assert { - (item.variable, item.operator, item.value) - for item in under_18.constraints + (item.variable, item.operator, item.value) for item in under_18.constraints } == {("age", ">=", 0), ("age", "<", 18)} assert { - (item.variable, item.operator, item.value) - for item in aged_18_to_64.constraints + (item.variable, item.operator, item.value) for item in aged_18_to_64.constraints } == {("age", ">=", 18), ("age", "<", 65)} assert { (item.variable, item.operator, item.value)