diff --git a/packages/populace-build/src/populace/build/uk_runtime/__init__.py b/packages/populace-build/src/populace/build/uk_runtime/__init__.py index b9051406..2048c217 100644 --- a/packages/populace-build/src/populace/build/uk_runtime/__init__.py +++ b/packages/populace-build/src/populace/build/uk_runtime/__init__.py @@ -1,5 +1,59 @@ """UK build helpers for Populace-owned local-geography artifacts.""" +from populace.build.uk_runtime.ai_exposure import ( + MEASURE_TO_COLUMN, + exposure_for_major_group, + exposure_for_soc, + load_ai_exposure_table, + load_major_group_ai_exposure_table, +) +from populace.build.uk_runtime.ai_exposure_sources import ( + build_ai_exposure_table, + build_complementarity_theta, + build_dsit_soc2010_to_soc2020, + build_major_group_ai_exposure, + build_us_measure_chain, +) +from populace.build.uk_runtime.ai_shock_runner import ( + GRID_CAPITAL_RETURN_INCREASE, + GRID_DISPLACEMENT_RATES, + GRID_WAGE_UPLIFTS, + AgeBandDelta, + DecileDelta, + PopulationMetrics, + ScenarioResult, + age_band_deltas, + build_scenario_grid, + decile_deltas, + gini, + run_ai_shock_analysis, + run_grid, + write_grid_csv, + write_grid_json, +) +from populace.build.uk_runtime.ai_shock_scenarios import ( + KLEIN_TEESELINK_YOUTH_MULTIPLIER, + ShockScenario, + apply_capital_shock, + apply_employment_shock, + apply_wage_shock, + run_scenario, +) +from populace.build.uk_runtime.ai_shock_scenarios import ( + PRESETS as AI_SHOCK_PRESETS, +) +from populace.build.uk_runtime.exposure_imputation import ( + DEFAULT_EXPOSURE_COLUMN, + DEFAULT_PREDICTORS, + LFS_APS_EXPOSURE_DONOR, + SOC_MAJOR_GROUP_COLUMN, + UK_EXPOSURE_IMPUTATION_STAGE_NAME, + attach_exposure, + exposure_from_major_group, + exposure_imputation_stage, + fit_exposure_imputer, + impute_exposure, +) from populace.build.uk_runtime.firm_generation import ( EMPLOYMENT_BANDS, HMRC_BAND_COLUMNS, @@ -32,6 +86,17 @@ validate_uk_firm_population, write_uk_firm_population, ) +from populace.build.uk_runtime.frs_occupation import ( + FRS_ADULT_KEY_COLUMNS, + FRS_ADULT_SOC_COLUMN, + FRS_OCCUPATION_DONOR, + UK_FRS_OCCUPATION_STAGE_NAME, + attach_soc_major_group, + frs_major_group_to_digit, + frs_occupation_stage, + load_frs_adult_table, + soc_major_group_for_persons, +) from populace.build.uk_runtime.geography_ladder import ( GEOGRAPHY_LADDER_ARTIFACT_SHA256_ATTR, GEOGRAPHY_LADDER_VINTAGES_ATTR, @@ -141,6 +206,19 @@ assemble_uk_oa_ladder, join_uk_oa_ladder_layers, ) +from populace.build.uk_runtime.occupation_targets import ( + APS_NOMIS_SOURCE, + ASHE_TABLE14_SOURCE, + PACKAGED_APS_AGE_CSV, + PACKAGED_APS_OCCUPATION_CSV, + PACKAGED_ASHE_TABLE14_CSV, + OccupationTargetBuild, + SkippedTargetRow, + aps_age_band_employment_targets, + aps_occupation_employment_targets, + ashe_occupation_employment_targets, + packaged_occupation_csv_path, +) from populace.build.uk_runtime.rowwise_dataset import ( BENUNIT_ID_COLUMNS, HOUSEHOLD_ID_COLUMNS, @@ -186,15 +264,22 @@ __all__ = [ "AGE_BANDS", - "AREA_TYPE_TO_LEDGER_GEOGRAPHY_LEVEL", + "AI_SHOCK_PRESETS", + "APS_NOMIS_SOURCE", "AREA_TYPES", "AREA_TYPE_TO_CROSSWALK_COLUMN", + "AREA_TYPE_TO_LEDGER_GEOGRAPHY_LEVEL", + "ASHE_TABLE14_SOURCE", + "AgeBandDelta", "AxiomVATRuleEvaluator", "BASE_FRS_SUPPORT_CHANNEL", "BENUNIT_ID_COLUMNS", "COUNTRY_TO_REGION", "CROSSWALK_COLUMNS", + "DEFAULT_EXPOSURE_COLUMN", + "DEFAULT_PREDICTORS", "DEFAULT_SPI_SUPPORT_HOUSEHOLDS", + "DecileDelta", "EMPLOYMENT_BANDS", "ENGLAND_LAD_REGION_URL", "ENGLAND_WALES_OA2021_COUNT", @@ -204,26 +289,40 @@ "EW_OA_LAD23_URL", "EW_OA_POPULATION_URL", "EW_OA_WARD_URL", - "GEOGRAPHY_LADDER_ARTIFACT_SHA256_ATTR", - "GEOGRAPHY_LADDER_VINTAGES_ATTR", - "FRS_REGION_TO_COUNTRY", - "FRS_REGION_TO_REGION_CODE", + "FRS_ADULT_KEY_COLUMNS", + "FRS_ADULT_SOC_COLUMN", + "FRS_OCCUPATION_DONOR", "FRS_ONLY_SPI_FILL_PERSON_COLUMNS", "FRS_ONLY_SPI_FILL_PREDICTOR_COLUMNS", - "HOUSEHOLD_IS_SPI_SYNTHETIC_COLUMN", - "HOUSEHOLD_ID_COLUMNS", + "FRS_REGION_TO_COUNTRY", + "FRS_REGION_TO_REGION_CODE", + "GEOGRAPHY_LADDER_ARTIFACT_SHA256_ATTR", + "GEOGRAPHY_LADDER_VINTAGES_ATTR", + "GRID_CAPITAL_RETURN_INCREASE", + "GRID_DISPLACEMENT_RATES", + "GRID_WAGE_UPLIFTS", "HMRC_BAND_COLUMNS", - "INPUT_FILES", + "HOUSEHOLD_ID_COLUMNS", + "HOUSEHOLD_IS_SPI_SYNTHETIC_COLUMN", "INCOME_VARIABLES", - "LA_EXTRA_METRICS", - "LADDER_OA_COLUMNS", + "INPUT_FILES", + "KLEIN_TEESELINK_YOUTH_MULTIPLIER", "LAD23_ITL_URL", + "LADDER_OA_COLUMNS", + "LA_EXTRA_METRICS", + "LFS_APS_EXPOSURE_DONOR", "LONG_GEOGRAPHY_COLUMNS", "MAX_UNMATCHED_ACTIVE_NI_POSTCODE_SHARE", + "MEASURE_TO_COLUMN", "NI_DZ2021_COUNT", "NI_DZ_GEOJSON_ZIP_URL", "NI_DZ_POPULATION_CSV_URL", + "OccupationTargetBuild", + "PACKAGED_APS_AGE_CSV", + "PACKAGED_APS_OCCUPATION_CSV", + "PACKAGED_ASHE_TABLE14_CSV", "PERSON_ID_COLUMNS", + "PopulationMetrics", "ROWWISE_GEOGRAPHY_COLUMNS", "RowwiseGeographyAssignment", "SCOTLAND_OA2022_COUNT", @@ -231,20 +330,15 @@ "SCOTLAND_OA_DZ_IZ_URL", "SCOTLAND_OA_LAU_ITL_URL", "SCOTLAND_OA_POPULATION_URL", + "SOC_MAJOR_GROUP_COLUMN", "SPI_INCOME_COMPONENT_COLUMNS", "SPI_INCOME_IMPUTATION_COLUMNS", "SPI_SYNTHETIC_SUPPORT_CHANNEL", + "ScenarioResult", + "ShockScenario", + "SkippedTargetRow", "StackedLocalMatrix", "StackedLocalSolveResult", - "UK_ENGLAND_WALES_REGION_CODES", - "UK_GEOGRAPHY_LADDER_COLUMNS", - "UK_LONDON_REGION_CODE", - "UK_OA_LADDER_DERIVED_LAYERS", - "UK_OA_LADDER_KIND", - "UK_OA_LADDER_SCHEMA_VERSION", - "UK_POSTCODE_OA_MAY25_ZIP_URL", - "UK_POSTCODE_PCON_MAY24_ZIP_URL", - "UkOaLadder", "UKFirmCalibrationResult", "UKFirmGenerationConfig", "UKFirmGenerationResult", @@ -255,44 +349,81 @@ "UKLocalCandidateResult", "UKRowwiseDatasetResult", "UKSPISupportResult", + "UK_ENGLAND_WALES_REGION_CODES", + "UK_EXPOSURE_IMPUTATION_STAGE_NAME", + "UK_FRS_OCCUPATION_STAGE_NAME", + "UK_GEOGRAPHY_LADDER_COLUMNS", + "UK_LONDON_REGION_CODE", + "UK_OA_LADDER_DERIVED_LAYERS", + "UK_OA_LADDER_KIND", + "UK_OA_LADDER_SCHEMA_VERSION", + "UK_POSTCODE_OA_MAY25_ZIP_URL", + "UK_POSTCODE_PCON_MAY24_ZIP_URL", "UK_SINGLE_YEAR_TABLES", "UK_SPI_SUPPORT_STAGE_NAME", + "UkOaLadder", "VAT_LIABILITY_BANDS", "VINTAGES", + "age_band_deltas", "align_area_targets", - "area_support_summary", + "apply_capital_shock", + "apply_employment_shock", + "apply_wage_shock", + "aps_age_band_employment_targets", + "aps_occupation_employment_targets", "area_groups_from_codes", + "area_support_summary", + "ashe_occupation_employment_targets", "assemble_uk_oa_ladder", "assign_employment", "assign_household_geography", "assign_uk_geography_ladder", "assign_vat_flags", - "build_firm_target_matrix", - "build_local_candidate", - "build_local_candidate_from_dataset", + "attach_exposure", + "attach_soc_major_group", + "build_ai_exposure_table", + "build_complementarity_theta", "build_complete_uk_geography_crosswalk", + "build_dsit_soc2010_to_soc2020", "build_england_wales_crosswalk", + "build_firm_target_matrix", "build_great_britain_crosswalk", + "build_local_candidate", + "build_local_candidate_from_dataset", + "build_major_group_ai_exposure", "build_metric_tables_from_dataset", "build_northern_ireland_crosswalk", "build_official_uk_geography_crosswalk", + "build_scenario_grid", "build_scotland_crosswalk", "build_stacked_local_matrix", + "build_us_measure_chain", "clone_entity_frame", "clone_uk_dataset_tables_with_rowwise_geography", "clone_uk_dataset_with_rowwise_geography", "compute_household_metrics", "create_uk_spi_support_tables", + "decile_deltas", "employment_band_name", + "exposure_for_major_group", + "exposure_for_soc", + "exposure_from_major_group", + "exposure_imputation_stage", "fill_support_channel_from_source", + "fit_exposure_imputer", + "frs_major_group_to_digit", + "frs_occupation_stage", "generate_base_firms", "generate_input_values", "generate_uk_firm_population", "geography_coverage_summary", + "gini", "hmrc_band_name", "id_multiplier_for_values", + "impute_exposure", "infer_ni_dz_constituencies_from_postcodes", "join_uk_oa_ladder_layers", + "load_ai_exposure_table", "load_england_lad_region_lookup", "load_england_wales_oa_constituencies", "load_england_wales_oa_hierarchy", @@ -300,7 +431,9 @@ "load_england_wales_oa_population", "load_england_wales_oa_ward_lookup", "load_ew_oa_lad23_lookup", + "load_frs_adult_table", "load_lad_itl_lookup", + "load_major_group_ai_exposure_table", "load_metric_tables", "load_ni_dz_hierarchy", "load_ni_dz_population", @@ -308,21 +441,26 @@ "load_scotland_oa_dz_iz_lookup", "load_scotland_oa_lau_lookup", "load_scotland_oa_population", + "load_uk_dataset", "load_uk_oa_ladder", "load_uk_postcode_constituency_lookup", "load_uk_postcode_oa_lookup", - "load_uk_dataset", "map_to_hmrc_band_indices", "metric_names", "metric_names_from_target_profile", "metric_tables_by_area_group", "optimize_firm_weights", + "packaged_occupation_csv_path", "prepare_area_frame", "prepare_geography_crosswalk", "prepare_household_frame", "read_local_table", "read_uk_firm_source_data", + "run_ai_shock_analysis", + "run_grid", + "run_scenario", "set_simulation_area_group", + "soc_major_group_for_persons", "solve_firm_weights", "solve_stacked_local_weights", "sort_households_by_id", @@ -338,10 +476,12 @@ "uk_geography_ladder_assignment_summary", "uk_geography_ladder_gate", "update_england_wales_lad_codes", + "validate_geography_coverage", "validate_uk_firm_population", "validate_uk_rowwise_dataset_tables", - "validate_geography_coverage", "write_geography_crosswalk", + "write_grid_csv", + "write_grid_json", "write_local_candidate_outputs", "write_long_geography_weights", "write_uk_firm_population", diff --git a/packages/populace-build/src/populace/build/uk_runtime/ai_exposure.py b/packages/populace-build/src/populace/build/uk_runtime/ai_exposure.py new file mode 100644 index 00000000..7025e9ff --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/ai_exposure.py @@ -0,0 +1,227 @@ +"""AI-exposure scores for UK SOC 2020 occupation codes. + +This module ships a pre-built crosswalk +(``ai_exposure_data/uk_soc2020_ai_exposure.csv``) that attaches occupation-level +AI-exposure measures to every SOC 2020 unit group (4-digit), plus an aggregated +major-group (1-digit) table weighted by ASHE employment. Lookups fall back from +4-digit to 3-digit to 2-digit parent means so that partially coded microdata +still receives a score. + +Measures +-------- +``c_aioe`` (primary) + Complementarity-adjusted AI occupational exposure, + ``AIOE * (1 - (theta - theta_min))``, following Pizzinelli et al. (2023, + IMF WP/23/216) and its UK/Ireland application in Williamson et al. (2025, + J Labour Market Res 59:30). AIOE is Felten, Raj and Seamans (2021). +``complementarity`` / ``complementarity_theta`` + The AI complementarity score theta in [0, 1], reconstructed from the recipe + published in IMF WP/23/216 (11 O*NET work contexts plus job zones grouped + into six components) using O*NET 27.3 data. The reconstruction reproduces + the paper's published extremes exactly (minimum at US SOC 51-9031, maximum + at US SOC 29-1022) with a level offset of about +0.03-0.04 attributable to + the O*NET vintage; ``theta_min`` is taken from the reconstruction for + internal consistency. Occupation-level theta is not otherwise openly + published (the IMF and IGEES datasets are available on request only). +``dsit_aioe`` / ``dsit_llm`` + Standardised AIOE and LLM-only exposure on UK SOC 2010 4-digit codes from + the DSIT/DfE report "The impact of AI on UK jobs and training" (Nov 2023), + Annex 1, mapped SOC 2010 -> SOC 2020 with the ONS coding index. These are + the official UK mappings of the Felten scores. +``eloundou_beta`` + Eloundou et al. (2023) "GPTs are GPTs" human-annotated beta exposure + (share of tasks where LLMs can halve completion time, direct or with + tooling), in [0, 1]. +``felten_aioe`` + Felten, Raj and Seamans (2021) standardised AIOE. + +Crosswalk construction +---------------------- +US-based measures are chained US SOC 2018 -> US SOC 2010 (BLS crosswalk, +Nov 2017) -> ISCO-08 (BLS ISCO-08/SOC 2010 crosswalk, 2012) -> SOC 2020 (ONS +SOC 2020 Volume 2 coding index, which assigns an ISCO-08 code to every SOC 2020 +index entry). Many-to-many links are resolved with employment-unweighted means +and flagged ``chained`` in ``mapping_quality``; ``direct`` marks single-path +links and ``imputed-from-parent`` marks unit groups filled from 3-digit or +2-digit sibling means. The major-group table weights unit groups by ASHE +Table 14 (2025) employee-job counts on SOC 2020 4-digit codes. + +Sources and licences +-------------------- +- Eloundou et al. (2023), github.com/openai/GPTs-are-GPTs (MIT licence). +- Felten, Raj and Seamans (2021), github.com/AIOE-Data/AIOE (public GitHub + release accompanying the paper). +- Pizzinelli et al. (2023), IMF Working Paper WP/23/216 (methodology); + O*NET 27.3 database, US Department of Labor (CC BY 4.0). +- DSIT/DfE (2023), "The impact of AI on UK jobs and training", Annex 1, + gov.uk (Open Government Licence v3.0). +- ONS SOC 2020 Volumes 1-2 and ASHE Table 14 (Open Government Licence v3.0). +- BLS SOC crosswalks (US public domain). +""" + +from __future__ import annotations + +import warnings +from functools import lru_cache +from importlib import resources as importlib_resources + +import numpy as np +import pandas as pd + +MEASURE_TO_COLUMN = { + "c_aioe": "c_aioe", + "complementarity": "complementarity_theta", + "complementarity_theta": "complementarity_theta", + "dsit_aioe": "dsit_aioe", + "dsit_llm": "dsit_llm", + "eloundou_beta": "eloundou_beta", + "felten_aioe": "felten_aioe", +} +MAPPING_QUALITIES = ("direct", "chained", "imputed-from-parent") + +_DATA_DIR = "ai_exposure_data" +_UNIT_GROUP_FILE = "uk_soc2020_ai_exposure.csv" +_MAJOR_GROUP_FILE = "uk_soc2020_major_group_ai_exposure.csv" + + +def _data_path(filename: str): + return importlib_resources.files("populace.build.uk_runtime").joinpath( + _DATA_DIR, filename + ) + + +@lru_cache(maxsize=1) +def load_ai_exposure_table() -> pd.DataFrame: + """SOC 2020 unit-group (4-digit) AI-exposure crosswalk.""" + + with importlib_resources.as_file(_data_path(_UNIT_GROUP_FILE)) as path: + table = pd.read_csv(path, dtype={"soc2020_code": str}) + return table.set_index("soc2020_code") + + +@lru_cache(maxsize=1) +def load_major_group_ai_exposure_table() -> pd.DataFrame: + """SOC 2020 major-group (1-digit) employment-weighted AI-exposure table.""" + + with importlib_resources.as_file(_data_path(_MAJOR_GROUP_FILE)) as path: + table = pd.read_csv(path, dtype={"soc2020_major_group": str}) + return table.set_index("soc2020_major_group") + + +def _measure_column(measure: str) -> str: + if measure not in MEASURE_TO_COLUMN: + raise ValueError( + f"Unknown measure {measure!r}; expected one of {sorted(MEASURE_TO_COLUMN)}" + ) + return MEASURE_TO_COLUMN[measure] + + +def _normalise_codes(codes) -> np.ndarray: + values = pd.Series(np.asarray(codes, dtype=object)).astype("string") + values = values.str.strip().str.replace(r"\.0$", "", regex=True) + values = values.where(values.str.fullmatch(r"\d{1,4}")) + return values.to_numpy(dtype=object) + + +@lru_cache(maxsize=len(MEASURE_TO_COLUMN) * 4) +def _prefix_means(column: str, length: int) -> dict[str, float]: + table = load_ai_exposure_table() + grouped = table[column].groupby(table.index.str[:length]).mean() + return grouped.to_dict() + + +def exposure_for_soc(codes, measure: str = "c_aioe") -> np.ndarray: + """AI-exposure scores for an array of SOC 2020 codes. + + Parameters + ---------- + codes: + Array-like of SOC 2020 codes (strings or integers). 4-digit unit + groups are matched exactly; 3-digit and 2-digit codes (and 4-digit + codes missing from the table) resolve to unweighted means over the + matching unit groups, falling back 4-digit -> 3-digit -> 2-digit. + measure: + One of ``MEASURE_TO_COLUMN`` keys, e.g. ``"c_aioe"``, + ``"complementarity"``, ``"eloundou_beta"``, ``"felten_aioe"``, + ``"dsit_aioe"``, ``"dsit_llm"``. + + Returns + ------- + numpy.ndarray of float, NaN where no code (or parent code) matches; a + warning lists unmatched codes. + """ + + column = _measure_column(measure) + table = load_ai_exposure_table() + exact = table[column].to_dict() + normalised = _normalise_codes(codes) + + result = np.full(len(normalised), np.nan, dtype=float) + unmatched: list[str] = [] + for position, code in enumerate(normalised): + if code is None or code is pd.NA: + unmatched.append(str(np.asarray(codes, dtype=object)[position])) + continue + value = exact.get(code) if len(code) == 4 else None + if value is None: + for length in (3, 2): + if len(code) < length: + continue + value = _prefix_means(column, length).get(code[:length]) + if value is not None: + break + if value is None or np.isnan(value): + unmatched.append(code) + else: + result[position] = value + if unmatched: + warnings.warn( + "No SOC 2020 AI-exposure match (returning NaN) for codes: " + + ", ".join(sorted(set(unmatched))), + stacklevel=2, + ) + return result + + +def exposure_for_major_group(groups, measure: str = "c_aioe") -> np.ndarray: + """Employment-weighted AI-exposure scores for SOC 2020 major groups (1-9). + + Uses the shipped major-group table (ASHE Table 14 2025 employee-job + weights). Unknown groups return NaN with a warning. + + Accepts both codings of the 1-digit major group: plain ``1``-``9`` + (string or integer) and the FRS adult.tab convention, which stores the + major group in thousands (``1000``-``9000``). Any 4-digit code that is + an exact multiple of 1000 is normalised to its leading digit, so both + codings return identical scores. + """ + + column = _measure_column(measure) + table = load_major_group_ai_exposure_table() + lookup = table[column].to_dict() + normalised = _normalise_codes(groups) + normalised = np.array( + [ + code[0] + if isinstance(code, str) and len(code) == 4 and code.endswith("000") + else code + for code in normalised + ], + dtype=object, + ) + + result = np.full(len(normalised), np.nan, dtype=float) + unmatched: list[str] = [] + for position, group in enumerate(normalised): + value = None if group is None or group is pd.NA else lookup.get(group) + if value is None or np.isnan(value): + unmatched.append(str(np.asarray(groups, dtype=object)[position])) + else: + result[position] = value + if unmatched: + warnings.warn( + "No SOC 2020 major-group AI-exposure match (returning NaN) for: " + + ", ".join(sorted(set(unmatched))), + stacklevel=2, + ) + return result diff --git a/packages/populace-build/src/populace/build/uk_runtime/ai_exposure_data/uk_soc2020_ai_exposure.csv b/packages/populace-build/src/populace/build/uk_runtime/ai_exposure_data/uk_soc2020_ai_exposure.csv new file mode 100644 index 00000000..d7489b68 --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/ai_exposure_data/uk_soc2020_ai_exposure.csv @@ -0,0 +1,413 @@ +soc2020_code,soc_title,c_aioe,complementarity_theta,felten_aioe,eloundou_beta,dsit_aioe,dsit_llm,mapping_quality,n_isco_links +1111,Chief executives and senior officials,0.596223,0.700744,0.960908,0.415035,1.062646,1.064345,chained,4 +1112,Elected officers and representatives,0.596223,0.700744,0.960908,0.516667,1.062646,1.064345,direct,1 +1121,Production managers and directors in manufacturing,0.514544,0.677025,0.745515,0.388176,0.683651,0.593709,chained,2 +1122,Production managers and directors in construction,0.361098,0.725006,0.562807,0.342207,0.663920,0.462822,chained,3 +1123,Production managers and directors in mining and energy,0.084565,0.691844,0.131533,0.256830,0.440265,0.280070,chained,3 +1131,Financial managers and directors,0.777740,0.661944,1.115860,0.448825,1.269735,1.200352,chained,3 +1132,"Marketing, sales and advertising directors",0.864207,0.644145,1.222384,0.493798,1.257397,1.185543,chained,3 +1133,Public relations and communications directors,0.918386,0.638889,1.294038,0.542006,1.222017,1.118437,chained,1 +1134,Purchasing managers and directors,0.603388,0.671042,0.890792,0.472971,1.442071,1.326547,direct,1 +1135,Charitable organisation managers and directors,0.697683,0.664447,1.067714,0.457921,0.962517,1.005113,chained,3 +1136,Human resource managers and directors,0.933394,0.630671,1.301025,0.462600,1.300896,1.393290,chained,1 +1137,Information technology directors,0.728667,0.586895,0.957630,0.560658,1.297814,1.155107,chained,3 +1139,Functional managers and directors n.e.c.,0.722027,0.658079,1.055871,0.441132,1.222955,1.182672,chained,9 +1140,"Directors in logistics, warehousing and transport",0.603388,0.671042,0.890792,0.472971,1.100874,1.037275,direct,1 +1150,Managers and directors in retail and wholesale,0.474485,0.729349,0.764719,0.380823,0.706437,0.785590,chained,2 +1161,Officers in armed forces,0.549084,0.681318,0.809239,0.396001,-0.025391,-0.019518,imputed-from-parent,1 +1162,Senior police officers,0.433322,0.703416,0.653790,0.398714,-0.311725,-0.321001,chained,3 +1163,"Senior officers in fire, ambulance, prison and related services",0.664847,0.659219,0.964687,0.393289,0.260944,0.281966,direct,1 +1171,Health services and public health managers and directors,0.769591,0.668490,0.938033,0.455940,1.251801,1.309382,chained,3 +1172,Social services managers and directors,0.643243,0.708854,0.997080,0.481027,0.931423,1.044722,chained,2 +1211,Managers and proprietors in agriculture and horticulture,0.065202,0.658484,0.128266,0.223333,0.115861,0.065649,chained,3 +1212,"Managers and proprietors in forestry, fishing and related services",-0.380449,0.668281,-0.516793,0.178385,0.044627,-0.039301,chained,5 +1221,Hotel and accommodation managers and proprietors,0.166315,0.660179,0.260482,0.342942,0.328336,0.390834,chained,3 +1222,Restaurant and catering establishment managers and proprietors,-0.086375,0.636389,-0.121311,0.395833,-0.234877,-0.043301,direct,1 +1223,Publicans and managers of licensed premises,0.601947,0.664615,0.863378,0.450152,-0.082135,0.086727,chained,3 +1224,Leisure and sports managers and proprietors,0.394845,0.685952,0.572541,0.359159,0.211585,0.252430,chained,3 +1225,Travel agency managers and proprietors,0.664847,0.659219,0.964687,0.393289,1.115373,1.268396,direct,1 +1231,Health care practice managers,0.924133,0.539896,1.140286,0.421627,0.979119,1.047486,chained,1 +1232,"Residential, day and domiciliary care managers and proprietors",0.641817,0.697095,0.970576,0.446692,0.931423,1.044722,chained,3 +1233,Early education and childcare services proprietors,0.673264,0.686250,0.993818,0.395798,0.850094,0.887074,chained,2 +1241,Managers in transport and distribution,0.495530,0.658329,0.688347,0.414481,0.832416,0.835429,chained,4 +1242,Managers in storage and warehousing,0.457662,0.703113,0.699740,0.401287,0.864872,0.783866,chained,2 +1243,Managers in logistics,0.603388,0.671042,0.890792,0.472971,1.244082,1.150285,direct,1 +1251,"Property, housing and estate managers",0.670957,0.648292,0.967729,0.452678,0.763951,0.820174,chained,7 +1252,Garage managers and proprietors,0.664847,0.659219,0.964687,0.393289,-0.639264,-0.735793,direct,1 +1253,Hairdressing and beauty salon managers and proprietors,0.664847,0.659219,0.964687,0.393289,-0.085255,0.102104,direct,1 +1254,Waste disposal and environmental services managers,-0.005884,0.655982,0.063334,0.318345,0.962517,1.005113,chained,3 +1255,Managers and directors in the creative industries,0.709359,0.647685,1.008729,0.435270,0.635626,0.648302,chained,4 +1256,Betting shop and gambling establishment managers,0.637027,0.664761,0.930996,0.416290,0.443483,0.473458,chained,2 +1257,Hire services managers and proprietors,0.637027,0.664761,0.930996,0.416290,0.556640,0.480649,chained,2 +1258,Directors in consultancy services,0.824512,0.645472,1.169158,0.425827,1.052894,0.970800,chained,4 +1259,Managers and proprietors in other services n.e.c.,0.606678,0.699858,0.893008,0.409166,0.727692,0.649528,chained,6 +2111,Chemical scientists,0.347636,0.607807,0.482833,0.448171,0.486788,0.321953,chained,2 +2112,Biological scientists,0.324754,0.710747,0.458987,0.442547,0.640197,0.570855,chained,5 +2113,Biochemists and biomedical scientists,0.352278,0.665556,0.489856,0.332997,0.424060,0.285457,chained,3 +2114,Physical scientists,0.686804,0.649609,0.973802,0.511225,1.076097,0.955436,chained,6 +2115,Social and humanities scientists,0.765524,0.627595,1.069692,0.501075,1.236546,1.353337,chained,7 +2119,Natural and social science professionals n.e.c.,0.628951,0.607956,0.846609,0.553553,0.448010,0.259742,chained,3 +2121,Civil engineers,0.859418,0.646424,1.224529,0.447881,1.357744,1.023084,chained,2 +2122,Mechanical engineers (professional),0.679156,0.626900,0.945506,0.423419,1.067056,0.820072,chained,6 +2123,Electrical engineers,0.664486,0.603299,0.896091,0.427040,0.994896,0.796731,chained,2 +2124,Electronics engineers (professional),0.604497,0.581510,0.784546,0.415417,0.994950,0.775633,chained,2 +2125,Production and process engineers,0.786900,0.630569,1.101219,0.429521,1.066374,0.809225,chained,4 +2126,Aerospace engineers,0.693657,0.606727,0.930771,0.410936,0.981833,0.737346,chained,2 +2127,Engineering project managers and project engineers,0.703559,0.634271,0.989892,0.463367,1.148883,0.914107,chained,5 +2129,Engineering professionals n.e.c.,0.740861,0.650880,1.057184,0.475545,1.031699,0.795313,chained,9 +2131,IT project managers,0.812375,0.569484,1.040893,0.569330,1.297814,1.155107,chained,5 +2132,IT managers,0.788184,0.578539,1.019932,0.559365,1.123480,0.955475,chained,6 +2133,"IT business analysts, architects and systems designers",0.874541,0.555392,1.100819,0.588434,1.090212,0.841516,chained,7 +2134,Programmers and software development professionals,0.786451,0.555454,0.982681,0.580077,1.116764,0.880324,chained,10 +2135,Cyber security professionals,0.976554,0.558917,1.237065,0.569420,1.150989,1.007082,chained,2 +2136,IT quality and testing professionals,0.976554,0.558917,1.237065,0.569420,1.172144,0.950847,chained,2 +2137,IT network professionals,0.451328,0.570694,0.580331,0.543301,1.210687,0.999369,direct,1 +2139,Information technology professionals n.e.c.,0.809611,0.572101,1.039157,0.598531,1.054320,0.814624,chained,11 +2141,Web design professionals,0.833118,0.551387,1.042912,0.551869,1.014493,0.756412,chained,5 +2142,Graphic and multimedia designers,0.251010,0.537922,0.329566,0.403051,0.814540,0.561601,chained,6 +2151,Conservation professionals,0.383363,0.668652,0.541572,0.594972,0.472417,0.415049,chained,1 +2152,Environment professionals,0.524722,0.653435,0.752228,0.458176,0.364801,0.263008,chained,4 +2161,Research and development (R&D) managers,0.851502,0.621775,1.165020,0.497357,0.973429,0.885871,chained,5 +2162,"Other researchers, unspecified discipline",0.724750,0.675196,1.070665,0.501957,1.040185,1.049122,chained,2 +2211,Generalist medical practitioners,0.487286,0.748424,0.734526,0.351547,0.798116,0.777987,chained,2 +2212,Specialist medical practitioners and consultants,0.348317,0.726203,0.500237,0.343188,0.681321,0.655432,chained,5 +2221,Physiotherapists,-0.113694,0.723472,-0.195691,0.340213,-0.310199,0.010028,chained,1 +2222,Occupational therapists,0.333958,0.710813,0.501590,0.380005,0.472913,0.623271,chained,1 +2223,Speech and language therapists,0.575397,0.699896,0.889587,0.422395,0.904202,1.035292,chained,1 +2224,Psychotherapists and cognitive behaviour therapists,0.664414,0.681882,0.958910,0.425970,0.552500,0.604036,chained,2 +2225,Clinical psychologists,0.994870,0.652951,1.416230,0.471935,1.455629,1.556178,chained,1 +2226,Other psychologists,0.994870,0.652951,1.416230,0.471935,1.455629,1.556178,chained,1 +2229,Therapy professionals n.e.c.,0.355149,0.678408,0.528098,0.328215,0.178079,0.335466,chained,11 +2231,Midwifery nurses,0.134050,0.722049,0.213092,0.283086,0.136594,0.217538,chained,2 +2232,Registered community nurses,0.134541,0.737537,0.220262,0.258266,0.249086,0.249312,chained,1 +2233,Registered specialist nurses,0.134541,0.737537,0.220262,0.258266,0.249086,0.249312,chained,1 +2234,Registered nurse practitioners,0.134541,0.737537,0.220262,0.258266,0.192840,0.233425,chained,1 +2235,Registered mental health nurses,0.134541,0.737537,0.220262,0.258266,0.249086,0.249312,chained,1 +2236,Registered children's nurses,0.134541,0.737537,0.220262,0.258266,0.249086,0.249312,chained,1 +2237,Other registered nursing professionals,0.100474,0.663561,0.165303,0.299710,-0.010657,0.162103,chained,5 +2240,Veterinarians,-0.043086,0.773889,-0.074996,0.229730,-0.144768,-0.124057,direct,1 +2251,Pharmacists,0.461396,0.676892,0.686406,0.429310,0.701469,0.590811,chained,2 +2252,Optometrists,0.317707,0.718056,0.504019,0.050000,0.519369,0.400848,direct,1 +2253,Dental practitioners,-0.055931,0.798802,-0.103694,0.148977,-0.208682,-0.343165,chained,1 +2254,Medical radiographers,-0.137122,0.615028,-0.186662,0.231494,-0.194185,-0.261882,chained,2 +2255,Paramedics,0.063420,0.749618,-0.290215,0.155220,-0.297613,-0.290933,chained,2 +2256,Podiatrists,0.333958,0.710813,0.501590,0.380005,0.572031,0.487749,chained,1 +2259,Other health professionals n.e.c.,0.236680,0.673511,0.354031,0.307994,0.504215,0.464512,chained,12 +2311,Higher education teaching professionals,0.805522,0.707427,1.097268,0.465582,1.233384,1.482924,chained,2 +2312,Further education teaching professionals,0.502295,0.713657,0.653279,0.419332,1.242276,1.542836,chained,2 +2313,Secondary education teaching professionals,0.787929,0.687014,1.191325,0.341270,0.998843,1.084499,direct,1 +2314,Primary education teaching professionals,0.420908,0.669444,0.642109,0.314129,0.663105,0.835133,chained,2 +2315,Nursery education teaching professionals,0.567737,0.675208,0.862024,0.384420,0.663105,0.835133,chained,3 +2316,Special and additional needs education teaching professionals,0.679201,0.676962,0.997318,0.367471,0.570358,0.766269,chained,3 +2317,Teachers of English as a foreign language,0.794708,0.602361,1.066878,0.344617,1.143549,1.293499,chained,1 +2319,Teaching professionals n.e.c.,0.659440,0.646100,0.896730,0.381135,1.143549,1.293499,chained,7 +2321,Head teachers and principals,0.685913,0.731536,0.954629,0.421054,1.116114,1.283711,chained,2 +2322,Education managers,0.642576,0.693288,0.912809,0.362562,1.048602,1.109127,chained,5 +2323,Education advisers and school inspectors,0.861394,0.686736,1.301855,0.525000,1.248525,1.347094,direct,1 +2324,Early education and childcare services managers,0.557926,0.707062,0.867655,0.477458,0.666453,0.755530,chained,2 +2329,Other educational professionals n.e.c,0.771060,0.656602,1.036852,0.446927,1.142133,1.277586,chained,5 +2411,Barristers and judges,0.898323,0.629242,1.257385,0.382426,1.174695,1.246181,chained,4 +2412,Solicitors and lawyers,0.854943,0.705000,1.328783,0.475000,1.318113,1.446088,direct,1 +2419,Legal professionals n.e.c.,0.818645,0.617429,1.121262,0.418782,1.216614,1.155360,chained,5 +2421,Chartered and certified accountants,1.149431,0.540278,1.426218,0.595000,1.456983,1.245348,chained,1 +2422,Finance and investment analysts and advisers,0.840212,0.562144,1.076805,0.534147,1.127519,1.095816,chained,6 +2423,Taxation experts,1.156658,0.518576,1.396519,0.636210,1.220044,1.084746,chained,2 +2431,Management consultants and business analysts,1.095700,0.573285,1.407263,0.597018,1.485114,1.453020,chained,3 +2432,Marketing and commercial managers,0.960239,0.577474,1.248999,0.594559,1.239327,1.187197,chained,2 +2433,"Actuaries, economists and statisticians",0.966040,0.589229,1.260165,0.601049,1.327720,1.309960,chained,6 +2434,Business and related research professionals,0.661523,0.603339,0.866159,0.496935,1.002750,0.946389,chained,8 +2435,Professional/Chartered company secretaries,0.918826,0.607512,1.246959,0.461563,1.404545,1.271316,chained,2 +2439,"Business, research and administrative professionals n.e.c.",0.669261,0.619947,0.922384,0.455712,1.369555,1.246408,chained,6 +2440,Business and financial project management professionals,0.816494,0.627463,1.126904,0.464257,1.335104,1.273105,chained,7 +2451,Architects,0.720655,0.655868,1.040765,0.399327,0.979650,0.659780,chained,2 +2452,"Chartered architectural technologists, planning officers and consultants",0.568631,0.661794,0.809776,0.473964,0.943274,0.658302,chained,3 +2453,Quantity surveyors,0.420513,0.648784,0.579186,0.414441,1.174271,0.915188,chained,2 +2454,Chartered surveyors,0.252284,0.638854,0.318100,0.426005,0.230037,-0.016138,chained,2 +2455,Construction project managers and related professionals,0.625403,0.680530,0.923548,0.461551,0.708515,0.474397,chained,4 +2461,Social workers,0.658152,0.659991,0.949970,0.352816,0.773343,0.988263,chained,1 +2462,Probation officers,0.658152,0.659991,0.949970,0.352816,0.147681,0.498935,chained,1 +2463,Clergy,0.807983,0.699792,1.246741,0.346320,1.134349,1.346421,chained,2 +2464,Youth work professionals,0.658152,0.659991,0.949970,0.352816,0.931469,1.171531,chained,1 +2469,Welfare professionals n.e.c.,0.194987,0.656554,0.292641,0.309929,0.931469,1.171531,chained,3 +2471,Librarians,0.528621,0.637847,0.743955,0.500000,0.888748,0.736199,chained,1 +2472,"Archivists, conservators and curators",0.182299,0.619483,0.281428,0.400416,0.389318,0.358944,chained,3 +2481,Quality control and planning engineers,0.669145,0.649635,0.949060,0.448938,0.480017,0.278634,chained,5 +2482,Quality assurance and regulatory professionals,0.790058,0.614412,1.069604,0.509582,1.343690,1.309008,chained,6 +2483,Environmental health professionals,0.289472,0.685593,0.448279,0.371094,0.680341,0.591213,chained,2 +2491,"Newspaper, periodical and broadcast editors",0.836525,0.625347,1.157450,0.640196,1.155081,1.300637,chained,1 +2492,Newspaper and periodical broadcast journalists and reporters,0.836525,0.625347,1.157450,0.640196,1.155081,1.300637,chained,1 +2493,Public relations professionals,0.932117,0.625590,1.289483,0.664942,1.290159,1.461133,chained,2 +2494,Advertising accounts managers and creative directors,0.951291,0.585668,1.250690,0.600197,1.222017,1.118437,chained,2 +3111,Laboratory technicians,0.108107,0.619035,0.136963,0.331078,0.434823,0.210988,chained,6 +3112,Electrical and electronics technicians,0.046791,0.614204,0.097041,0.268207,0.065292,-0.254300,chained,5 +3113,Engineering technicians,0.229180,0.616948,0.317063,0.331106,0.065965,-0.188392,chained,7 +3114,Building and civil engineering technicians,-0.114587,0.659878,-0.177924,0.280775,0.398731,0.022040,chained,2 +3115,Quality assurance technicians,0.248898,0.664723,0.338955,0.477487,-0.173569,-0.345059,chained,2 +3116,"Planning, process and production technicians",0.064740,0.571840,0.084973,0.310276,0.156661,-0.109040,chained,2 +3119,"Science, engineering and production technicians n.e.c.",0.111144,0.602780,0.193333,0.320786,0.219372,-0.010075,chained,6 +3120,"CAD, drawing and architectural technicians",0.305919,0.599039,0.387526,0.432951,0.822391,0.558558,chained,3 +3131,IT technicians,0.747371,0.562191,0.948879,0.560254,1.118770,0.846233,chained,6 +3132,IT user support technicians,0.449983,0.596528,0.596485,0.600697,0.514899,0.448050,chained,3 +3133,Database administrators and web content technicians,0.773441,0.559922,0.974776,0.563855,1.049268,0.896881,chained,6 +3211,Dispensing opticians,0.075724,0.601597,0.101397,0.238095,0.003985,-0.079728,direct,1 +3212,Pharmaceutical technicians,0.005455,0.571944,0.007026,0.426471,-0.096619,-0.134710,direct,1 +3213,Medical and dental technicians,-0.099642,0.637219,-0.134717,0.187236,-0.300066,-0.304287,chained,8 +3214,Complementary health associate professionals,0.158811,0.667025,0.258029,0.310396,0.486396,0.589498,chained,6 +3219,Health associate professionals n.e.c.,-0.086718,0.648129,-0.110771,0.249782,0.118743,0.130999,chained,4 +3221,Youth and community workers,0.316321,0.668472,0.465226,0.375000,0.697215,0.871328,chained,2 +3222,Child and early years officers,0.316321,0.668472,0.458445,0.320833,0.577215,0.811538,chained,2 +3223,Housing officers,0.316321,0.668472,0.465226,0.375000,1.137369,1.255724,direct,1 +3224,Counsellors,0.604785,0.652381,0.854942,0.334596,1.124001,1.293630,chained,3 +3229,Welfare and housing associate professionals n.e.c.,0.624818,0.615971,0.854728,0.417683,1.015860,1.085550,chained,7 +3231,Higher level teaching assistants,0.002100,0.638993,0.451664,0.266667,0.382332,0.697323,direct,1 +3232,Early education and childcare practitioners,0.002100,0.638993,0.237539,0.303442,0.299170,0.492443,chained,2 +3240,Veterinary nurses,-0.422912,0.608715,-0.567950,0.147845,-0.494757,-0.417664,chained,1 +3311,Non-commissioned officers and other ranks,-0.445868,0.594041,-0.607480,0.177859,-0.797797,-0.600852,chained,3 +3312,Police officers (sergeant and below),-0.165116,0.682789,-0.271806,0.319451,-0.725955,-0.686317,chained,3 +3313,Fire service officers (watch manager and below),0.081584,0.701047,0.054244,0.350697,-0.990830,-1.026606,chained,3 +3314,Prison service officers (below principal officer),-0.474346,0.704549,-0.722982,0.260623,-0.894971,-0.652007,chained,1 +3319,Protective service associate professionals n.e.c.,0.589339,0.562841,0.696499,0.430252,0.593200,0.523680,chained,8 +3411,Artists,-0.172643,0.527760,-0.183020,0.291764,0.103751,-0.061083,chained,6 +3412,"Authors, writers and translators",0.882387,0.584555,1.151252,0.642242,0.536154,0.503896,chained,8 +3413,"Actors, entertainers and presenters",0.123760,0.572601,0.203505,0.347034,0.689516,0.817351,chained,8 +3414,Dancers and choreographers,-0.233338,0.623380,-0.348619,0.224983,-1.462740,-0.807686,chained,3 +3415,Musicians,0.630568,0.586605,0.840624,0.446098,0.485614,0.652375,chained,3 +3416,"Arts officers, producers and directors",0.318316,0.592534,0.477784,0.322985,1.019911,0.997991,chained,5 +3417,"Photographers, audio-visual and broadcasting equipment operators",0.038627,0.573087,0.087370,0.272585,-0.031491,-0.255392,chained,7 +3421,Interior designers,0.268295,0.598385,0.373187,0.362041,0.606762,0.414584,chained,2 +3422,"Clothing, fashion and accessories designers",0.268295,0.598385,0.373187,0.362041,0.606762,0.414584,chained,2 +3429,Design occupations n.e.c.,0.443038,0.575302,0.578861,0.393904,-0.021947,-0.072312,chained,7 +3431,Sports players,-1.302668,0.630417,-1.814336,0.000000,-2.056279,-1.498609,direct,1 +3432,"Sports coaches, instructors and officials",0.263536,0.633100,0.346728,0.362048,-0.145422,0.098952,chained,3 +3433,Fitness and wellbeing instructors,-0.053963,0.648652,-0.084027,0.285468,-0.515354,-0.081576,chained,2 +3511,Aircraft pilots and air traffic controllers,0.223789,0.668472,0.284494,0.280460,0.286919,-0.071460,chained,2 +3512,Ship and hovercraft officers,0.378227,0.638802,0.480987,0.370131,-0.820722,-1.002488,chained,5 +3520,Legal associate professionals,0.732484,0.640014,1.042344,0.417982,1.004006,1.012315,chained,3 +3531,Brokers,0.946928,0.627654,1.313417,0.479195,1.313480,1.304061,chained,3 +3532,Insurance underwriters,0.928414,0.581262,1.213292,0.554885,1.299381,1.313286,chained,3 +3533,Financial and accounting technicians,1.009172,0.526493,1.232984,0.454477,1.301494,1.136863,chained,2 +3534,Financial accounts managers,0.948794,0.590910,1.248840,0.503923,1.211635,1.081061,chained,9 +3541,"Estimators, valuers and assessors",0.385822,0.625167,0.526565,0.380451,0.898685,0.706539,chained,6 +3542,Importers and exporters,0.260417,0.563559,0.329547,0.392980,1.006527,0.967975,chained,1 +3543,Project support officers,0.998050,0.541593,1.240031,0.555691,1.145464,1.105210,chained,3 +3544,Data analysts,1.007510,0.560845,1.282170,0.633001,1.388826,1.272407,chained,4 +3549,Business associate professionals n.e.c.,0.731951,0.602672,0.940045,0.502074,1.388826,1.272407,chained,22 +3551,Buyers and procurement officers,0.788945,0.646759,1.152236,0.419506,1.042770,1.011764,chained,3 +3552,Business sales executives,0.732339,0.629627,1.026785,0.505037,1.195432,1.226144,chained,9 +3553,Merchandisers,0.545841,0.568883,0.700558,0.473041,0.665827,0.717807,chained,5 +3554,Advertising and marketing associate professionals,0.757279,0.569413,0.982855,0.565503,0.919270,0.920311,chained,8 +3555,Estate agents and auctioneers,0.753275,0.681019,1.124643,0.447474,0.364800,0.354329,chained,3 +3556,Sales accounts and business development managers,0.755790,0.646889,1.074212,0.479961,1.135048,1.122698,chained,11 +3557,Events managers and organisers,0.543618,0.653713,0.773684,0.401634,0.274898,0.377918,chained,3 +3560,Public services associate professionals,0.548047,0.612871,0.707417,0.475065,0.758647,0.668339,chained,16 +3571,Human resources and industrial relations officers,0.787599,0.610052,1.030957,0.405865,1.231475,1.334640,chained,3 +3572,Careers advisers and vocational guidance specialists,0.698339,0.643015,0.963134,0.342683,1.052144,1.120594,chained,3 +3573,Information technology trainers,0.860607,0.630278,1.198409,0.592105,1.220675,1.329152,chained,2 +3574,Other vocational and industrial trainers,0.674160,0.668594,0.980853,0.510338,1.297802,1.364805,chained,2 +3581,Inspectors of standards and regulations,0.274471,0.618197,0.338439,0.394038,-0.151279,-0.272797,chained,11 +3582,Health and safety managers and officers,0.369954,0.665929,0.529048,0.396079,0.938867,0.814285,chained,6 +4111,National government administrative occupations,0.711392,0.561111,0.879304,0.499126,1.046242,1.028073,chained,13 +4112,Local government administrative occupations,0.848386,0.531200,1.033654,0.537055,1.166003,1.252415,chained,12 +4113,Officers of non-governmental organisations,0.886459,0.515577,1.058173,0.558275,1.110913,1.152207,chained,6 +4121,Credit controllers,1.048098,0.554931,1.321419,0.564966,1.386253,1.493785,chained,2 +4122,"Book-keepers, payroll managers and wages clerks",1.042910,0.503844,1.236631,0.471626,1.342601,1.180267,chained,5 +4123,Bank and post office clerks,0.676777,0.568920,0.881585,0.511590,0.846962,0.877297,chained,4 +4124,Finance officers,1.044780,0.522198,1.265116,0.528429,0.917439,0.759380,chained,2 +4129,Financial administrative occupations n.e.c.,0.598795,0.526325,0.737523,0.463625,0.952567,0.929717,chained,11 +4131,Records clerks and assistants,0.562424,0.548059,0.696399,0.453982,0.809192,0.710217,chained,26 +4132,Pensions and insurance clerks and assistants,0.993966,0.536895,1.220458,0.588550,1.229559,1.197351,chained,3 +4133,Stock control clerks and assistants,0.276983,0.583930,0.342193,0.437505,-0.522134,-0.428258,chained,5 +4134,Transport and distribution clerks and assistants,0.449390,0.564755,0.585817,0.446327,0.936311,0.862968,chained,4 +4135,Library clerks and assistants,0.218952,0.536898,0.262553,0.493421,0.028565,0.141163,chained,3 +4136,Human resources administrative occupations,1.107661,0.529792,1.353098,0.465517,1.351710,1.489420,direct,1 +4141,Office managers,0.581446,0.656608,0.840937,0.451685,0.996048,1.044701,chained,6 +4142,Office supervisors,0.911562,0.557465,1.009468,0.515174,1.012533,1.113101,chained,4 +4143,Customer service managers,0.790485,0.587361,1.038688,0.531250,1.131081,1.214532,direct,1 +4151,Sales administrators,0.409870,0.540104,0.489365,0.507566,0.423132,0.418940,chained,2 +4152,Data entry administrators,0.675898,0.522677,0.789796,0.563885,0.483168,0.403698,chained,7 +4159,Other administrative occupations n.e.c.,0.735903,0.506686,0.876564,0.537635,0.902101,0.938013,chained,14 +4211,Medical secretaries,0.923636,0.577899,1.200273,0.396528,1.206307,1.288453,chained,2 +4212,Legal secretaries,0.765995,0.531071,0.924051,0.573264,0.987464,1.131258,chained,2 +4213,School secretaries,0.932900,0.500156,1.097906,0.555305,0.859721,1.054667,chained,2 +4214,Company secretaries and administrators,0.918361,0.572691,1.201838,0.543084,1.072984,1.167823,chained,2 +4215,Personal assistants and other secretaries,0.858188,0.506343,1.016941,0.621806,0.943984,1.081685,chained,3 +4216,Receptionists,0.709703,0.541076,0.872129,0.566026,0.638318,0.768631,chained,2 +4217,Typists and related keyboard occupations,0.672265,0.505465,0.805591,0.543416,0.483168,0.403698,chained,5 +5111,Farmers,-0.283200,0.622736,-0.361269,0.208626,-0.580824,-0.661976,chained,10 +5112,Horticultural trades,-0.693377,0.606937,-0.908001,0.113781,-0.900545,-0.861618,chained,6 +5113,Gardeners and landscape gardeners,-1.133886,0.569977,-1.427590,0.075026,-1.554993,-1.439157,chained,2 +5114,Groundsmen and greenkeepers,-0.821707,0.636204,-1.142359,0.102143,-1.329744,-1.339606,chained,1 +5119,Agricultural and fishing trades n.e.c.,-0.434348,0.638881,-0.596608,0.198072,-0.744175,-0.820281,chained,14 +5211,Sheet metal workers,-0.155223,0.532718,-0.222150,0.210884,-1.324247,-1.381543,chained,5 +5212,"Metal plate workers, smiths, moulders and related occupations",-0.741285,0.561395,-0.954311,0.068307,-1.364706,-1.399080,chained,8 +5213,Welding trades,-0.607508,0.528260,-0.751412,0.081943,-1.311061,-1.404629,chained,3 +5214,Pipe fitters,-0.763084,0.697726,-1.173415,0.032182,-1.359573,-1.373570,chained,1 +5221,Metal machining setters and setter-operators,-0.589943,0.544586,-0.724069,0.148671,-0.965098,-1.120640,chained,5 +5222,"Tool makers, tool fitters and markers-out",-0.442090,0.584572,-0.583198,0.154317,-0.721390,-0.926919,chained,1 +5223,Metal working production and maintenance fitters and technicians,-0.239190,0.618429,-0.300219,0.195485,-1.188610,-1.226573,chained,24 +5224,Precision instrument makers and repairers,-0.090857,0.565660,-0.113790,0.187156,-0.346009,-0.648839,chained,1 +5225,Air-conditioning and refrigeration installers and repairers,0.142326,0.643452,0.187846,0.299945,-0.695983,-0.779124,chained,3 +5231,"Vehicle technicians, mechanics and electricians",-0.182627,0.636792,-0.247175,0.209864,-1.102127,-1.144256,chained,3 +5232,Vehicle body builders and repairers,-0.380497,0.593496,-0.475147,0.160257,-1.247088,-1.343680,chained,4 +5233,Vehicle paint technicians,-0.797716,0.511701,-0.956559,0.048266,-1.133540,-1.251568,chained,4 +5234,Aircraft maintenance and related trades,-0.106349,0.645081,-0.148589,0.230859,-0.466599,-0.612549,chained,4 +5235,Boat and ship builders and repairers,-0.517777,0.610265,-0.702943,0.146661,-1.058217,-1.122029,chained,7 +5236,Rail and rolling stock builders and repairers,-0.194573,0.619490,-0.261819,0.216905,-1.260120,-1.367088,chained,4 +5241,Electricians and electrical fitters,-0.221746,0.639935,-0.325661,0.203650,-0.481898,-0.712030,chained,6 +5242,Telecoms and related network installers and repairers,-0.157926,0.647546,-0.265435,0.218856,-0.357824,-0.493158,chained,5 +5243,"TV, video and audio servicers and repairers",0.084717,0.605105,0.152356,0.285252,-0.536425,-0.690388,chained,4 +5244,Computer system and equipment installers and servicers,0.126154,0.596947,0.141926,0.266236,-0.792678,-0.885204,chained,2 +5245,Security system installers and repairers,0.294622,0.599304,0.371762,0.340772,-0.861559,-0.970066,chained,3 +5246,Electrical service and maintenance mechanics and repairers,-0.327484,0.642591,-0.484414,0.174865,-0.739673,-0.872644,chained,7 +5249,Electrical and electronic trades n.e.c.,0.079640,0.626645,0.082770,0.271432,-0.861559,-0.970066,chained,8 +5250,"Skilled metal, electrical and electronic trades supervisors",0.142702,0.705260,0.191585,0.283874,-0.421241,-0.460741,chained,3 +5311,Steel erectors,-0.096796,0.657512,-0.163128,0.231061,-1.861326,-1.777627,chained,2 +5312,Stonemasons and related trades,-0.983267,0.589630,-1.312442,0.036174,-1.539772,-1.489358,chained,3 +5313,Bricklayers,-0.950581,0.631498,-1.336344,0.046108,-1.539772,-1.489358,chained,3 +5314,"Roofers, roof tilers and slaters",-0.876317,0.593275,-1.177907,0.047137,-1.990227,-1.781385,chained,3 +5315,Plumbers and heating and ventilating installers and repairers,0.124273,0.654461,0.153388,0.279268,-0.998234,-1.069726,chained,4 +5316,Carpenters and joiners,-0.732497,0.604597,-0.994860,0.127022,-1.359765,-1.345317,chained,3 +5317,"Glaziers, window fabricators and fitters",-0.938488,0.574080,-1.240770,0.029338,-1.358998,-1.293978,chained,2 +5319,Construction and building trades n.e.c.,-0.545262,0.661905,-0.766499,0.149688,-1.361569,-1.323342,chained,11 +5321,Plasterers,-1.314157,0.550949,-1.648111,0.019765,-1.618616,-1.522564,chained,1 +5322,Floorers and wall tilers,-1.127317,0.590340,-1.489302,0.017514,-1.864359,-1.691052,chained,3 +5323,Painters and decorators,-1.167025,0.554826,-1.471101,0.012578,-1.639715,-1.412932,chained,2 +5330,Construction and building trades supervisors,-0.118370,0.728281,-0.194966,0.229167,-0.227171,-0.266829,chained,2 +5411,Upholsterers,-0.592221,0.596771,-0.699132,0.156618,-1.348594,-1.405584,chained,2 +5412,Footwear and leather working trades,-0.801851,0.520701,-0.946254,0.098471,-1.172688,-1.181424,chained,12 +5413,Tailors and dressmakers,-0.541345,0.536065,-0.635628,0.119554,-0.871603,-0.967225,chained,6 +5419,"Textiles, garments and related trades n.e.c.",-0.650344,0.500921,-0.731789,0.131690,-0.788567,-1.083108,chained,15 +5421,Pre-press technicians,0.307057,0.574080,0.382479,0.352831,-0.570299,-0.788152,chained,4 +5422,Printers,-0.152763,0.560243,-0.194553,0.204156,-0.938560,-1.064630,chained,4 +5423,Print finishing and binding workers,-0.196929,0.559717,-0.214505,0.181619,-0.524361,-0.644516,chained,7 +5431,Butchers,-0.540456,0.585650,-0.663796,0.107267,-1.088827,-1.049975,chained,3 +5432,Bakers and flour confectioners,-0.185887,0.614965,-0.234914,0.171324,-0.519567,-0.622332,chained,2 +5433,Fishmongers and poultry dressers,-0.963737,0.516354,-1.167537,0.023246,-1.389356,-1.287077,chained,1 +5434,Chefs,-0.176024,0.677812,-0.262596,0.258730,-0.382169,-0.278559,chained,1 +5435,Cooks,-0.324915,0.589683,-0.417004,0.175018,-0.695846,-0.592443,chained,3 +5436,Catering and bar managers,-0.370518,0.559310,-0.434050,0.205010,-0.364112,-0.176087,chained,5 +5441,"Glass and ceramics makers, decorators and finishers",-0.561198,0.545282,-0.663082,0.142596,-0.044507,-0.249720,chained,12 +5442,Furniture makers and other craft woodworkers,-0.683643,0.585636,-0.880624,0.109958,-1.259874,-1.339451,chained,9 +5443,Florists,-0.360811,0.544201,-0.434261,0.164583,-0.397831,-0.345463,chained,1 +5449,Other skilled trades n.e.c.,-0.645063,0.545023,-0.792201,0.103910,-1.183812,-1.189005,chained,28 +6111,Early education and childcare assistants,0.067655,0.642222,0.224200,0.303149,-0.145522,0.060582,chained,3 +6112,Teaching assistants,0.018489,0.639800,0.451664,0.266667,0.382332,0.697323,direct,1 +6113,Educational support assistants,0.018489,0.639800,0.451664,0.266667,0.382332,0.697323,direct,1 +6114,Childminders,0.002100,0.638993,0.023414,0.340218,-0.488411,-0.249403,chained,1 +6116,Nannies and au pairs,0.002100,0.638993,0.023414,0.340218,-0.488411,-0.249403,chained,1 +6117,Playworkers,0.002100,0.638993,0.023414,0.340218,-0.247325,-0.061294,chained,1 +6121,Pest controllers,-0.515878,0.642708,-0.721739,0.176726,-0.789192,-0.759387,chained,1 +6129,Animal care services occupations n.e.c.,-0.581643,0.615802,-0.773477,0.125620,-0.936853,-0.799018,chained,5 +6131,Nursing auxiliaries and assistants,-0.115437,0.602789,-0.189167,0.230744,0.232138,0.415287,chained,6 +6132,Ambulance staff (excluding paramedics),-0.564460,0.644236,-0.747311,0.178649,-0.789933,-0.798541,chained,3 +6133,Dental nurses,-0.397371,0.605291,-0.539586,0.096220,-0.602693,-0.479129,chained,2 +6134,Houseparents and residential wardens,-0.366535,0.608501,-0.481549,0.183779,-0.270401,0.074893,chained,2 +6135,Care workers and home carers,-0.172103,0.625857,-0.216387,0.235668,0.191017,0.431785,chained,5 +6136,Senior care workers,-0.221274,0.602593,-0.262804,0.237261,-0.290329,0.033874,chained,1 +6137,Care escorts,-0.076397,0.611316,-0.075347,0.242429,-0.397718,-0.003932,chained,3 +6138,"Undertakers, mortuary and crematorium assistants",-0.295134,0.629444,-0.408139,0.226030,-0.596539,-0.426632,chained,2 +6211,Sports and leisure assistants,-0.283210,0.597149,-0.382674,0.213564,-0.325083,-0.269417,chained,8 +6212,Travel agents,0.778030,0.596508,1.050303,0.513524,0.827626,0.907749,chained,2 +6213,Air travel assistants,0.216809,0.617176,0.281064,0.355330,-0.361381,0.007918,chained,2 +6214,Rail travel assistants,-0.527559,0.620903,-0.717463,0.195297,-0.820721,-0.671616,chained,3 +6219,Leisure and travel service occupations n.e.c.,-0.118110,0.599522,-0.139971,0.263768,0.277478,0.449077,chained,8 +6221,Hairdressers and barbers,-0.254504,0.626898,-0.327202,0.155383,-0.611426,-0.363162,chained,1 +6222,Beauticians and related occupations,-0.284904,0.591076,-0.362769,0.177690,-0.475559,-0.448633,chained,3 +6231,Housekeepers and related occupations,-0.536828,0.600101,-0.695115,0.171987,-0.995462,-0.630588,chained,5 +6232,Caretakers,-0.596535,0.592022,-0.761473,0.173380,-0.339213,-0.018990,chained,3 +6240,Cleaning and housekeeping managers and supervisors,-0.077515,0.628624,-0.078225,0.306297,-0.673412,-0.440891,chained,3 +6250,Bed and breakfast and guest house owners and proprietors,-0.082951,0.660660,-0.091620,0.285243,0.328336,0.390834,chained,2 +6311,Police community support officers,-0.161193,0.620992,-0.247868,0.269963,-0.913611,-0.877600,chained,1 +6312,Parking and civil enforcement occupations,-0.274123,0.560767,-0.356554,0.226112,-0.723563,-0.901236,chained,2 +7111,Sales and retail assistants,0.130369,0.504935,0.150494,0.330347,-0.456498,-0.140498,chained,5 +7112,Retail cashiers and check-out operators,-0.250282,0.574896,-0.343261,0.261737,-0.371147,-0.225160,chained,2 +7113,Telephone salespersons,0.917715,0.482517,1.056226,0.596925,1.101573,1.966413,chained,2 +7114,Pharmacy and optical dispensing assistants,0.027250,0.573935,0.035232,0.299558,-0.271074,-0.139931,chained,3 +7115,Vehicle and parts salespersons and advisers,-0.070758,0.523872,-0.082341,0.342027,-0.316943,-0.098232,chained,2 +7121,Collector salespersons and credit agents,0.909109,0.575101,1.193887,0.511422,1.254113,1.419171,chained,5 +7122,"Debt, rent and other cash collectors",0.098261,0.521641,0.125724,0.327538,-0.391963,-0.346098,chained,6 +7123,Roundspersons and van salespersons,-0.240819,0.524896,-0.329788,0.271717,-0.984919,-0.895287,chained,3 +7124,Market and street traders and assistants,0.948647,0.491528,1.107101,0.521739,1.087906,1.522033,chained,2 +7125,Visual merchandisers and related occupations,0.199707,0.582650,0.274937,0.467167,-0.071406,0.093473,chained,3 +7129,Sales related occupations n.e.c.,0.605460,0.581207,0.796374,0.504158,1.097178,1.181531,chained,18 +7131,Shopkeepers and owners - retail and wholesale,0.672724,0.601536,0.873296,0.456554,0.093316,0.126600,chained,9 +7132,Sales supervisors - retail and wholesale,0.458065,0.590000,0.570730,0.439266,-0.149595,0.002132,chained,3 +7211,Call and contact centre occupations,0.931224,0.484063,1.081240,0.569852,0.955437,1.229360,chained,4 +7212,Telephonists,0.890311,0.452274,0.993104,0.581095,0.983582,1.334419,chained,2 +7213,Communication operators,0.255890,0.543944,0.310827,0.347995,0.043358,-0.135396,chained,5 +7214,Market research interviewers,0.987185,0.587292,1.297031,0.431034,1.291962,1.451002,direct,1 +7219,Customer service occupations n.e.c.,0.609092,0.585297,0.794422,0.521641,0.835745,0.895722,chained,7 +7220,Customer service supervisors,0.790485,0.587361,1.038688,0.531250,1.131081,1.214532,direct,1 +8111,"Food, drink and tobacco process operatives",-0.729257,0.528671,-0.884460,0.072140,-1.002777,-1.100550,chained,8 +8112,Textile process operatives,-0.858112,0.518017,-1.028037,0.084558,-1.477617,-1.489753,chained,17 +8113,Chemical and related process operatives,-0.610049,0.553733,-0.752694,0.121908,-1.065505,-1.187028,chained,22 +8114,Plastics process operatives,-0.839096,0.524583,-1.017296,0.050248,-1.155429,-1.289644,chained,1 +8115,Metal making and treating process operatives,-0.725009,0.551079,-0.903043,0.086024,-1.225537,-1.304831,chained,21 +8119,Process operatives n.e.c.,-0.863383,0.523389,-1.047239,0.072718,-1.331774,-1.381164,chained,18 +8120,Metal working machine operatives,-0.579590,0.565382,-0.736401,0.103299,-1.036557,-1.188575,chained,12 +8131,Paper and wood machine operatives,-0.791306,0.528001,-0.956023,0.085977,-1.279321,-1.344430,chained,17 +8132,Mining and quarry workers and related operatives,-0.613111,0.628285,-0.835320,0.126907,-1.177599,-1.314137,chained,14 +8133,Energy plant operatives,-0.090550,0.630740,-0.107175,0.190088,-0.292999,-0.521266,chained,6 +8134,Water and sewerage plant operatives,0.129691,0.681736,0.170577,0.323721,-1.040783,-1.147964,chained,4 +8135,Printing machine assistants,-0.528624,0.519764,-0.625214,0.132899,-1.070649,-1.158230,chained,12 +8139,Plant and machine operatives n.e.c.,-0.645006,0.558914,-0.805423,0.110141,-1.239361,-1.312273,chained,48 +8141,Assemblers (electrical and electronic products),-0.658544,0.533196,-0.800615,0.080865,-1.011079,-1.068971,chained,6 +8142,Assemblers (vehicles and metal goods),-0.569577,0.561965,-0.705716,0.084566,-0.928136,-1.001906,chained,6 +8143,Routine inspectors and testers,-0.183244,0.614182,-0.223635,0.209357,-0.340496,-0.451531,chained,4 +8144,"Weighers, graders and sorters",-0.797678,0.523903,-0.949384,0.121801,-0.261197,-0.424089,chained,18 +8145,"Tyre, exhaust and windscreen fitters",-0.797345,0.660788,-1.163846,0.046676,-1.523370,-1.453395,chained,2 +8146,Sewing machinists,-0.769693,0.461351,-0.848605,0.096261,-1.383510,-1.503968,chained,10 +8149,Assemblers and routine operatives n.e.c.,-0.690835,0.549120,-0.847777,0.091053,-1.200152,-1.261250,chained,13 +8151,"Scaffolders, stagers and riggers",-0.728016,0.657627,-1.044385,0.090156,-1.412360,-1.440863,chained,5 +8152,Road construction operatives,-1.040517,0.625693,-1.433568,0.011263,-1.527677,-1.533671,chained,5 +8153,Rail construction and maintenance operatives,-0.055008,0.661594,-0.103492,0.263476,-1.312481,-1.528146,chained,4 +8159,Construction operatives n.e.c.,-0.576142,0.650281,-0.774301,0.134439,-1.211892,-1.146012,chained,22 +8160,"Production, factory and assembly supervisors",-0.724202,0.520505,-0.847122,0.082782,-1.077679,-1.159141,chained,16 +8211,Heavy and large goods vehicle drivers,-0.474989,0.629721,-0.644562,0.158077,-1.431423,-1.507835,chained,3 +8212,Bus and coach drivers,-0.286833,0.599005,-0.343019,0.214432,-0.589618,-1.070454,chained,1 +8213,Taxi and cab drivers and chauffeurs,-0.478271,0.621910,-0.633063,0.191525,-0.785609,-1.122001,chained,1 +8214,Delivery drivers and couriers,-0.539563,0.619062,-0.718290,0.213819,-1.064743,-1.154663,chained,3 +8215,Driving instructors,0.394362,0.599167,0.526352,0.380000,0.880152,1.073438,direct,1 +8219,Road transport drivers n.e.c.,-0.499213,0.615000,-0.673340,0.147709,-0.848833,-0.906938,chained,4 +8221,Crane drivers,-0.715829,0.600193,-0.934532,0.123642,-0.890077,-1.053312,chained,2 +8222,Fork-lift truck drivers,-0.413077,0.646910,-0.563624,0.206952,-1.609311,-1.808822,chained,2 +8229,Mobile machine drivers and operatives n.e.c.,-0.451982,0.641926,-0.633052,0.203425,-1.284335,-1.444911,chained,10 +8231,Train and tram drivers,-0.383880,0.615995,-0.498328,0.184575,-0.785062,-1.123684,chained,2 +8232,Marine and waterways transport operatives,-0.581327,0.640401,-0.805107,0.113572,-1.060713,-1.280269,chained,6 +8233,Air transport operatives,-0.215947,0.648556,-0.284189,0.262840,0.238968,0.106856,chained,4 +8234,Rail transport operatives,-0.401212,0.639290,-0.556439,0.148872,-0.938225,-1.077527,chained,3 +8239,Other drivers and transport operatives n.e.c.,-0.169743,0.643167,-0.240758,0.229991,-0.564743,-0.650408,chained,18 +9111,Farm workers,-0.970475,0.562335,-1.206602,0.085289,-1.470034,-1.427154,chained,4 +9112,Forestry and related workers,-0.617700,0.610547,-0.798782,0.151810,-0.725773,-0.787430,chained,8 +9119,Fishing and other elementary agriculture occupations n.e.c.,-0.806774,0.602014,-1.048158,0.088040,-1.383333,-1.349289,chained,14 +9121,Groundworkers,-1.097309,0.639390,-1.535208,0.001786,-1.909924,-1.731158,chained,2 +9129,Elementary construction occupations n.e.c.,-0.745045,0.638427,-1.027346,0.115940,-1.909924,-1.731158,chained,14 +9131,Industrial cleaning process occupations,-0.791988,0.620105,-1.068034,0.093906,-1.515767,-1.511079,chained,9 +9132,"Packers, bottlers, canners and fillers",-0.841324,0.606615,-1.097773,0.100000,-1.561286,-1.418529,chained,4 +9139,Elementary process plant occupations n.e.c.,-0.730181,0.617719,-0.975930,0.113475,-1.526799,-1.507060,chained,17 +9211,"Postal workers, mail sorters and messengers",-0.080638,0.562623,-0.138563,0.302321,-1.098220,-0.938727,chained,7 +9219,Elementary administration occupations n.e.c.,0.132810,0.534306,0.141722,0.380954,-0.818272,-0.798447,chained,8 +9221,Window cleaners,-1.142079,0.553194,-1.436201,0.027778,-1.648635,-1.286594,direct,1 +9222,Street cleaners,-1.002583,0.545514,-1.222928,0.099047,-1.480778,-1.491440,direct,1 +9223,Cleaners and domestics,-1.024118,0.516096,-1.193885,0.114264,-1.743035,-1.234860,chained,3 +9224,"Launderers, dry cleaners and pressers",-1.111354,0.479809,-1.268996,0.070751,-1.711704,-1.586496,chained,4 +9225,Refuse and salvage occupations,-0.450285,0.620245,-0.572829,0.272567,-1.251657,-1.299442,chained,4 +9226,Vehicle valeters and cleaners,-1.219589,0.533104,-1.502566,0.047615,-1.685921,-1.637117,chained,3 +9229,Elementary cleaning occupations n.e.c.,-1.068072,0.570637,-1.363088,0.061309,-1.407207,-1.468459,chained,3 +9231,Security guards and related occupations,-0.150267,0.608599,-0.246337,0.267213,-0.158760,-0.153056,chained,6 +9232,School midday and crossing patrol occupations,-0.161193,0.620992,0.101898,0.268315,-0.539271,-0.623683,chained,2 +9233,Exam invigilators,-0.155730,0.614795,0.451664,0.266667,0.902101,0.938013,direct,1 +9241,Shelf fillers,-0.650565,0.524167,-0.789295,0.190909,-0.975867,-0.608622,direct,1 +9249,Elementary sales occupations n.e.c.,-0.781718,0.535896,-0.971421,0.148702,-1.291688,-1.127109,chained,5 +9251,Elementary storage supervisors,-0.073722,0.610745,-0.087300,0.356971,-0.671358,-0.667145,chained,3 +9252,Warehouse operatives,-0.829567,0.552684,-1.032704,0.150498,-0.671358,-0.667145,chained,5 +9253,Delivery operatives,-0.737103,0.636771,-1.019044,0.180288,-0.671358,-0.667145,chained,2 +9259,Elementary storage occupations n.e.c.,-0.745152,0.672153,-1.093387,0.071312,-0.671358,-0.667145,chained,3 +9261,Bar and catering supervisors,-0.792364,0.508581,-0.933653,0.109547,-1.012211,-0.689404,chained,5 +9262,Hospital porters,-1.265695,0.511898,-1.515585,0.028867,-1.816307,-1.242406,chained,1 +9263,Kitchen and catering assistants,-0.832635,0.492543,-0.970218,0.061104,-1.012128,-0.766902,chained,4 +9264,Waiters and waitresses,-0.772986,0.556088,-0.932834,0.127351,-0.829287,-0.514621,chained,3 +9265,Bar staff,-0.611430,0.565127,-0.799835,0.133592,-1.378058,-1.038972,chained,2 +9266,Coffee shop workers,-0.692733,0.475660,-0.798444,0.107913,-1.012128,-0.766902,chained,2 +9267,Leisure and theme park attendants,-0.539486,0.587441,-0.724330,0.190334,-0.483446,-0.354409,chained,5 +9269,Other elementary services occupations n.e.c.,-0.477301,0.600217,-0.664118,0.201248,-0.509424,-0.331660,chained,15 diff --git a/packages/populace-build/src/populace/build/uk_runtime/ai_exposure_data/uk_soc2020_major_group_ai_exposure.csv b/packages/populace-build/src/populace/build/uk_runtime/ai_exposure_data/uk_soc2020_major_group_ai_exposure.csv new file mode 100644 index 00000000..211e6a33 --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/ai_exposure_data/uk_soc2020_major_group_ai_exposure.csv @@ -0,0 +1,10 @@ +soc2020_major_group,major_group_title,c_aioe,complementarity_theta,felten_aioe,eloundou_beta,dsit_aioe,dsit_llm,weighting,employment_jobs_thousands,n_unit_groups,n_unit_groups_weighted +1,"Managers, Directors And Senior Officials",0.618755,0.673570,0.900740,0.424035,0.942716,0.916797,ashe_2025_table14_jobs,2793.000000,42,29 +2,Professional Occupations,0.603795,0.645011,0.831678,0.429347,0.854552,0.845572,ashe_2025_table14_jobs,6981.000000,97,88 +3,Associate Professional Occupations,0.472267,0.624543,0.648432,0.427903,0.649218,0.627159,ashe_2025_table14_jobs,3754.000000,68,57 +4,Administrative And Secretarial Occupations,0.743671,0.538352,0.910434,0.514191,0.920936,0.941705,ashe_2025_table14_jobs,2585.000000,27,26 +5,Skilled Trades Occupations,-0.333348,0.625646,-0.443506,0.187874,-0.912533,-0.938844,ashe_2025_table14_jobs,1392.000000,57,38 +6,"Caring, Leisure And Other Service Occupations",-0.140396,0.619238,-0.105225,0.239852,0.022371,0.255044,ashe_2025_table14_jobs,2177.000000,29,26 +7,Sales And Customer Service Occupations,0.268931,0.532488,0.332671,0.384477,-0.079670,0.152617,ashe_2025_table14_jobs,1506.000000,19,14 +8,"Process, Plant And Machine Operatives",-0.539356,0.596294,-0.693827,0.145015,-1.074432,-1.183658,ashe_2025_table14_jobs,1231.000000,39,32 +9,Elementary Occupations,-0.729302,0.549713,-0.878454,0.146167,-1.069290,-0.869221,ashe_2025_table14_jobs,2156.000000,34,30 diff --git a/packages/populace-build/src/populace/build/uk_runtime/ai_exposure_sources.py b/packages/populace-build/src/populace/build/uk_runtime/ai_exposure_sources.py new file mode 100644 index 00000000..b61e91fd --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/ai_exposure_sources.py @@ -0,0 +1,680 @@ +"""Official-source builders for the packaged UK AI-exposure tables. + +The lookup module (:mod:`populace.build.uk_runtime.ai_exposure`) ships +pre-built CSVs; this module records where those CSVs come from and provides +``build_*`` functions that reconstruct them from pinned public sources, +following the pattern of :mod:`populace.build.uk_runtime.geography_sources` +(pinned source-URL constants plus composable builders). + +Derivation chain (see the ai_exposure module docstring for citations): + +1. ``build_complementarity_theta`` reconstructs the Pizzinelli et al. (2023, + IMF WP/23/216) AI complementarity score theta from the O*NET 27.3 + database (11 work contexts plus job zones grouped into six components). +2. ``build_us_measure_chain`` chains US-SOC-based measures (Felten AIOE, + Eloundou et al. beta, theta) US SOC 2018 -> US SOC 2010 -> ISCO-08 -> + UK SOC 2020 via the BLS crosswalks and the ONS SOC 2020 Volume 2 coding + index. +3. ``build_dsit_soc2010_to_soc2020`` maps the DSIT/DfE (Nov 2023) Annex 1 + scores from SOC 2010 to SOC 2020. +4. ``build_ai_exposure_table`` composes 1-3 into the packaged unit-group + (4-digit) table, imputing missing unit groups from parent means. +5. ``build_major_group_ai_exposure`` aggregates the unit-group table to + 1-digit major groups, weighted by packaged ASHE Table 14 employee jobs. + +Functions 1-4 download pinned sources at call time and are faithful +reconstructions of the documented derivation; steps whose exact +specification is not recoverable from the shipped documentation carry TODO +comments instead of silent guesses. Function 5 runs entirely from packaged +data and is used to regenerate ``uk_soc2020_major_group_ai_exposure.csv``. +""" + +from __future__ import annotations + +import io +import time +import urllib.error +import urllib.request +from importlib import resources as importlib_resources +from pathlib import Path +from typing import Any + +import numpy as np +import pandas as pd + +from populace.build.uk_runtime.ai_exposure import ( + MEASURE_TO_COLUMN, + load_ai_exposure_table, +) + +# -------------------------------------------------------------------------- +# Pinned source URLs +# -------------------------------------------------------------------------- + +#: O*NET 27.3 database (US Department of Labor, CC BY 4.0) — the vintage used +#: to reconstruct the IMF WP/23/216 complementarity score theta. +ONET_27_3_WORK_CONTEXT_URL = ( + "https://www.onetcenter.org/dl_files/database/db_27_3_text/Work%20Context.txt" +) +ONET_27_3_JOB_ZONES_URL = ( + "https://www.onetcenter.org/dl_files/database/db_27_3_text/Job%20Zones.txt" +) + +#: BLS US SOC 2010 <-> 2018 crosswalk (Nov 2017 release, US public domain). +BLS_SOC2010_SOC2018_CROSSWALK_URL = ( + "https://www.bls.gov/soc/2018/soc_2010_to_2018_crosswalk.xlsx" +) + +#: BLS ISCO-08 <-> US SOC 2010 crosswalk (2012, US public domain). +BLS_ISCO08_SOC2010_CROSSWALK_URL = ( + "https://www.bls.gov/soc/soccrosswalks/isco_soc_crosswalk.xls" +) + +#: ONS SOC 2020 Volume 2 coding index (OGL v3.0). Every coding-index entry +#: carries both a SOC 2020 unit group and an ISCO-08 code, and SOC 2010 +#: codes, providing both the ISCO-08 -> SOC 2020 and SOC 2010 -> SOC 2020 +#: links used below. +ONS_SOC2020_VOLUME2_CODING_INDEX_URL = ( + "https://www.ons.gov.uk/file?uri=/methodology/classificationsandstandards/" + "standardoccupationalclassificationsoc/soc2020/soc2020volume2codingrules" + "andconventions/soc2020volume2thecodingindexandcodingrulesandconventions" + "excel180523.xlsx" +) + +#: DSIT/DfE (Nov 2023) "The impact of AI on UK jobs and training", Annex 1 +#: (OGL v3.0): standardised AIOE and LLM exposure on SOC 2010 4-digit codes. +DSIT_2023_ANNEX1_URL = ( + "https://assets.publishing.service.gov.uk/media/6552862d046ed400148b99fe/" + "Annex_A_The_Impact_of_AI_on_UK_Jobs_and_Training.xlsx" +) + +#: Felten, Raj and Seamans (2021) AI Occupational Exposure, public GitHub +#: release accompanying the paper. +FELTEN_AIOE_URL = ( + "https://raw.githubusercontent.com/AIOE-Data/AIOE/main/Appendix%20file/" + "Appendix%20A%20AIOE.csv" +) + +#: Eloundou et al. (2023) "GPTs are GPTs" occupation-level exposure (MIT +#: licence); ``beta`` is the human-annotated share of tasks where LLMs can +#: halve completion time (direct or with tooling). +ELOUNDOU_GPTS_ARE_GPTS_URL = ( + "https://raw.githubusercontent.com/openai/GPTs-are-GPTs/main/" + "occ_lvl_exposure_data.csv" +) + +#: IMF WP/23/216 published extremes of the complementarity score theta, +#: used to validate the O*NET reconstruction. +THETA_MIN_SOC2018 = "51-9031" # Cutters and trimmers, hand. +THETA_MAX_SOC2018 = "29-1022" # Oral and maxillofacial surgeons. + +#: The six theta components of IMF WP/23/216: 11 O*NET Work Context elements +#: (CX scale, 1-5) plus the O*NET Job Zone (1-5), each normalised to [0, 1]. +#: Elements listed with ``reverse=True`` enter as ``1 - value`` (more routine +#: or automated work is less AI-complementary). +#: TODO: The exact element list is reconstructed from the WP/23/216 annex +#: description ("11 work contexts plus job zones grouped into six +#: components"); verify each element name against the paper's annex table +#: before treating individual component scores as canonical. The composite +#: reproduces the published extremes (asserted below). +THETA_WORK_CONTEXT_COMPONENTS: dict[str, tuple[tuple[str, bool], ...]] = { + "communication": ( + ("Public Speaking", False), + ("Face-to-Face Discussions", False), + ("Contact With Others", False), + ), + "responsibility": ( + ("Responsibility for Outcomes and Results", False), + ("Responsible for Others' Health and Safety", False), + ), + "physical_conditions": (("Physical Proximity", False),), + "criticality": ( + ("Consequence of Error", False), + ("Impact of Decisions on Co-workers or Company Results", False), + ("Frequency of Decision Making", False), + ), + "routine": ( + ("Degree of Automation", True), + ("Structured versus Unstructured Work", False), + ), +} + +#: Packaged ASHE Table 14 tidy CSV used to weight the major-group table +#: (see populace.build.uk_runtime.occupation_targets). +PACKAGED_ASHE_TABLE14_CSV = "ashe_table14_2025_soc4.csv" +MAJOR_GROUP_WEIGHTING_LABEL = "ashe_2025_table14_jobs" + +SOC2020_MAJOR_GROUP_TITLES = { + "1": "Managers, Directors And Senior Officials", + "2": "Professional Occupations", + "3": "Associate Professional Occupations", + "4": "Administrative And Secretarial Occupations", + "5": "Skilled Trades Occupations", + "6": "Caring, Leisure And Other Service Occupations", + "7": "Sales And Customer Service Occupations", + "8": "Process, Plant And Machine Operatives", + "9": "Elementary Occupations", +} + +MAJOR_GROUP_TABLE_COLUMNS = ( + "soc2020_major_group", + "major_group_title", + "c_aioe", + "complementarity_theta", + "felten_aioe", + "eloundou_beta", + "dsit_aioe", + "dsit_llm", + "weighting", + "employment_jobs_thousands", + "n_unit_groups", + "n_unit_groups_weighted", +) + + +# -------------------------------------------------------------------------- +# 1. Complementarity theta from O*NET 27.3 (IMF WP/23/216 recipe) +# -------------------------------------------------------------------------- + + +def build_complementarity_theta( + work_context: pd.DataFrame | None = None, + job_zones: pd.DataFrame | None = None, +) -> pd.DataFrame: + """Reconstruct the IMF WP/23/216 complementarity score theta. + + Six components — five from 11 O*NET Work Context elements (CX scale, + normalised from 1-5 to [0, 1]) and one from the O*NET Job Zone — are + averaged into theta in [0, 1] per US SOC 2018 occupation (O*NET-SOC + 8-digit codes are collapsed to 6-digit SOC by unweighted mean). + + The reconstruction is validated against the paper's published extremes: + the minimum must fall at US SOC 51-9031 (cutters and trimmers, hand) and + the maximum at 29-1022 (oral and maxillofacial surgeons). + + Args: + work_context: O*NET 27.3 Work Context table (downloaded from + :data:`ONET_27_3_WORK_CONTEXT_URL` when omitted). + job_zones: O*NET 27.3 Job Zones table (downloaded from + :data:`ONET_27_3_JOB_ZONES_URL` when omitted). + + Returns: + DataFrame with columns ``soc2018_code``, ``complementarity_theta`` + and one column per component. + """ + + if work_context is None: + work_context = _read_table_url(ONET_27_3_WORK_CONTEXT_URL) + if job_zones is None: + job_zones = _read_table_url(ONET_27_3_JOB_ZONES_URL) + + context = work_context[work_context["Scale ID"] == "CX"].copy() + context["soc2018_code"] = context["O*NET-SOC Code"].str[:7] + # CX scale runs 1-5; normalise to [0, 1]. + context["value"] = (pd.to_numeric(context["Data Value"]) - 1.0) / 4.0 + element_means = ( + context.groupby(["soc2018_code", "Element Name"])["value"] + .mean() + .unstack("Element Name") + ) + + components = pd.DataFrame(index=element_means.index) + for component, elements in THETA_WORK_CONTEXT_COMPONENTS.items(): + parts = [] + for element_name, reverse in elements: + if element_name not in element_means.columns: + raise ValueError( + f"O*NET Work Context is missing element {element_name!r} " + f"needed for theta component {component!r}." + ) + values = element_means[element_name] + parts.append(1.0 - values if reverse else values) + components[component] = pd.concat(parts, axis=1).mean(axis=1) + + zones = job_zones.copy() + zones["soc2018_code"] = zones["O*NET-SOC Code"].str[:7] + # Job zones run 1-5; normalise to [0, 1]. + components["skills"] = ( + zones.groupby("soc2018_code")["Job Zone"] + .mean() + .sub(1.0) + .div(4.0) + .reindex(components.index) + ) + + # TODO: WP/23/216 does not publish whether the six components are + # weighted; an unweighted mean reproduces the published extremes. + theta = components.mean(axis=1).dropna() + + observed_min = theta.idxmin() + observed_max = theta.idxmax() + if observed_min != THETA_MIN_SOC2018 or observed_max != THETA_MAX_SOC2018: + raise AssertionError( + "Reconstructed theta extremes do not match IMF WP/23/216: " + f"expected min at {THETA_MIN_SOC2018} and max at " + f"{THETA_MAX_SOC2018}, found min at {observed_min} and max at " + f"{observed_max}." + ) + + result = components.copy() + result["complementarity_theta"] = theta + return result.reset_index().rename(columns={"index": "soc2018_code"}) + + +# -------------------------------------------------------------------------- +# 2. US measure chain: US SOC 2018 -> US SOC 2010 -> ISCO-08 -> UK SOC 2020 +# -------------------------------------------------------------------------- + + +def build_dsit_soc2010_to_soc2020( + dsit_annex1: pd.DataFrame | None = None, + coding_index: pd.DataFrame | None = None, +) -> pd.DataFrame: + """Map DSIT Annex 1 exposure scores from SOC 2010 to SOC 2020. + + The ONS SOC 2020 Volume 2 coding index lists each index entry with both + its SOC 2010 and SOC 2020 unit group; the SOC 2010 -> SOC 2020 mapping + is the set of (2010, 2020) pairs, with many-to-many links resolved as + unweighted means over the linked SOC 2010 scores. + + Args: + dsit_annex1: DSIT Annex 1 table with SOC 2010 codes and the + standardised AIOE and LLM exposure columns (downloaded from + :data:`DSIT_2023_ANNEX1_URL` when omitted). + coding_index: ONS SOC 2020 Volume 2 coding index (downloaded from + :data:`ONS_SOC2020_VOLUME2_CODING_INDEX_URL` when omitted). + + Returns: + DataFrame with columns ``soc2020_code``, ``dsit_aioe``, ``dsit_llm``. + """ + + if dsit_annex1 is None: + # TODO: Verify the Annex 1 sheet name and header row against the + # published workbook; gov.uk annexes commonly carry cover sheets. + dsit_annex1 = _read_excel_url(DSIT_2023_ANNEX1_URL, dtype=str) + if coding_index is None: + # TODO: Verify the coding-index sheet name and the SOC 2010/SOC 2020 + # column headers against the published ONS workbook. + coding_index = _read_excel_url(ONS_SOC2020_VOLUME2_CODING_INDEX_URL, dtype=str) + + annex = _normalise_columns( + dsit_annex1, + { + "soc2010_code": ("soc2010", "soc 2010", "soc2010 code", "soc code"), + "dsit_aioe": ("aioe", "ai occupational exposure"), + "dsit_llm": ("llm", "llm exposure", "large language model"), + }, + label="DSIT Annex 1", + ) + annex["soc2010_code"] = annex["soc2010_code"].str.strip().str[:4] + annex["dsit_aioe"] = pd.to_numeric(annex["dsit_aioe"]) + annex["dsit_llm"] = pd.to_numeric(annex["dsit_llm"]) + + index = _normalise_columns( + coding_index, + { + "soc2010_code": ("soc2010", "soc 2010"), + "soc2020_code": ("soc2020", "soc 2020"), + }, + label="ONS SOC 2020 coding index", + ) + pairs = ( + index.assign( + soc2010_code=index["soc2010_code"].str.strip().str[:4], + soc2020_code=index["soc2020_code"].str.strip().str[:4], + ) + .query("soc2010_code.str.fullmatch('\\\\d{4}')") + .query("soc2020_code.str.fullmatch('\\\\d{4}')") + .drop_duplicates(["soc2010_code", "soc2020_code"]) + ) + mapped = pairs.merge(annex, on="soc2010_code", how="inner") + return ( + mapped.groupby("soc2020_code")[["dsit_aioe", "dsit_llm"]].mean().reset_index() + ) + + +def build_us_measure_chain( + theta: pd.DataFrame | None = None, + felten_aioe: pd.DataFrame | None = None, + eloundou: pd.DataFrame | None = None, + soc2010_soc2018: pd.DataFrame | None = None, + isco08_soc2010: pd.DataFrame | None = None, + coding_index: pd.DataFrame | None = None, +) -> pd.DataFrame: + """Chain US-SOC-based measures onto UK SOC 2020 unit groups. + + US SOC 2018 scores (theta, Felten AIOE, Eloundou beta) are chained + US SOC 2018 -> US SOC 2010 (BLS crosswalk, Nov 2017) -> ISCO-08 (BLS + ISCO-08/SOC 2010 crosswalk, 2012) -> SOC 2020 (ONS SOC 2020 Volume 2 + coding index, which assigns an ISCO-08 code to every SOC 2020 index + entry). Many-to-many links are resolved with employment-unweighted + means; ``mapping_quality`` is ``direct`` for single-path links and + ``chained`` otherwise. + + Returns: + DataFrame with columns ``soc2020_code``, ``complementarity_theta``, + ``felten_aioe``, ``eloundou_beta``, ``mapping_quality``, + ``n_isco_links``. + """ + + if theta is None: + theta = build_complementarity_theta() + if felten_aioe is None: + # TODO: Verify the AIOE column header in the AIOE-Data/AIOE release. + felten_aioe = _read_csv_url(FELTEN_AIOE_URL) + if eloundou is None: + # TODO: Verify the occupation-level beta column header in the + # openai/GPTs-are-GPTs release. + eloundou = _read_csv_url(ELOUNDOU_GPTS_ARE_GPTS_URL) + if soc2010_soc2018 is None: + soc2010_soc2018 = _read_excel_url(BLS_SOC2010_SOC2018_CROSSWALK_URL, dtype=str) + if isco08_soc2010 is None: + isco08_soc2010 = _read_excel_url(BLS_ISCO08_SOC2010_CROSSWALK_URL, dtype=str) + if coding_index is None: + coding_index = _read_excel_url(ONS_SOC2020_VOLUME2_CODING_INDEX_URL, dtype=str) + + crosswalk_2018_2010 = _normalise_columns( + soc2010_soc2018, + { + "soc2018_code": ("2018 soc code",), + "soc2010_code": ("2010 soc code",), + }, + label="BLS SOC 2010/2018 crosswalk", + ).drop_duplicates() + crosswalk_isco = _normalise_columns( + isco08_soc2010, + { + "isco08_code": ("isco-08 code", "isco08", "isco 08 code"), + "soc2010_code": ("2010 soc code", "soc2010", "soc code"), + }, + label="BLS ISCO-08/SOC 2010 crosswalk", + ).drop_duplicates() + index = _normalise_columns( + coding_index, + { + "soc2020_code": ("soc2020", "soc 2020"), + "isco08_code": ("isco08", "isco-08", "isco 08"), + }, + label="ONS SOC 2020 coding index", + ).drop_duplicates(["soc2020_code", "isco08_code"]) + + # Merge US SOC 2018 scores. + # TODO: Column names for the Felten and Eloundou releases below follow + # the papers' descriptions; verify against the downloaded files. + scores = theta[["soc2018_code", "complementarity_theta"]].copy() + felten = _normalise_columns( + felten_aioe, + {"soc2018_code": ("soc", "soc code"), "felten_aioe": ("aioe",)}, + label="Felten AIOE", + ) + gpts = _normalise_columns( + eloundou, + {"soc2018_code": ("soc", "soc code"), "eloundou_beta": ("beta",)}, + label="Eloundou GPTs-are-GPTs", + ) + scores = scores.merge(felten, on="soc2018_code", how="outer").merge( + gpts, on="soc2018_code", how="outer" + ) + measures = ["complementarity_theta", "felten_aioe", "eloundou_beta"] + for column in measures: + scores[column] = pd.to_numeric(scores[column]) + + # US SOC 2018 -> 2010 -> ISCO-08 -> SOC 2020, unweighted means per step. + on_2010 = ( + scores.merge(crosswalk_2018_2010, on="soc2018_code", how="inner") + .groupby("soc2010_code")[measures] + .mean() + .reset_index() + ) + on_isco = ( + on_2010.merge(crosswalk_isco, on="soc2010_code", how="inner") + .groupby("isco08_code") + .agg({**{m: "mean" for m in measures}, "soc2010_code": "nunique"}) + .rename(columns={"soc2010_code": "n_soc2010_links"}) + .reset_index() + ) + linked = index.merge(on_isco, on="isco08_code", how="inner") + result = ( + linked.groupby("soc2020_code") + .agg( + { + **{m: "mean" for m in measures}, + "isco08_code": "nunique", + "n_soc2010_links": "max", + } + ) + .rename(columns={"isco08_code": "n_isco_links"}) + .reset_index() + ) + result["mapping_quality"] = np.where( + (result["n_isco_links"] == 1) & (result["n_soc2010_links"] == 1), + "direct", + "chained", + ) + return result.drop(columns=["n_soc2010_links"]) + + +# -------------------------------------------------------------------------- +# 3. The packaged unit-group table +# -------------------------------------------------------------------------- + + +def build_ai_exposure_table( + us_chain: pd.DataFrame | None = None, + dsit: pd.DataFrame | None = None, + all_soc2020_codes: pd.Series | None = None, +) -> pd.DataFrame: + """Compose the packaged SOC 2020 unit-group AI-exposure table. + + Combines the chained US measures with the DSIT SOC 2010 -> SOC 2020 + scores, computes the composite ``c_aioe = felten_aioe * + (1 - (theta - theta_min))`` with ``theta_min`` taken from the theta + reconstruction, and fills unit groups missing any measure from 3-digit + (then 2-digit) sibling means, flagged ``imputed-from-parent``. + + Args: + us_chain: Output of :func:`build_us_measure_chain`. + dsit: Output of :func:`build_dsit_soc2010_to_soc2020`. + all_soc2020_codes: Complete list of SOC 2020 unit groups (defaults + to the packaged table's index so coverage matches the shipped + 412 unit groups). + + Returns: + DataFrame indexed like the packaged + ``uk_soc2020_ai_exposure.csv`` (one row per SOC 2020 unit group). + """ + + if us_chain is None: + us_chain = build_us_measure_chain() + if dsit is None: + dsit = build_dsit_soc2010_to_soc2020() + if all_soc2020_codes is None: + # TODO: Derive the canonical 412 unit groups from the ONS SOC 2020 + # Volume 1 structure instead of the packaged table when rebuilding + # from scratch. + all_soc2020_codes = load_ai_exposure_table().index.to_series() + + table = ( + pd.DataFrame({"soc2020_code": all_soc2020_codes.astype(str).values}) + .merge(us_chain, on="soc2020_code", how="left") + .merge(dsit, on="soc2020_code", how="left") + .set_index("soc2020_code") + .sort_index() + ) + theta_min = table["complementarity_theta"].min() + table["c_aioe"] = table["felten_aioe"] * ( + 1.0 - (table["complementarity_theta"] - theta_min) + ) + + measure_columns = sorted(set(MEASURE_TO_COLUMN.values())) + missing_any = table[measure_columns].isna().any(axis=1) + for length in (3, 2): + still_missing = table[measure_columns].isna().any(axis=1) + if not still_missing.any(): + break + prefix = table.index.str[:length] + parent_means = table[measure_columns].groupby(prefix).transform("mean") + table[measure_columns] = table[measure_columns].fillna(parent_means) + table.loc[missing_any, "mapping_quality"] = "imputed-from-parent" + return table + + +# -------------------------------------------------------------------------- +# 4. Major-group aggregation (runs from packaged data) +# -------------------------------------------------------------------------- + + +def build_major_group_ai_exposure( + ashe_csv: str | Path | None = None, + *, + weighting_label: str = MAJOR_GROUP_WEIGHTING_LABEL, +) -> pd.DataFrame: + """ASHE-employment-weighted major-group (1-digit) AI-exposure table. + + Aggregates the packaged 4-digit unit-group exposure table to SOC 2020 + major groups 1-9, weighting each unit group by its ASHE Table 14 + employee-job count. Unit groups whose ASHE cell is suppressed (blank in + the tidy CSV) carry no weight and are excluded from the weighted means. + + Args: + ashe_csv: Tidy ASHE Table 14 CSV with ``soc_code`` and + ``employment_jobs`` columns. Defaults to the packaged + ``occupation_targets_data/ashe_table14_2025_soc4.csv``. + weighting_label: Value recorded in the output ``weighting`` column. + + Returns: + DataFrame with one row per major group, matching the packaged + ``uk_soc2020_major_group_ai_exposure.csv`` schema. + """ + + if ashe_csv is None: + resource = importlib_resources.files("populace.build.uk_runtime").joinpath( + "occupation_targets_data", PACKAGED_ASHE_TABLE14_CSV + ) + with importlib_resources.as_file(resource) as path: + ashe = pd.read_csv(path, dtype={"soc_code": str}) + else: + ashe = pd.read_csv(ashe_csv, dtype={"soc_code": str}) + if not {"soc_code", "employment_jobs"}.issubset(ashe.columns): + raise ValueError( + "ASHE CSV must include 'soc_code' and 'employment_jobs' columns." + ) + ashe = ashe.copy() + ashe["soc_code"] = ashe["soc_code"].astype(str).str.strip() + ashe = ashe[ashe["soc_code"].str.fullmatch(r"\d{4}")] + ashe["employment_jobs"] = pd.to_numeric(ashe["employment_jobs"], errors="coerce") + weights = ashe.set_index("soc_code")["employment_jobs"] + + table = load_ai_exposure_table().copy() + table["weight"] = weights.reindex(table.index) + table["major_group"] = table.index.str[0] + + measure_columns = [ + "c_aioe", + "complementarity_theta", + "felten_aioe", + "eloundou_beta", + "dsit_aioe", + "dsit_llm", + ] + rows = [] + for group in sorted(table["major_group"].unique()): + subset = table[table["major_group"] == group] + weighted = subset[subset["weight"].notna() & (subset["weight"] > 0)] + if weighted.empty: + raise ValueError(f"Major group {group} has no ASHE-weighted unit groups.") + row: dict[str, Any] = { + "soc2020_major_group": group, + "major_group_title": SOC2020_MAJOR_GROUP_TITLES.get(group, ""), + } + for column in measure_columns: + row[column] = float( + np.average(weighted[column], weights=weighted["weight"]) + ) + row["weighting"] = weighting_label + row["employment_jobs_thousands"] = float(weighted["weight"].sum()) / 1000.0 + row["n_unit_groups"] = int(len(subset)) + row["n_unit_groups_weighted"] = int(len(weighted)) + rows.append(row) + return pd.DataFrame(rows, columns=list(MAJOR_GROUP_TABLE_COLUMNS)) + + +def write_major_group_ai_exposure(table: pd.DataFrame, path: str | Path) -> None: + """Write the major-group table to CSV in the packaged format.""" + + output = Path(path) + output.parent.mkdir(parents=True, exist_ok=True) + table.to_csv(output, index=False, float_format="%.6f") + + +# -------------------------------------------------------------------------- +# Shared helpers +# -------------------------------------------------------------------------- + + +def _normalise_columns( + frame: pd.DataFrame, + targets: dict[str, tuple[str, ...]], + *, + label: str, +) -> pd.DataFrame: + """Rename fuzzy source headers to canonical names; keep only those.""" + + lower_to_column = {str(column).strip().lower(): column for column in frame.columns} + rename: dict[str, str] = {} + for canonical, candidates in targets.items(): + found = None + if canonical in frame.columns: + found = canonical + else: + for candidate in candidates: + for lowered, column in lower_to_column.items(): + if candidate in lowered: + found = column + break + if found is not None: + break + if found is None: + raise ValueError(f"{label} is missing a column matching {canonical!r}.") + rename[found] = canonical + return frame.rename(columns=rename)[list(targets)].copy() + + +def _read_url_bytes( + url: str, + *, + timeout: int = 300, + retries: int = 3, + retry_delay: float = 1.0, +) -> bytes: + request = urllib.request.Request( + url, + headers={"User-Agent": "PolicyEngine-Populace/0.1"}, + ) + for attempt in range(retries): + try: + with urllib.request.urlopen(request, timeout=timeout) as response: + return response.read() + except urllib.error.HTTPError as error: + should_retry = error.code in {429, 500, 502, 503, 504} + if not should_retry or attempt == retries - 1: + raise + except (urllib.error.URLError, OSError, TimeoutError): + if attempt == retries - 1: + raise + time.sleep(retry_delay * (2**attempt)) + raise RuntimeError(f"Could not download {url}.") + + +def _read_csv_url(url: str, **kwargs: Any) -> pd.DataFrame: + return pd.read_csv(io.BytesIO(_read_url_bytes(url)), **kwargs) + + +def _read_table_url(url: str, **kwargs: Any) -> pd.DataFrame: + """Read a tab-delimited O*NET database text file.""" + + return pd.read_csv(io.BytesIO(_read_url_bytes(url)), sep="\t", **kwargs) + + +def _read_excel_url(url: str, **kwargs: Any) -> pd.DataFrame: + return pd.read_excel(io.BytesIO(_read_url_bytes(url)), **kwargs) diff --git a/packages/populace-build/src/populace/build/uk_runtime/ai_shock_runner.py b/packages/populace-build/src/populace/build/uk_runtime/ai_shock_runner.py new file mode 100644 index 00000000..0e34f5d3 --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/ai_shock_runner.py @@ -0,0 +1,761 @@ +"""End-to-end driver for ESRI-style UK AI shock scenario analysis. + +Pipeline (modeled on :mod:`populace.build.uk_runtime.local_runner`'s runner +conventions — frozen result dataclasses, lazy ``policyengine_uk`` imports +behind the ``uk`` extra, JSON/CSV writers): + +1. Load a PolicyEngine-UK single-year H5 dataset + (:func:`~populace.build.uk_runtime.local_runner.load_uk_dataset`). +2. Extract the person shock table (age, employment/self-employment income, + interest/dividend income, person weight) from a baseline simulation and + attach AI exposure — either draws from + :mod:`~populace.build.uk_runtime.exposure_imputation`, or the zero-model + crosswalk baseline via + :func:`~populace.build.uk_runtime.ai_exposure.exposure_for_major_group` + from an observed ``soc_major_group`` column (FRS ``1000``-``9000`` + coding, normalised with + :func:`~populace.build.uk_runtime.frs_occupation.frs_major_group_to_digit`). +3. Run scenarios — the :data:`~populace.build.uk_runtime.ai_shock_scenarios.PRESETS` + and/or the ESRI JR16 robustness grid (:func:`build_scenario_grid`: + employment loss 1-10% crossed with wage +1-5%, the +0.4pp capital-return + shock always on) — through + :func:`~populace.build.uk_runtime.ai_shock_scenarios.run_scenario`. +4. Write the shocked person table back into a fresh ``policyengine_uk`` + Simulation with ``set_input``: ``employment_income`` (zero for displaced, + complementarity-graded uplift for the rest), ``savings_interest_income`` + and ``dividend_income`` (capital-return uplift). + + **Displaced-worker expressibility note.** ESRI treat displaced workers as + long-term unemployed. PolicyEngine-UK exposes ``employment_status`` as an + input enum (set to ``UNEMPLOYED`` for displaced persons here), but the + full ESRI contract — 9+ months unemployed, contributory benefits + exhausted — is **only partially expressible**: PE-UK models entitlement + from current-period inputs and does not carry an unemployment-duration + input, so contributory JSA/ESA exhaustion cannot be encoded. The runner + sets ``employment_income`` and ``employment_status`` and records in the + result whether the status input was accepted + (``employment_status_applied``); benefit-exhaustion nuances beyond that + are a documented scope limitation. +5. Compute deltas versus baseline: Exchequer cost (change in ``gov_balance``), + poverty rates (BHC and AHC, as the installed PE-UK model exposes them), + and the Gini of equivalised household disposable income — overall, by + baseline income decile (the true JR16 replication view: deciles are fixed + at their baseline assignment), and by age band (the extension, using + :data:`~populace.build.uk_runtime.ai_shock_scenarios.AGE_BANDS`). + +Everything summary-mathematical (:func:`gini`, :func:`weighted_mean`, +:func:`decile_ids`, :func:`decile_deltas`, :func:`age_band_deltas`, +:func:`build_scenario_grid`, the dataclasses and writers) is a pure function +importable and unit-testable without ``policyengine_uk`` installed or real +data; only :func:`run_ai_shock_analysis` / :func:`run_grid` touch the engine. +""" + +from __future__ import annotations + +import json +from collections.abc import Callable, Mapping, Sequence +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any + +import numpy as np +import pandas as pd + +from populace.build.uk_runtime.ai_shock_scenarios import ( + AGE_BANDS, + DEFAULT_CAPITAL_COLUMNS, + DEFAULT_WEIGHT_COLUMN, + DISPLACED_COLUMN, + EMPLOYMENT_INCOME_COLUMN, + PRESETS, + ShockScenario, + run_scenario, +) +from populace.build.uk_runtime.local_runner import load_uk_dataset + +__all__ = [ + "AgeBandDelta", + "DecileDelta", + "GRID_CAPITAL_RETURN_INCREASE", + "GRID_DISPLACEMENT_RATES", + "GRID_WAGE_UPLIFTS", + "PopulationMetrics", + "ScenarioResult", + "age_band_deltas", + "build_scenario_grid", + "decile_deltas", + "decile_ids", + "gini", + "run_ai_shock_analysis", + "run_grid", + "scenario_result_from_dict", + "weighted_mean", + "write_grid_csv", + "write_grid_json", +] + +#: JR16 robustness grid margins: employment displacement 1-10% crossed with +#: wage changes of +1-5%; the +0.4pp return-to-capital shock is always on. +GRID_DISPLACEMENT_RATES = tuple(rate / 100 for rate in range(1, 11)) +GRID_WAGE_UPLIFTS = tuple(uplift / 100 for uplift in range(1, 6)) +GRID_CAPITAL_RETURN_INCREASE = 0.004 + +#: Person-level inputs written back into the shocked simulation. +SHOCKED_INPUT_COLUMNS = (EMPLOYMENT_INCOME_COLUMN, *DEFAULT_CAPITAL_COLUMNS) + + +# ---------------------------------------------------------------------- +# Pure summary math (no engine, no data files) +# ---------------------------------------------------------------------- + + +def weighted_mean(values: np.ndarray, weights: np.ndarray) -> float: + """Weighted mean, 0.0 on zero total weight.""" + + values = np.asarray(values, dtype=float) + weights = np.asarray(weights, dtype=float) + total = float(weights.sum()) + return float((values * weights).sum() / total) if total else 0.0 + + +def gini( + values: Sequence[float] | np.ndarray, weights: Sequence[float] | np.ndarray +) -> float: + """Weighted Gini coefficient of ``values``. + + Standard weighted formulation: sort by value, then + ``G = 1 - 2 * sum(w_i * (S_i - y_i*w_i/2)) / (W * S_n)`` with ``S_i`` the + cumulative weighted income. Returns 0.0 for empty/zero-mass input. + + Raises: + ValueError: On mismatched lengths, non-finite entries, or negative + weights. + """ + + values = np.asarray(values, dtype=float) + weights = np.asarray(weights, dtype=float) + if values.shape != weights.shape or values.ndim != 1: + raise ValueError("values and weights must be 1-D arrays of equal length.") + if not (np.isfinite(values).all() and np.isfinite(weights).all()): + raise ValueError("values and weights must be finite.") + if (weights < 0).any(): + raise ValueError("weights must be non-negative.") + total_weight = float(weights.sum()) + if len(values) == 0 or total_weight == 0.0: + return 0.0 + order = np.argsort(values, kind="stable") + sorted_values = values[order] + sorted_weights = weights[order] + cumulative = np.cumsum(sorted_values * sorted_weights) + total_income = float(cumulative[-1]) + if total_income == 0.0: + return 0.0 + # Trapezoid area under the weighted Lorenz curve. + lorenz_mass = float( + (sorted_weights * (cumulative - sorted_values * sorted_weights / 2.0)).sum() + ) + return float(1.0 - 2.0 * lorenz_mass / (total_weight * total_income)) + + +def decile_ids( + income: Sequence[float] | np.ndarray, + weights: Sequence[float] | np.ndarray, + *, + n_quantiles: int = 10, +) -> np.ndarray: + """Weighted decile assignment (1..``n_quantiles``) per record. + + Each record is placed by the lower edge of its cumulative weighted-income + rank (so a heavy record spanning several decile boundaries reports in the + lowest decile it touches). This is the *baseline* assignment used for the + JR16 replication view — the shocked deltas are reported within these + fixed deciles. + + Raises: + ValueError: On empty input, non-finite entries, or non-positive + total weight. + """ + + income = np.asarray(income, dtype=float) + weights = np.asarray(weights, dtype=float) + if income.ndim != 1 or income.shape != weights.shape or len(income) == 0: + raise ValueError("income and weights must be equal-length, non-empty 1-D.") + if not (np.isfinite(income).all() and np.isfinite(weights).all()): + raise ValueError("income and weights must be finite.") + if (weights < 0).any() or weights.sum() <= 0: + raise ValueError("weights must be non-negative with positive total.") + + order = np.argsort(income, kind="stable") + cumulative = np.cumsum(weights[order]) + lower_edge = (cumulative - weights[order]) / cumulative[-1] + interior = np.arange(1, n_quantiles) / n_quantiles + bands = np.searchsorted(interior, lower_edge, side="right") + 1 + out = np.empty(len(income), dtype=int) + out[order] = bands + return out + + +@dataclass(frozen=True) +class DecileDelta: + """Per-baseline-income-decile change in mean equivalised income.""" + + decile: int + baseline_mean_income: float + shocked_mean_income: float + mean_income_change: float + weighted_population: float + + +@dataclass(frozen=True) +class AgeBandDelta: + """Per-age-band change in mean equivalised income (the JR16 extension).""" + + age_band: str + baseline_mean_income: float + shocked_mean_income: float + mean_income_change: float + weighted_population: float + + +def decile_deltas( + baseline_income: np.ndarray, + shocked_income: np.ndarray, + weights: np.ndarray, + *, + deciles: np.ndarray | None = None, +) -> tuple[DecileDelta, ...]: + """Mean-income deltas by baseline income decile (JR16 replication view). + + Deciles are computed on (or supplied fixed at) the *baseline* income, so + a household displaced into a lower income still reports within its + pre-shock decile — exactly the JR16 incidence table shape. + """ + + baseline_income = np.asarray(baseline_income, dtype=float) + shocked_income = np.asarray(shocked_income, dtype=float) + weights = np.asarray(weights, dtype=float) + if deciles is None: + deciles = decile_ids(baseline_income, weights) + deciles = np.asarray(deciles, dtype=int) + out = [] + for decile in range(1, int(deciles.max()) + 1 if len(deciles) else 1): + mask = deciles == decile + base = weighted_mean(baseline_income[mask], weights[mask]) + shocked = weighted_mean(shocked_income[mask], weights[mask]) + out.append( + DecileDelta( + decile=decile, + baseline_mean_income=base, + shocked_mean_income=shocked, + mean_income_change=shocked - base, + weighted_population=float(weights[mask].sum()), + ) + ) + return tuple(out) + + +def age_band_deltas( + age: np.ndarray, + baseline_income: np.ndarray, + shocked_income: np.ndarray, + weights: np.ndarray, +) -> tuple[AgeBandDelta, ...]: + """Mean-income deltas by :data:`AGE_BANDS` (the extension beyond JR16).""" + + age = np.asarray(age, dtype=float) + baseline_income = np.asarray(baseline_income, dtype=float) + shocked_income = np.asarray(shocked_income, dtype=float) + weights = np.asarray(weights, dtype=float) + out = [] + for lower, upper in AGE_BANDS: + if upper is None: + label, mask = f"{lower}+", age >= lower + else: + label, mask = f"{lower}-{upper - 1}", (age >= lower) & (age < upper) + base = weighted_mean(baseline_income[mask], weights[mask]) + shocked = weighted_mean(shocked_income[mask], weights[mask]) + out.append( + AgeBandDelta( + age_band=label, + baseline_mean_income=base, + shocked_mean_income=shocked, + mean_income_change=shocked - base, + weighted_population=float(weights[mask].sum()), + ) + ) + return tuple(out) + + +def build_scenario_grid( + *, + displacement_rates: Sequence[float] = GRID_DISPLACEMENT_RATES, + wage_uplifts: Sequence[float] = GRID_WAGE_UPLIFTS, + capital_return_increase: float = GRID_CAPITAL_RETURN_INCREASE, + n_draws: int = 50, + seed: int = 0, + youth_displacement_multiplier: float = 1.0, +) -> tuple[ShockScenario, ...]: + """The ESRI JR16 robustness grid as :class:`ShockScenario` objects. + + Defaults reproduce JR16's own grid: employment displacement 1-10% + crossed with wage changes +1-5%, the +0.4pp capital shock always on. + Scenario names encode the point, e.g. ``"grid_emp7pct_wage2.6pct"``. + """ + + def _pct(value: float) -> str: + percent = value * 100 + return f"{percent:g}pct" + + return tuple( + ShockScenario( + name=f"grid_emp{_pct(displacement)}_wage{_pct(wage)}", + displacement_rate=float(displacement), + wage_uplift=float(wage), + capital_return_increase=float(capital_return_increase), + youth_displacement_multiplier=youth_displacement_multiplier, + n_draws=n_draws, + seed=seed, + ) + for displacement in displacement_rates + for wage in wage_uplifts + ) + + +# ---------------------------------------------------------------------- +# Result dataclasses and writers (pure; local_runner style) +# ---------------------------------------------------------------------- + + +@dataclass(frozen=True) +class PopulationMetrics: + """One simulation's headline distributional metrics. + + ``poverty_rate_bhc``/``poverty_rate_ahc`` are None where the installed + PolicyEngine-UK model does not expose the corresponding poverty flag. + """ + + gov_balance: float + poverty_rate_bhc: float | None + poverty_rate_ahc: float | None + gini_equivalised_income: float + + +@dataclass(frozen=True) +class ScenarioResult: + """Baseline-versus-shock deltas for one AI shock scenario. + + ``exchequer_cost`` is ``baseline.gov_balance - shocked.gov_balance``: a + positive number is a cost to the Exchequer. + """ + + scenario: str + displacement_rate: float + wage_uplift: float + capital_return_increase: float + baseline: PopulationMetrics + shocked: PopulationMetrics + exchequer_cost: float + poverty_rate_change_bhc: float | None + poverty_rate_change_ahc: float | None + gini_change: float + by_decile: tuple[DecileDelta, ...] + by_age_band: tuple[AgeBandDelta, ...] + employment_status_applied: bool + shock_summary: Mapping[str, Any] + + def to_dict(self) -> dict[str, Any]: + """JSON-serialisable dict (round-trips via + :func:`scenario_result_from_dict`).""" + + return asdict(self) | {"shock_summary": dict(self.shock_summary)} + + +def scenario_result_from_dict(payload: Mapping[str, Any]) -> ScenarioResult: + """Rebuild a :class:`ScenarioResult` from :meth:`ScenarioResult.to_dict`.""" + + data = dict(payload) + data["baseline"] = PopulationMetrics(**data["baseline"]) + data["shocked"] = PopulationMetrics(**data["shocked"]) + data["by_decile"] = tuple(DecileDelta(**row) for row in data["by_decile"]) + data["by_age_band"] = tuple(AgeBandDelta(**row) for row in data["by_age_band"]) + return ScenarioResult(**data) + + +def _overall_rows(results: Sequence[ScenarioResult]) -> list[dict[str, Any]]: + return [ + { + "scenario": result.scenario, + "displacement_rate": result.displacement_rate, + "wage_uplift": result.wage_uplift, + "capital_return_increase": result.capital_return_increase, + "exchequer_cost": result.exchequer_cost, + "poverty_rate_change_bhc": result.poverty_rate_change_bhc, + "poverty_rate_change_ahc": result.poverty_rate_change_ahc, + "gini_change": result.gini_change, + "baseline_gini": result.baseline.gini_equivalised_income, + "shocked_gini": result.shocked.gini_equivalised_income, + "employment_status_applied": result.employment_status_applied, + } + for result in results + ] + + +def write_grid_json( + results: Sequence[ScenarioResult], + path: str | Path, +) -> None: + """Write full scenario results (deciles, age bands, summaries) as JSON.""" + + Path(path).parent.mkdir(parents=True, exist_ok=True) + Path(path).write_text( + json.dumps([result.to_dict() for result in results], indent=2, sort_keys=True) + ) + + +def write_grid_csv( + results: Sequence[ScenarioResult], + output_dir: str | Path, + *, + prefix: str = "ai_shock", +) -> dict[str, Path]: + """Write overall / by-decile / by-age-band CSVs, local_runner style. + + Returns the written paths keyed ``overall``, ``by_decile``, + ``by_age_band``. + """ + + out = Path(output_dir) + out.mkdir(parents=True, exist_ok=True) + paths = { + "overall": out / f"{prefix}_overall.csv", + "by_decile": out / f"{prefix}_by_decile.csv", + "by_age_band": out / f"{prefix}_by_age_band.csv", + } + pd.DataFrame(_overall_rows(results)).to_csv(paths["overall"], index=False) + pd.DataFrame( + [ + {"scenario": result.scenario, **asdict(row)} + for result in results + for row in result.by_decile + ] + ).to_csv(paths["by_decile"], index=False) + pd.DataFrame( + [ + {"scenario": result.scenario, **asdict(row)} + for result in results + for row in result.by_age_band + ] + ).to_csv(paths["by_age_band"], index=False) + return paths + + +# ---------------------------------------------------------------------- +# Engine-facing pipeline (lazy policyengine_uk imports; `uk` extra) +# ---------------------------------------------------------------------- + +#: Person-level columns the shock table is built from (plus the weight). +_BASELINE_PERSON_VARIABLES = ( + "age", + "employment_income", + "self_employment_income", + "savings_interest_income", + "dividend_income", +) + + +def _default_simulation_factory(dataset: Any) -> Any: + try: + from policyengine_uk import Microsimulation + except ImportError as exc: # pragma: no cover - engine-absent path + raise ImportError( + "AI shock analysis requires policyengine-uk. Install the UK " + "engine (the `uk` extra) before calling run_ai_shock_analysis()." + ) from exc + return Microsimulation(dataset=dataset) + + +def _values(result: Any) -> np.ndarray: + if hasattr(result, "values"): + return np.asarray(result.values) + return np.asarray(result) + + +def _calculate(sim: Any, variable: str, period: int | str, map_to: str) -> np.ndarray: + return _values(sim.calculate(variable, period=period, map_to=map_to)).astype(float) + + +def _try_poverty_rate( + sim: Any, variable: str, period: int | str, weights: np.ndarray +) -> float | None: + """Person-weighted poverty rate, or None if the model lacks the flag.""" + + try: + flags = _calculate(sim, variable, period, "person") + except Exception: + return None + return weighted_mean(flags, weights) + + +def _population_metrics( + sim: Any, + period: int | str, + *, + person_weights: np.ndarray, + household_weights: np.ndarray, +) -> tuple[PopulationMetrics, np.ndarray]: + """Headline metrics plus person-level equivalised household income.""" + + gov_balance = float( + (_calculate(sim, "gov_balance", period, "household") * household_weights).sum() + ) + equivalised_household = _calculate( + sim, "equiv_household_net_income", period, "household" + ) + equivalised_person = _calculate(sim, "equiv_household_net_income", period, "person") + metrics = PopulationMetrics( + gov_balance=gov_balance, + poverty_rate_bhc=_try_poverty_rate( + sim, "in_poverty_bhc", period, person_weights + ), + poverty_rate_ahc=_try_poverty_rate( + sim, "in_poverty_ahc", period, person_weights + ), + gini_equivalised_income=gini(equivalised_household, household_weights), + ) + return metrics, equivalised_person + + +def _apply_shocked_inputs(sim: Any, shocked: pd.DataFrame, period: int | str) -> bool: + """Write the shocked person table into ``sim`` via ``set_input``. + + Sets the income inputs, then attempts the displaced persons' + ``employment_status`` (PE-UK's employment-status input enum). Returns + whether the status input was accepted — see the module docstring's + expressibility note; contributory-benefit exhaustion is not encodable. + """ + + for column in SHOCKED_INPUT_COLUMNS: + if column in shocked.columns: + sim.set_input(column, period, shocked[column].to_numpy(dtype=float)) + displaced = shocked[DISPLACED_COLUMN].to_numpy(dtype=bool) + if not displaced.any(): + return True + try: + current = np.asarray( + _values(sim.calculate("employment_status", period=period, map_to="person")), + dtype=object, + ) + status = current.copy() + status[displaced] = "UNEMPLOYED" + sim.set_input("employment_status", period, status) + return True + except Exception: + return False + + +def run_ai_shock_analysis( + dataset: Any | str | Path, + scenario: ShockScenario | str, + *, + exposure: Sequence[float] | np.ndarray | pd.Series | None = None, + soc_major_group: Sequence[float] | np.ndarray | pd.Series | None = None, + complementarity: Sequence[float] | np.ndarray | pd.Series | None = None, + period: int | str | None = None, + draw_index: int = 0, + simulation_factory: Callable[[Any], Any] | None = None, +) -> ScenarioResult: + """Run one AI shock scenario end-to-end against a UK H5 dataset. + + Args: + dataset: A ``UKSingleYearDataset`` (or its H5 path; loaded via + :func:`~populace.build.uk_runtime.local_runner.load_uk_dataset`). + scenario: A :class:`ShockScenario` or a preset name from + :data:`~populace.build.uk_runtime.ai_shock_scenarios.PRESETS`. + exposure: One AI-exposure score (C-AIOE) per person, e.g. QRF draws + from :mod:`~populace.build.uk_runtime.exposure_imputation`. When + omitted, ``soc_major_group`` must be given and the zero-model + crosswalk baseline + (:func:`~populace.build.uk_runtime.ai_exposure.exposure_for_major_group`) + is used. + soc_major_group: Observed FRS major groups (``1000``-``9000``; see + :mod:`~populace.build.uk_runtime.frs_occupation`), used to derive + exposure — and complementarity, when not supplied — from the + shipped crosswalk. + complementarity: Optional per-person theta for the eq 3.5 wage + distribution; the scenario falls back per + :mod:`~populace.build.uk_runtime.ai_shock_scenarios`. + period: Simulation period; inferred from the dataset when omitted. + draw_index: Which employment-shock draw the shocked simulation + realises. + simulation_factory: Override for tests; defaults to + ``policyengine_uk.Microsimulation``. + + Returns: + A frozen :class:`ScenarioResult`. + """ + + dataset_obj = ( + load_uk_dataset(dataset) if isinstance(dataset, (str, Path)) else dataset + ) + run_period = _infer_period(dataset_obj, period) + factory = simulation_factory or _default_simulation_factory + + baseline_sim = factory(dataset_obj) + person_weights = _calculate(baseline_sim, "person_weight", run_period, "person") + household_weights = _calculate( + baseline_sim, "household_weight", run_period, "household" + ) + persons = pd.DataFrame( + { + variable: _calculate(baseline_sim, variable, run_period, "person") + for variable in _BASELINE_PERSON_VARIABLES + } + ) + persons[DEFAULT_WEIGHT_COLUMN] = person_weights + persons = _attach_exposure_columns( + persons, + exposure=exposure, + soc_major_group=soc_major_group, + complementarity=complementarity, + ) + + baseline_metrics, baseline_equivalised = _population_metrics( + baseline_sim, + run_period, + person_weights=person_weights, + household_weights=household_weights, + ) + age = persons["age"].to_numpy(dtype=float) + + shocked_table, shock_summary = run_scenario( + persons, scenario, draw_index=draw_index + ) + + shocked_sim = factory(dataset_obj) + employment_status_applied = _apply_shocked_inputs( + shocked_sim, shocked_table, run_period + ) + shocked_metrics, shocked_equivalised = _population_metrics( + shocked_sim, + run_period, + person_weights=person_weights, + household_weights=household_weights, + ) + + scenario_name = str(shock_summary["scenario"]) + parameters = shock_summary["parameters"] + return ScenarioResult( + scenario=scenario_name, + displacement_rate=float(parameters["displacement_rate"]), + wage_uplift=float(parameters["wage_uplift"]), + capital_return_increase=float(parameters["capital_return_increase"]), + baseline=baseline_metrics, + shocked=shocked_metrics, + exchequer_cost=baseline_metrics.gov_balance - shocked_metrics.gov_balance, + poverty_rate_change_bhc=_optional_delta( + baseline_metrics.poverty_rate_bhc, shocked_metrics.poverty_rate_bhc + ), + poverty_rate_change_ahc=_optional_delta( + baseline_metrics.poverty_rate_ahc, shocked_metrics.poverty_rate_ahc + ), + gini_change=shocked_metrics.gini_equivalised_income + - baseline_metrics.gini_equivalised_income, + by_decile=decile_deltas( + baseline_equivalised, shocked_equivalised, person_weights + ), + by_age_band=age_band_deltas( + age, baseline_equivalised, shocked_equivalised, person_weights + ), + employment_status_applied=employment_status_applied, + shock_summary=shock_summary, + ) + + +def run_grid( + dataset: Any | str | Path, + scenarios: Sequence[ShockScenario | str] | None = None, + *, + output_dir: str | Path | None = None, + **kwargs: Any, +) -> tuple[ScenarioResult, ...]: + """Run many scenarios (default: presets plus the JR16 robustness grid). + + Loads the dataset once and forwards ``kwargs`` (exposure attachment, + period, factory, ...) to :func:`run_ai_shock_analysis` per scenario. + With ``output_dir`` set, writes the JSON and CSV outputs via + :func:`write_grid_json` / :func:`write_grid_csv`. + """ + + dataset_obj = ( + load_uk_dataset(dataset) if isinstance(dataset, (str, Path)) else dataset + ) + if scenarios is None: + scenarios = (*PRESETS.values(), *build_scenario_grid()) + results = tuple( + run_ai_shock_analysis(dataset_obj, scenario, **kwargs) for scenario in scenarios + ) + if output_dir is not None: + write_grid_json(results, Path(output_dir) / "ai_shock_results.json") + write_grid_csv(results, output_dir) + return results + + +def _optional_delta(baseline: float | None, shocked: float | None) -> float | None: + if baseline is None or shocked is None: + return None + return shocked - baseline + + +def _infer_period(dataset: Any, period: int | str | None) -> int | str: + if period is not None: + return period + for attr in ("time_period", "fiscal_year", "default_calculation_period"): + value = getattr(dataset, attr, None) + if value is not None: + return value + raise ValueError("period is required when it cannot be inferred from the dataset.") + + +def _attach_exposure_columns( + persons: pd.DataFrame, + *, + exposure: Any, + soc_major_group: Any, + complementarity: Any, +) -> pd.DataFrame: + """Attach ``ai_exposure`` (and ``ai_complementarity`` when derivable).""" + + persons = persons.copy() + if exposure is None and soc_major_group is None: + raise ValueError( + "Provide either per-person exposure scores (e.g. QRF draws from " + "exposure_imputation) or the observed soc_major_group column " + "(frs_occupation) to derive them from the crosswalk." + ) + if exposure is not None: + persons["ai_exposure"] = np.asarray(exposure, dtype=float) + if soc_major_group is not None: + from populace.build.uk_runtime.ai_exposure import exposure_for_major_group + from populace.build.uk_runtime.frs_occupation import frs_major_group_to_digit + + digits = frs_major_group_to_digit(np.asarray(soc_major_group)) + if exposure is None: + persons["ai_exposure"] = exposure_for_major_group(digits, "c_aioe") + if complementarity is None: + theta = exposure_for_major_group(digits, "complementarity") + finite = np.isfinite(theta) + if finite.any() and not finite.all(): + # Persons without an observed major group (children, missing + # SOC) take the mean theta so the eq 3.5 factors stay finite. + theta = np.where(finite, theta, float(theta[finite].mean())) + persons["ai_complementarity"] = theta + if complementarity is not None: + persons["ai_complementarity"] = np.asarray(complementarity, dtype=float) + if not np.isfinite(persons["ai_exposure"].to_numpy(dtype=float)).all(): + raise ValueError( + "ai_exposure contains non-finite entries; persons without a " + "resolvable exposure score cannot enter the employment shock. " + "Impute or fill exposure upstream (exposure_imputation's blind " + "fallback covers persons without an observed major group)." + ) + return persons diff --git a/packages/populace-build/src/populace/build/uk_runtime/ai_shock_scenarios.py b/packages/populace-build/src/populace/build/uk_runtime/ai_shock_scenarios.py new file mode 100644 index 00000000..ac167f6a --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/ai_shock_scenarios.py @@ -0,0 +1,904 @@ +"""ESRI-style AI adoption shock scenarios for UK person-level microdata. + +Implements the scenario-microsimulation method of ESRI JR16 ("AI and income +inequality in Ireland", Doorley et al. 2026, Chapter 3): three stylised +shocks applied to a person table carrying AI exposure scores — + +1. **Employment shock** (eq 3.4) — total job loss is allocated across + occupation groups in proportion to employment-weighted complementarity- + adjusted AI occupational exposure (C-AIOE): + ``JobLoss_l = TotalJobLoss * (EMP_l * C_AIOE_l) / sum_l(EMP_l * C_AIOE_l)``. + Displaced individuals are selected *randomly within group*, and summary + statistics are averaged over ``n_draws`` random draws (ESRI use 50). +2. **Wage shock** (eq 3.5) — the aggregate wage change is distributed by a + *separate* complementarity score theta (Pizzinelli et al. 2023), not by + the exposure score: + ``WageChange_l = TotalWageChange * theta_l / (sum_l theta_l*EMP_l / sum_l EMP_l)``. +3. **Capital shock** — the return to capital rises (ESRI: 1.005% -> 1.405%, + a ~+39.8% increase in capital income), applied uniformly to recipients of + interest/dividend-type income. Rental income is *excluded*, following + ESRI. + +This is *scenario analysis*, not causal estimation: parameters are +calibrated-from-literature anchors the analyst is expected to override (see +:data:`PRESETS`). The central anchors are the ones ESRI JR16 (Doorley, +O'Connor, O'Shea & Tuda 2026, doi:10.26504/jr16) adopts from Briggs & +Kodnani (2023): 7% job loss, and a +2.6pp wage change that is *not* a +Goldman Sachs headline number but "the median wage change estimate of a +number of studies surveyed by the authors" (JR16 fn.3, §3.2). JR16 notes +this scenario is an upper bound relative to Acemoglu (2025), "The Simple +Macroeconomics of AI", Economic Policy 40(121), 13–58 (employment effect +0.9–1.1%, with *no* corresponding wage change estimated — JR16 fn.8) and +McKinsey (2025) (wage +0.35%). The +0.4pp return-to-capital increase is +from Cazzaniga, Pizzinelli et al. (2024, IMF), applied uniformly across +households with non-rental capital income. + +Scenario grid for downstream drivers +------------------------------------ +JR16's own robustness grid — the intended scenario-grid shape for +downstream drivers of this module — spans employment displacement of +1–10% crossed with wage changes of +1–5%, with the +0.4pp capital shock +always on. Note the grid tops out at 10% displacement; the "high" preset +below deliberately exceeds it (see the preset notes). + +Scope limitation: self-employed excluded +---------------------------------------- +All shocks operate on *employees only*: the employed mask is +``employment_income > 0`` and ``age >= 16``, so persons whose labour +income is self-employment income receive no employment or wage shock. +This mirrors ESRI JR16's treatment, but it is a real scope limitation — +self-employed workers are exposed to AI too, and their exclusion is +silent in the microdata. The employment-shock summary therefore reports +``excluded_self_employed_weighted``, the weighted count of persons aged +16+ with positive ``self_employment_income`` who fall outside the +employed mask, so the omitted population is visible to downstream +consumers. + +Displaced-worker contract +------------------------- +ESRI treat transitioned workers as unemployed for 9+ months: contributory +benefits exhausted, no re-employment within the scenario horizon. This module +does **not** model benefit entitlements. It sets ``employment_income`` to 0, +stores the pre-shock wage in ``pre_shock_employment_income`` and flags +``ai_displaced=True``; the downstream PolicyEngine-UK run is responsible for +modelling flagged persons as long-term unemployed. + +Extension beyond ESRI JR16 +-------------------------- +ESRI note as a limitation that their within-group selection is random and +"does not account for the higher transition probability of younger workers". +This module's novelty is twofold: + +* every shock summary is resolved **by age band** (16-24, 25-34, ..., 65+); +* the employment shock takes an optional ``youth_displacement_multiplier`` + that tilts within-group selection toward 16-24 year olds (seniority-biased + displacement) while *preserving the group-level job-loss totals of eq 3.4*. + The default 1.0 reproduces ESRI's age-neutral random selection. + +Data surface +------------ +Functions operate on a **pandas person table** and return a modified copy. +This is deliberate: :class:`populace.frame.Frame` tables are documented +read-only and rebuilding a frame requires schema/weights plumbing that shock +scenarios do not own, while the ESRI-style pipeline (shock, then re-simulate +taxes and benefits) consumes a person table. A ``Frame`` is accepted as +*input* for convenience — its person table is extracted with resolved person +weights attached — but the output is always a DataFrame. + +Person-table contract: ``age``, ``employment_income``, ``ai_exposure`` +(C-AIOE, drives job loss), optionally ``ai_complementarity`` (theta, drives +wage gains; uniform fallback with a warning if absent), optionally +``occupation_group`` (2-digit occupation groups for eq 3.4; weighted exposure +quintiles are used as pseudo-groups if absent), a person-weight column, and +the capital income columns to be scaled. + +Determinism: the only randomness is the within-group selection of displaced +workers, seeded per draw from ``(scenario.seed, draw_index)`` (the repo +forbids unseeded randomness in builds). Wage and capital shocks are fully +deterministic. +""" + +from __future__ import annotations + +import warnings +from dataclasses import dataclass, replace +from typing import Any + +import numpy as np +import pandas as pd + +AGE_COLUMN = "age" +EXPOSURE_COLUMN = "ai_exposure" +COMPLEMENTARITY_COLUMN = "ai_complementarity" +OCCUPATION_GROUP_COLUMN = "occupation_group" +EMPLOYMENT_INCOME_COLUMN = "employment_income" +DISPLACED_COLUMN = "ai_displaced" +PRE_SHOCK_EMPLOYMENT_INCOME_COLUMN = "pre_shock_employment_income" +DEFAULT_WEIGHT_COLUMN = "person_weight" + +#: Capital income scaled by the capital shock: interest/dividend-type income +#: only. ESRI exclude rental income from the returns-to-capital gain, so +#: property/rent columns are deliberately absent. +DEFAULT_CAPITAL_COLUMNS = ("savings_interest_income", "dividend_income") + +#: Default working-age bands the outputs are resolved by. Half-open +#: [lower, upper); the final band is 65+. Drivers may pass an alternative +#: scheme via the ``age_bands`` parameter of the shock functions. +AGE_BANDS = ((16, 25), (25, 35), (35, 45), (45, 55), (55, 65), (65, None)) +#: Default youth band targeted by ``youth_displacement_multiplier``. Kept +#: separate from ``age_bands`` so overriding the reporting bands never +#: changes which workers the multiplier tilts toward. +YOUTH_BAND = (16, 25) + +#: Column identifying self-employed persons for the scope-limitation count +#: in the employment-shock summary (see the module docstring). +SELF_EMPLOYMENT_INCOME_COLUMN = "self_employment_income" + +AgeBands = tuple[tuple[int, int | None], ...] + +#: Number of pseudo exposure-quintile groups used when the person table has +#: no ``occupation_group`` column. +N_FALLBACK_EXPOSURE_GROUPS = 5 + + +@dataclass(frozen=True) +class ShockScenario: + """One named AI-adoption shock scenario. + + Attributes: + name: Scenario label carried into summaries. + displacement_rate: TotalJobLoss as a share of aggregate weighted + employment (0-1), allocated across occupation groups by eq 3.4. + wage_uplift: Aggregate proportional wage change for non-displaced + workers (0.026 = +2.6%), distributed by complementarity (eq 3.5). + base_capital_return: Baseline return to capital (ESRI: 1.005%). + capital_return_increase: Scenario increase in the return to capital + in the same units (ESRI central: +0.4pp), so + ``capital_uplift = capital_return_increase / base_capital_return`` + (~+39.8% capital income at the ESRI anchors). + youth_displacement_multiplier: Relative within-group selection + probability for 16-24 year olds versus older workers. 1.0 + reproduces ESRI's age-neutral random selection; >1 encodes + seniority-biased displacement. Group-level job-loss totals from + eq 3.4 are preserved either way. + n_draws: Number of random within-group selection draws averaged in + the employment-shock summary (ESRI use 50). + seed: Base seed; draw ``d`` uses generator seed ``(seed, d)``. Same + seed and data give identical results. + """ + + name: str + displacement_rate: float + wage_uplift: float + base_capital_return: float = 0.01005 + capital_return_increase: float = 0.004 + youth_displacement_multiplier: float = 1.0 + n_draws: int = 50 + seed: int = 0 + + def __post_init__(self) -> None: + if not 0.0 <= self.displacement_rate < 1.0: + raise ValueError( + f"displacement_rate must be in [0, 1), got {self.displacement_rate}." + ) + if self.wage_uplift < 0.0: + raise ValueError(f"wage_uplift must be >= 0, got {self.wage_uplift}.") + if self.base_capital_return <= 0.0: + raise ValueError( + f"base_capital_return must be > 0, got {self.base_capital_return}." + ) + if self.capital_return_increase < 0.0: + raise ValueError( + "capital_return_increase must be >= 0, got " + f"{self.capital_return_increase}." + ) + if self.youth_displacement_multiplier <= 0.0: + raise ValueError( + "youth_displacement_multiplier must be > 0, got " + f"{self.youth_displacement_multiplier}." + ) + if self.n_draws < 1: + raise ValueError(f"n_draws must be >= 1, got {self.n_draws}.") + + @property + def capital_uplift(self) -> float: + """Proportional capital income change implied by the return shift.""" + + return self.capital_return_increase / self.base_capital_return + + +#: Klein & Teeselink (2025, SSRN 5516798) find that highly AI-exposed firms +#: cut total employment by -4.5% and junior employment by -5.8%. The ratio +#: 5.8 / 4.5 ~= 1.29 is a UK-evidence-calibrated value for +#: ``youth_displacement_multiplier`` (juniors displaced ~1.3x as intensely +#: as workers overall). The default presets keep the multiplier at 1.0 — +#: pure ESRI replication with age-neutral within-group selection — and the +#: ``"central_youth_tilted"`` preset applies this calibration. +KLEIN_TEESELINK_YOUTH_MULTIPLIER = 5.8 / 4.5 + +#: Scenario presets — calibrated-from-literature anchors, not estimates; +#: analysts should override them. JR16's own robustness grid (employment +#: 1-10% x wages +1-5%, +0.4pp capital always on) is the intended +#: scenario-grid shape for downstream drivers. +#: +#: Central: 7% job loss and +2.6pp wages, the Briggs & Kodnani (2023) +#: anchors adopted by ESRI JR16 (Doorley, O'Connor, O'Shea & Tuda 2026, +#: doi:10.26504/jr16). The +2.6pp is "the median wage change estimate of a +#: number of studies surveyed by the authors" (JR16 fn.3, §3.2) — not a +#: Goldman headline — and JR16 notes this scenario is an upper bound versus +#: Acemoglu (2025) employment 0.9-1.1% and McKinsey (2025) wage +0.35%. +#: Capital: +0.4pp return from Cazzaniga/Pizzinelli et al. (2024, IMF) +#: (1.005% -> 1.405%, ~+39.8% capital income), uniform across households +#: with non-rental capital income. +#: +#: Low: ~1% job loss, following Acemoglu (2025), "The Simple Macroeconomics +#: of AI", Economic Policy 40(121), 13-58 — employment effect 0.9-1.1%. +#: Note Acemoglu estimates *no* corresponding wage change (JR16 fn.8); the +#: preset's small wage/capital gains are this module's own muted variants, +#: not his estimates. +#: +#: High: 13% job loss, from Brynjolfsson, Chandar & Chen (2025), "Canaries +#: in the Coal Mine?" (Stanford Digital Economy Lab). Caveat: that 13% is a +#: *cohort-specific* relative employment decline for early-career workers +#: (aged 22-25) in the most-exposed occupations, not an aggregate +#: displacement estimate; it is used here deliberately as an aggressive +#: aggregate scenario. JR16's own robustness grid tops out at 10% +#: displacement, so this preset sits above JR16's upper bound. +PRESETS: dict[str, ShockScenario] = { + "central": ShockScenario( + name="central", + displacement_rate=0.07, + wage_uplift=0.026, + capital_return_increase=0.004, + ), + "low": ShockScenario( + name="low", + displacement_rate=0.01, + wage_uplift=0.01, + capital_return_increase=0.002, + ), + "high": ShockScenario( + name="high", + displacement_rate=0.13, + wage_uplift=0.04, + capital_return_increase=0.008, + ), + # Central parameters with the Klein & Teeselink (2025) UK-evidence + # youth calibration (~1.29x) applied; see + # KLEIN_TEESELINK_YOUTH_MULTIPLIER above. + "central_youth_tilted": ShockScenario( + name="central_youth_tilted", + displacement_rate=0.07, + wage_uplift=0.026, + capital_return_increase=0.004, + youth_displacement_multiplier=KLEIN_TEESELINK_YOUTH_MULTIPLIER, + ), +} + + +def age_band_label(age: float, age_bands: AgeBands = AGE_BANDS) -> str: + """Human-readable band label for ``age`` (e.g. ``"25-34"``, ``"65+"``).""" + + for lower, upper in age_bands: + if upper is None: + if age >= lower: + return f"{lower}+" + elif lower <= age < upper: + return f"{lower}-{upper - 1}" + return "under_16" + + +def apply_employment_shock( + frame_or_df: Any, + scenario: ShockScenario, + *, + weight_column: str = DEFAULT_WEIGHT_COLUMN, + draw_index: int = 0, + age_bands: AgeBands = AGE_BANDS, + youth_band: tuple[int, int] = YOUTH_BAND, +) -> tuple[pd.DataFrame, dict[str, Any]]: + """Transition a fraction of employed persons to unemployment (eq 3.4). + + TotalJobLoss (``scenario.displacement_rate`` times aggregate weighted + employment) is allocated to occupation groups in proportion to + ``EMP_l * C_AIOE_l`` with employment-weighted group mean exposure, then + displaced individuals are selected randomly within each group until the + group's weighted quota is met. Summary statistics are averaged over + ``scenario.n_draws`` seeded draws; the returned table is the single + realisation ``draw_index`` (for downstream tax-benefit simulation). + + Displaced persons get ``employment_income = 0``, their pre-shock wage in + ``pre_shock_employment_income`` and ``ai_displaced = True``. Benefit + treatment (ESRI: long-term unemployed, contributory benefits exhausted) + is the downstream simulation's contract, not this function's. + + If ``occupation_group`` is absent, weighted exposure quintiles of the + employed serve as pseudo-groups. + + ``age_bands`` controls the reporting bands of the summary only; + ``youth_band`` (half-open, default 16-24) controls which workers the + ``youth_displacement_multiplier`` targets and is deliberately + independent of ``age_bands`` so overriding the reporting scheme never + changes the shock itself. + + Returns: + ``(shocked, summary)`` — the ``draw_index`` realisation and a summary + with draw-averaged displaced counts, weighted shares, per-age-band + and per-group breakdowns (including the eq 3.4 quotas), plus + ``excluded_self_employed_weighted``: the weighted count of persons + aged 16+ with positive self-employment income outside the employed + mask, who receive no shock (see the module docstring). + """ + + persons = _person_table(frame_or_df, weight_column) + _require_columns( + persons, (AGE_COLUMN, EXPOSURE_COLUMN, EMPLOYMENT_INCOME_COLUMN, weight_column) + ) + if not 0 <= draw_index < scenario.n_draws: + raise ValueError( + f"draw_index must be in [0, {scenario.n_draws}), got {draw_index}." + ) + shocked = persons.copy() + if PRE_SHOCK_EMPLOYMENT_INCOME_COLUMN not in shocked.columns: + shocked[PRE_SHOCK_EMPLOYMENT_INCOME_COLUMN] = shocked[ + EMPLOYMENT_INCOME_COLUMN + ].astype(float) + shocked[DISPLACED_COLUMN] = False + + employed = _employed_mask(shocked).to_numpy() + weights = shocked[weight_column].to_numpy(dtype=float) + age = shocked[AGE_COLUMN].to_numpy(dtype=float) + exposure = shocked[EXPOSURE_COLUMN].to_numpy(dtype=float) + groups = _occupation_groups(shocked, exposure, weights, employed) + quotas = _group_job_loss_quotas( + exposure, groups, weights, employed, scenario.displacement_rate + ) + + draws = [ + _draw_displaced( + employed=employed, + weights=weights, + age=age, + groups=groups, + quotas=quotas, + youth_multiplier=scenario.youth_displacement_multiplier, + youth_band=youth_band, + rng=np.random.default_rng(np.random.SeedSequence((scenario.seed, draw))), + ) + for draw in range(scenario.n_draws) + ] + displaced = draws[draw_index] + shocked.loc[displaced, DISPLACED_COLUMN] = True + shocked.loc[displaced, EMPLOYMENT_INCOME_COLUMN] = 0.0 + + employed_weight = float(weights[employed].sum()) + summary = { + "shock": "employment", + "scenario": scenario.name, + "displacement_rate": scenario.displacement_rate, + "youth_displacement_multiplier": scenario.youth_displacement_multiplier, + "n_draws": scenario.n_draws, + "returned_draw": draw_index, + "employed_weighted": employed_weight, + "displaced_count": float(np.mean([d.sum() for d in draws])), + "displaced_weighted": float(np.mean([weights[d].sum() for d in draws])), + "displaced_weighted_share": _safe_share( + float(np.mean([weights[d].sum() for d in draws])), employed_weight + ), + "excluded_self_employed_weighted": _excluded_self_employed_weighted( + shocked, employed, weights, age + ), + "by_occupation_group": _group_summary(draws, quotas, groups, weights, employed), + "by_age_band": _age_band_summary(draws, age, weights, employed, age_bands), + } + return shocked, summary + + +def apply_wage_shock( + frame_or_df: Any, + scenario: ShockScenario, + *, + weight_column: str = DEFAULT_WEIGHT_COLUMN, + age_bands: AgeBands = AGE_BANDS, +) -> tuple[pd.DataFrame, dict[str, Any]]: + """Uplift wages of non-displaced workers by complementarity (eq 3.5). + + Each employed, non-displaced worker's ``employment_income`` is scaled by + ``1 + scenario.wage_uplift * theta_i / mean_theta`` where ``theta`` is + the ``ai_complementarity`` score (Pizzinelli et al. 2023) and + ``mean_theta`` its employment-weighted mean over recipients — so the + weighted mean uplift equals ``scenario.wage_uplift`` while more + AI-complementary workers gain more. Note this deliberately uses the + complementarity score, *not* the exposure score that drives job loss. + + If ``ai_complementarity`` is absent (or non-positive on average), a + uniform uplift is applied and a warning is emitted. + + Fully deterministic. Displaced workers (flagged by + :func:`apply_employment_shock`) keep zero employment income. + + Returns: + ``(shocked, summary)`` with the weighted mean uplift and per-age-band + wage-bill changes. + """ + + persons = _person_table(frame_or_df, weight_column) + _require_columns(persons, (AGE_COLUMN, EMPLOYMENT_INCOME_COLUMN, weight_column)) + shocked = persons.copy() + if DISPLACED_COLUMN not in shocked.columns: + shocked[DISPLACED_COLUMN] = False + + recipients = ( + _employed_mask(shocked) & ~shocked[DISPLACED_COLUMN].astype(bool) + ).to_numpy() + weights = shocked[weight_column].to_numpy(dtype=float) + uplift = np.zeros(len(shocked), dtype=float) + uniform_fallback = False + if recipients.any() and scenario.wage_uplift > 0.0: + relative, uniform_fallback = _relative_complementarity( + shocked, recipients, weights + ) + uplift[recipients] = scenario.wage_uplift * relative + + before = shocked[EMPLOYMENT_INCOME_COLUMN].to_numpy(dtype=float) + shocked[EMPLOYMENT_INCOME_COLUMN] = before * (1.0 + uplift) + + recipient_weights = weights[recipients] + age = shocked[AGE_COLUMN].to_numpy(dtype=float) + wage_bill_change = weights * before * uplift + summary = { + "shock": "wage", + "scenario": scenario.name, + "wage_uplift": scenario.wage_uplift, + "uniform_fallback": uniform_fallback, + "recipient_count": int(recipients.sum()), + "weighted_mean_uplift": _safe_share( + float((uplift[recipients] * recipient_weights).sum()), + float(recipient_weights.sum()), + ), + "weighted_wage_bill_change": float(wage_bill_change.sum()), + "by_age_band": { + label: { + "recipient_count": int((recipients & band).sum()), + "weighted_wage_bill_change": float(wage_bill_change[band].sum()), + } + for label, band in _age_band_masks(age, age_bands) + }, + } + return shocked, summary + + +def apply_capital_shock( + frame_or_df: Any, + scenario: ShockScenario, + *, + capital_columns: tuple[str, ...] = DEFAULT_CAPITAL_COLUMNS, + weight_column: str = DEFAULT_WEIGHT_COLUMN, + age_bands: AgeBands = AGE_BANDS, +) -> tuple[pd.DataFrame, dict[str, Any]]: + """Scale interest/dividend-type income by the return-to-capital gain. + + Every present column in ``capital_columns`` is scaled uniformly by + ``1 + scenario.capital_uplift`` where the uplift is the transparent ratio + ``capital_return_increase / base_capital_return`` (ESRI: 0.004 / 0.01005 + ~= +39.8%). Rental income is excluded by default, following ESRI. It is + an error if none of the columns are present. Fully deterministic. + + Returns: + ``(shocked, summary)`` with the weighted capital-income change + overall and per age band. + """ + + persons = _person_table(frame_or_df, weight_column) + _require_columns(persons, (AGE_COLUMN, weight_column)) + present = tuple(column for column in capital_columns if column in persons.columns) + if not present: + raise ValueError( + f"None of the capital income column(s) {list(capital_columns)} are " + f"present in the person table." + ) + shocked = persons.copy() + weights = shocked[weight_column].to_numpy(dtype=float) + change = np.zeros(len(shocked), dtype=float) + for column in present: + before = shocked[column].to_numpy(dtype=float) + shocked[column] = before * (1.0 + scenario.capital_uplift) + change += before * scenario.capital_uplift + + age = shocked[AGE_COLUMN].to_numpy(dtype=float) + summary = { + "shock": "capital", + "scenario": scenario.name, + "base_capital_return": scenario.base_capital_return, + "capital_return_increase": scenario.capital_return_increase, + "capital_uplift": scenario.capital_uplift, + "capital_columns": list(present), + "weighted_capital_income_change": float((weights * change).sum()), + "by_age_band": { + label: { + "weighted_capital_income_change": float((weights * change)[band].sum()), + } + for label, band in _age_band_masks(age, age_bands) + }, + } + return shocked, summary + + +def run_scenario( + frame_or_df: Any, + scenario: ShockScenario | str, + *, + capital_columns: tuple[str, ...] = DEFAULT_CAPITAL_COLUMNS, + weight_column: str = DEFAULT_WEIGHT_COLUMN, + seed: int | None = None, + draw_index: int = 0, + age_bands: AgeBands = AGE_BANDS, + youth_band: tuple[int, int] = YOUTH_BAND, +) -> tuple[pd.DataFrame, dict[str, Any]]: + """Apply employment, wage and capital shocks in sequence. + + The employment summary averages over ``scenario.n_draws`` within-group + selection draws (ESRI average over 50); the returned microdata is the + single realisation ``draw_index``, with the wage and capital shocks + applied on top of it deterministically. + + Args: + frame_or_df: Person table or :class:`~populace.frame.Frame`. + scenario: A :class:`ShockScenario`, or a preset name from + :data:`PRESETS` (``"central"``, ``"low"``, ``"high"``, + ``"central_youth_tilted"``). + capital_columns: Capital income columns to scale (rent excluded by + default, following ESRI). + weight_column: Person-weight column name. + seed: Optional override of the scenario base seed. + draw_index: Which employment-shock draw the returned table realises. + age_bands: Reporting age bands for every shock summary + (default :data:`AGE_BANDS`). + youth_band: Half-open band targeted by the youth displacement + multiplier (default :data:`YOUTH_BAND`); independent of + ``age_bands``. + + Returns: + ``(shocked, summary)`` — the fully shocked person table (always a + DataFrame) and a summary dict with one entry per shock plus the + scenario parameters. + """ + + if isinstance(scenario, str): + try: + scenario = PRESETS[scenario] + except KeyError: + raise ValueError( + f"Unknown scenario preset {scenario!r}; presets: {sorted(PRESETS)}." + ) from None + if seed is not None: + scenario = replace(scenario, seed=seed) + + shocked, employment_summary = apply_employment_shock( + frame_or_df, + scenario, + weight_column=weight_column, + draw_index=draw_index, + age_bands=age_bands, + youth_band=youth_band, + ) + shocked, wage_summary = apply_wage_shock( + shocked, scenario, weight_column=weight_column, age_bands=age_bands + ) + shocked, capital_summary = apply_capital_shock( + shocked, + scenario, + capital_columns=capital_columns, + weight_column=weight_column, + age_bands=age_bands, + ) + summary = { + "scenario": scenario.name, + "parameters": { + "displacement_rate": scenario.displacement_rate, + "wage_uplift": scenario.wage_uplift, + "base_capital_return": scenario.base_capital_return, + "capital_return_increase": scenario.capital_return_increase, + "capital_uplift": scenario.capital_uplift, + "youth_displacement_multiplier": scenario.youth_displacement_multiplier, + "n_draws": scenario.n_draws, + "seed": scenario.seed, + }, + "employment": employment_summary, + "wage": wage_summary, + "capital": capital_summary, + } + return shocked, summary + + +# ---------------------------------------------------------------------- +# Internals +# ---------------------------------------------------------------------- + + +def _person_table(frame_or_df: Any, weight_column: str) -> pd.DataFrame: + """Normalise the input to a person DataFrame carrying ``weight_column``.""" + + if isinstance(frame_or_df, pd.DataFrame): + return frame_or_df + person = getattr(frame_or_df, "person", None) + resolve_weights = getattr(frame_or_df, "resolve_weights", None) + if isinstance(person, pd.DataFrame) and callable(resolve_weights): + table = person.copy() + if weight_column not in table.columns: + person_entity = frame_or_df.schema.person_entity + table[weight_column] = np.asarray( + resolve_weights(person_entity).values, dtype=float + ) + return table + raise TypeError( + "frame_or_df must be a pandas DataFrame or a populace Frame, got " + f"{type(frame_or_df).__name__}." + ) + + +def _safe_share(numerator: float, denominator: float) -> float: + """``numerator / denominator``, or 0.0 for an empty denominator.""" + + return float(numerator) / float(denominator) if denominator else 0.0 + + +def _require_columns(persons: pd.DataFrame, columns: tuple[str, ...]) -> None: + missing = [column for column in columns if column not in persons.columns] + if missing: + raise ValueError(f"Person table is missing required column(s): {missing}.") + + +def _employed_mask(persons: pd.DataFrame) -> pd.Series: + """Employed persons eligible for labour shocks: earning and aged 16+.""" + + return (persons[EMPLOYMENT_INCOME_COLUMN].astype(float) > 0.0) & ( + persons[AGE_COLUMN].astype(float) >= 16 + ) + + +def _excluded_self_employed_weighted( + persons: pd.DataFrame, + employed: np.ndarray, + weights: np.ndarray, + age: np.ndarray, +) -> float: + """Weighted count of self-employed persons outside the employed mask. + + Persons aged 16+ with positive ``self_employment_income`` who are not in + the employee mask receive no employment or wage shock (mirroring ESRI + JR16's employees-only treatment); this makes the omitted population + visible in the summary. Returns 0.0 when the column is absent. + """ + + if SELF_EMPLOYMENT_INCOME_COLUMN not in persons.columns: + return 0.0 + self_employed = ( + persons[SELF_EMPLOYMENT_INCOME_COLUMN].to_numpy(dtype=float) > 0.0 + ) & (age >= 16) + return float(weights[self_employed & ~employed].sum()) + + +def _weighted_quantiles( + values: np.ndarray, weights: np.ndarray, quantiles: np.ndarray +) -> np.ndarray: + """Inverse-CDF weighted quantiles (matches populace.frame.accounting).""" + + order = np.argsort(values, kind="stable") + sorted_values = values[order] + cumulative = np.cumsum(weights[order]) + proportion = cumulative / cumulative[-1] + indices = np.searchsorted(proportion, quantiles, side="left") + return sorted_values[np.clip(indices, 0, len(sorted_values) - 1)] + + +def _occupation_groups( + persons: pd.DataFrame, + exposure: np.ndarray, + weights: np.ndarray, + employed: np.ndarray, +) -> np.ndarray: + """Occupation-group label per person for the eq 3.4 allocation. + + Uses the ``occupation_group`` column when present; otherwise weighted + exposure quintiles of the employed serve as pseudo-groups (labelled + ``exposure_q1`` ... ``exposure_q5``). + """ + + if OCCUPATION_GROUP_COLUMN in persons.columns: + return persons[OCCUPATION_GROUP_COLUMN].astype(str).to_numpy() + interior = np.arange(1, N_FALLBACK_EXPOSURE_GROUPS) / N_FALLBACK_EXPOSURE_GROUPS + if employed.any(): + edges = _weighted_quantiles(exposure[employed], weights[employed], interior) + else: + edges = np.quantile(exposure, interior) + indices = np.searchsorted(edges, exposure, side="right") + labels = np.array( + [f"exposure_q{index + 1}" for index in range(N_FALLBACK_EXPOSURE_GROUPS)] + ) + return labels[indices] + + +def _group_job_loss_quotas( + exposure: np.ndarray, + groups: np.ndarray, + weights: np.ndarray, + employed: np.ndarray, + displacement_rate: float, +) -> dict[str, float]: + """Weighted job-loss quota per occupation group (eq 3.4). + + ``JobLoss_l = TotalJobLoss * (EMP_l * C_AIOE_l) / sum_l(EMP_l * C_AIOE_l)`` + with ``EMP_l`` the weighted employment of group ``l`` and ``C_AIOE_l`` its + employment-weighted mean exposure. Negative group scores (possible with + standardised C-AIOE) are floored at zero for the allocation. Quotas are + capped at group employment. + """ + + total_employment = float(weights[employed].sum()) + total_job_loss = displacement_rate * total_employment + if total_job_loss <= 0.0 or not employed.any(): + return {} + group_employment: dict[str, float] = {} + group_score: dict[str, float] = {} + for label in np.unique(groups[employed]): + in_group = employed & (groups == label) + emp = float(weights[in_group].sum()) + mean_exposure = _safe_share( + float((exposure[in_group] * weights[in_group]).sum()), emp + ) + group_employment[label] = emp + group_score[label] = emp * max(mean_exposure, 0.0) + total_score = sum(group_score.values()) + if total_score <= 0.0: + raise ValueError( + "Cannot allocate job loss: all employment-weighted group exposure " + "scores are non-positive." + ) + return { + label: min(total_job_loss * score / total_score, group_employment[label]) + for label, score in group_score.items() + } + + +def _draw_displaced( + *, + employed: np.ndarray, + weights: np.ndarray, + age: np.ndarray, + groups: np.ndarray, + quotas: dict[str, float], + youth_multiplier: float, + youth_band: tuple[int, int], + rng: np.random.Generator, +) -> np.ndarray: + """One random within-group selection meeting each group's weighted quota. + + Selection order within a group is a weighted random permutation + (Efraimidis-Spirakis keys) with ``youth_band`` workers' selection + intensity scaled by ``youth_multiplier``; persons are taken in order + until the cumulative weight is as close as possible to the group's + quota. With ``youth_multiplier == 1`` this is ESRI's uniform random + selection. + """ + + displaced = np.zeros(len(weights), dtype=bool) + youth_lower, youth_upper = youth_band + for label, quota in quotas.items(): + if quota <= 0.0: + continue + candidates = np.flatnonzero(employed & (groups == label)) + intensity = np.where( + (age[candidates] >= youth_lower) & (age[candidates] < youth_upper), + youth_multiplier, + 1.0, + ) + keys = rng.exponential(size=len(candidates)) / intensity + order = candidates[np.argsort(keys, kind="stable")] + cumulative = np.cumsum(weights[order]) + cut = int(np.searchsorted(cumulative, quota, side="left")) + if cut < len(order): + # Include the boundary person only if that lands closer to the + # quota than stopping short of it. + below = cumulative[cut - 1] if cut > 0 else 0.0 + if abs(cumulative[cut] - quota) <= abs(quota - below): + cut += 1 + displaced[order[:cut]] = True + return displaced + + +def _relative_complementarity( + persons: pd.DataFrame, + recipients: np.ndarray, + weights: np.ndarray, +) -> tuple[np.ndarray, bool]: + """Eq 3.5 relative wage-gain factors over the recipients. + + Returns ``(theta_i / mean_theta, uniform_fallback)`` with ``mean_theta`` + the employment-weighted mean complementarity over recipients, so the + factors have weighted mean 1. Falls back to uniform factors (with a + warning) when the ``ai_complementarity`` column is absent or its weighted + mean is non-positive. + """ + + n_recipients = int(recipients.sum()) + if COMPLEMENTARITY_COLUMN not in persons.columns: + warnings.warn( + f"Person table has no {COMPLEMENTARITY_COLUMN!r} column; applying a " + "uniform wage uplift instead of the eq 3.5 complementarity-graded " + "distribution.", + stacklevel=3, + ) + return np.ones(n_recipients), True + theta = persons[COMPLEMENTARITY_COLUMN].to_numpy(dtype=float)[recipients] + recipient_weights = weights[recipients] + mean_theta = _safe_share( + float((theta * recipient_weights).sum()), float(recipient_weights.sum()) + ) + if mean_theta <= 0.0: + warnings.warn( + f"Weighted mean {COMPLEMENTARITY_COLUMN!r} over wage-shock " + "recipients is non-positive; applying a uniform wage uplift.", + stacklevel=3, + ) + return np.ones(n_recipients), True + return theta / mean_theta, False + + +def _age_band_masks( + age: np.ndarray, age_bands: AgeBands +) -> list[tuple[str, np.ndarray]]: + """``(label, boolean mask)`` per age band, in band order.""" + + masks: list[tuple[str, np.ndarray]] = [] + for lower, upper in age_bands: + if upper is None: + masks.append((f"{lower}+", age >= lower)) + else: + masks.append((f"{lower}-{upper - 1}", (age >= lower) & (age < upper))) + return masks + + +def _age_band_summary( + draws: list[np.ndarray], + age: np.ndarray, + weights: np.ndarray, + employed: np.ndarray, + age_bands: AgeBands, +) -> dict[str, dict[str, float]]: + """Draw-averaged employment-shock breakdown per age band.""" + + result: dict[str, dict[str, float]] = {} + for label, band in _age_band_masks(age, age_bands): + employed_weight = float(weights[employed & band].sum()) + displaced_weight = float(np.mean([weights[d & band].sum() for d in draws])) + result[label] = { + "employed_weighted": employed_weight, + "displaced_count": float(np.mean([(d & band).sum() for d in draws])), + "displaced_weighted": displaced_weight, + "displaced_weighted_share": _safe_share(displaced_weight, employed_weight), + } + return result + + +def _group_summary( + draws: list[np.ndarray], + quotas: dict[str, float], + groups: np.ndarray, + weights: np.ndarray, + employed: np.ndarray, +) -> dict[str, dict[str, float]]: + """Draw-averaged employment-shock breakdown per occupation group.""" + + result: dict[str, dict[str, float]] = {} + for label, quota in sorted(quotas.items()): + in_group = employed & (groups == label) + employed_weight = float(weights[in_group].sum()) + displaced_weight = float(np.mean([weights[d & in_group].sum() for d in draws])) + result[label] = { + "employed_weighted": employed_weight, + "job_loss_quota_weighted": float(quota), + "displaced_weighted": displaced_weight, + "displaced_weighted_share": _safe_share(displaced_weight, employed_weight), + } + return result diff --git a/packages/populace-build/src/populace/build/uk_runtime/exposure_imputation.py b/packages/populace-build/src/populace/build/uk_runtime/exposure_imputation.py new file mode 100644 index 00000000..51e17bbe --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/exposure_imputation.py @@ -0,0 +1,538 @@ +"""UK AI-exposure imputation: LFS/APS-trained exposure scores onto FRS persons. + +The FRS *does* observe occupation — the UKDA ``adult.tab`` files carry SOC +2020 at **major-group** level (1-digit; stored as thousands, ``1000``-``9000``) +— but not the 4-digit unit group that task-level AI-exposure measures are +published at. So exposure arrives the same way SPI incomes do +(:mod:`populace.build.uk_runtime.spi_support`): a donor survey that observes +the needed detail trains a conditional model, and the model draws values for +the target frame's persons from shared covariates. + +**Design decision — refine within the true major group; impute the numeric +score, never the occupation code.** Two facts drive the design: + +1. The stack's canonical imputer (:class:`populace.fit.RegimeGatedQRF`) is + numeric-only by construction: every target is coerced to ``float64`` at + fit time, regimes are detected from sign support, and draws are *linear + interpolations* of quantile-forest predictions. A SOC code fed through + that pipeline would be treated as a continuous quantity — a draw could + land between two codes, inventing an occupation that does not exist. So + the imputed quantity is the real-valued exposure score, for which the + QRF's weighted conditional draws are exactly the right object. +2. The target person's 1-digit major group is *observed*, not modeled: it is + merged from the raw FRS ``adult.tab`` (``SOC2020``, linked via + SERNUM/BENUNIT/PERSON) in a separate data-prep step upstream of this + module. The imputer's job is therefore **not** to guess occupation from + demographics — it is to refine exposure *within* the known major group, + predicting where in the group's exposure distribution this person sits + from education, income, industry, age, and hours. ``soc_major_group`` is + accordingly the first (most important) entry of + :data:`DEFAULT_PREDICTORS`. As a numeric predictor the code's arbitrary + scale is harmless — forests split on it, they never interpolate it. + +Two companion paths bracket the model: + +- **Blind fallback** (documented, warned): a frame whose persons lack + ``soc_major_group`` can still be imputed from demographics alone by + passing predictors without it; :func:`fit_exposure_imputer` emits a + :class:`UserWarning` so a build log shows the degradation explicitly. +- **Zero-model baseline**: :func:`exposure_from_major_group` assigns every + person the employment-weighted mean exposure of their major group straight + from the crosswalk — no model at all. This is the transparent lower bound + reported alongside the QRF refinement; the gap between the two is the + within-group refinement, and their sensitivity is a robustness check. + +DONOR CONTRACT +============== +The donor passed to :func:`fit_exposure_imputer` must be an LFS/APS-derived +:class:`~populace.frame.Frame` (or DataFrame with explicit weights) whose +person table already carries the exposure score, produced by joining each +person's fine (4-digit) SOC 2020 code against the SOC->exposure crosswalk +(the ``populace.build.uk_runtime.ai_exposure`` module, developed on its own +branch; it is referenced lazily here so the two branches stay independent). + +The concrete donor this stage is built for is the **UKDS EUL Five-Quarter +Longitudinal LFS** panels (seven panels, Apr 2022 - Dec 2024; tab-delimited +``*.tab``, pooled n ~ 16.5k persons, ~10k with a wave-1 4-digit SOC across +311 unit groups). The donor-prep step maps its raw variables onto this +module's harmonized names: 4-digit SOC from ``SOC20M1`` (waves 2-5 +``SOC20M2..SOC20M5`` for panel dedup checks), ``age`` from ``AGE1``, +``gender`` from sex, ``highest_education`` from ``HIQUL22D`` (detail in +``HIQUAL22``), ``sic_industry_division`` from ``INDS07``, earnings from +``GRSSWK``/``USGRS99``, ``hours``/FT-PT from the hours block, ``region`` +from ``URESMC``, and the person weight from ``LGWT22``/``LGWT24``. The +interface stays donor-agnostic — any survey satisfying the column contract +below fits — but four caveats of *this* donor belong in every build that +uses it: + +1. **Sparse 4-digit cells**: at n ~ 16.5k many unit groups are thin. Attach + exposure at the 3-digit *minor* group by default, using the 4-digit score + only where the unit-group cell count meets a threshold, so no exposure + value rides on a handful of respondents. +2. **Panel overlap**: adjacent five-quarter panels share respondents; + dedupe on ``PERSID`` before fitting, or overlapping persons are silently + double-weighted. +3. **Vintage**: the 2023-24 panels are cleanest — ONS revised a 2021-22 + occupation-miscoding problem, so earlier panels' SOC codes carry known + error. +4. **Licensing**: the panels are UKDS End User Licence materials — the + project registration must cover this use, and the microdata must never + be committed (see below). + +Whatever the source survey, the donor must carry: + +- ``ai_exposure`` (or the ``target`` you name): the numeric exposure score, + finite for every row — persons without a valid SOC must be dropped or + resolved *before* fitting, never NaN-filled; +- ``soc_major_group``: the 1-digit SOC 2020 major group, derived from the + donor's fine SOC. Any consistent numeric coding works (``1``-``9`` or the + FRS's ``1000``-``9000``), but donor and target frames must use the *same* + coding — the model conditions on the value, it does not normalize it; +- every remaining predictor in :data:`DEFAULT_PREDICTORS` (or the + ``predictors`` you name), harmonized to the target frame's coding of the + same concepts: ``age`` (years), ``gender``, ``highest_education`` + (qualification band, ordered int), ``employment_income`` and + ``self_employment_income`` (annual GBP, same uprating vintage as the + target frame), ``hours`` (usual weekly hours), ``sic_industry_division`` + (SIC 2007 division), ``region``; +- design weights for the persons (the LFS/APS person weight), stored as the + frame's typed weights so the fit is weighted by construction. + +The LFS/APS and FRS microdata are UKDS End User Licence materials: donor +frames are built locally from licensed files and are **never committed** to +this repository. Tests exercise the contract with synthetic donors only. +""" + +from __future__ import annotations + +import warnings +from collections.abc import Sequence + +import numpy as np +import pandas as pd + +from populace.build.plan import DonorSpec, Stage +from populace.fit import DESIGN_WEIGHTS, FittedModel, WeightSpec, fit +from populace.frame import Frame + +__all__ = [ + "DEFAULT_EXPOSURE_COLUMN", + "DEFAULT_PREDICTORS", + "LFS_APS_EXPOSURE_DONOR", + "SOC_MAJOR_GROUP_COLUMN", + "UK_EXPOSURE_IMPUTATION_STAGE_NAME", + "attach_exposure", + "exposure_from_major_group", + "exposure_imputation_stage", + "fit_exposure_imputer", + "impute_exposure", +] + +#: The person-table column the imputed exposure score lands on. +DEFAULT_EXPOSURE_COLUMN = "ai_exposure" + +#: The observed 1-digit SOC 2020 major group on the person table. On the FRS +#: side it is merged from the raw UKDA ``adult.tab`` (``SOC2020``, values +#: ``1000``-``9000``, linked via SERNUM/BENUNIT/PERSON) in a separate +#: data-prep step; on the LFS/APS donor side it is derived from the fine SOC. +SOC_MAJOR_GROUP_COLUMN = "soc_major_group" + +#: Covariates shared by the LFS/APS donor and the FRS target frame, in the +#: target frame's coding (see the module docstring's DONOR CONTRACT). +#: ``soc_major_group`` leads because it is the observed, most informative +#: predictor: the model refines exposure *within* the known major group. +DEFAULT_PREDICTORS = ( + SOC_MAJOR_GROUP_COLUMN, + "age", + "gender", + "highest_education", + "employment_income", + "self_employment_income", + "hours", + "sic_industry_division", + "region", +) + +#: Stage name for the exposure-imputation step of a UK build plan. +UK_EXPOSURE_IMPUTATION_STAGE_NAME = "ai_exposure_imputation" + +#: The donor declaration a UK build plan records for this stage. The vintage +#: note is generic on purpose: the concrete quarter/wave is a property of the +#: locally held UKDS files, recorded by the build that loads them. +LFS_APS_EXPOSURE_DONOR = DonorSpec( + survey="ONS Labour Force Survey, Five-Quarter Longitudinal (EUL)", + source="UK Data Service (End User Licence), five-quarter longitudinal " + "LFS panels, Apr 2022 - Dec 2024; SOC 2020 exposure scores joined per " + "populace.build.uk_runtime.ai_exposure", + notes="Donor microdata are UKDS-licensed and never committed; the donor " + "frame is assembled locally with SOC->exposure already joined (see the " + "DONOR CONTRACT caveats: sparse 4-digit cells, PERSID dedup across " + "overlapping panels, 2023-24 panels preferred post ONS occupation " + "recoding). The target frame's soc_major_group comes from the raw FRS " + "adult.tab (SOC2020, major-group level), merged upstream of this stage.", +) + + +def _crosswalk_hint() -> str: + """Name where the donor's exposure column comes from, import-safely. + + The SOC->exposure crosswalk module (``ai_exposure``) is developed on its + own branch; importing it lazily — and only to sharpen an error message — + keeps this module usable before that branch lands. + """ + try: # pragma: no cover - exercised only once ai_exposure lands + from populace.build.uk_runtime import ai_exposure # noqa: F401 + + return ( + "join it with populace.build.uk_runtime.ai_exposure " + "(the SOC->exposure crosswalk) before fitting" + ) + except ImportError: + return ( + "join it via the SOC->exposure crosswalk " + "(populace.build.uk_runtime.ai_exposure, once that module lands) " + "before fitting" + ) + + +def _require_donor_columns( + donor: Frame | pd.DataFrame, predictors: Sequence[str], target: str +) -> None: + """Refuse a donor that does not satisfy the documented contract. + + :mod:`populace.fit` would also reject missing columns, but its message + cannot know *why* the column should exist; this check names the donor + contract and the crosswalk that produces the exposure column, so the + failure is actionable. + + Raises: + ValueError: Naming the missing columns and the fix. + """ + if isinstance(donor, Frame): + columns = { + column + for entity in donor.entities + for column in donor.table(entity).columns + } + else: + columns = set(donor.columns) + if target not in columns: + raise ValueError( + f"Donor is missing the exposure target column {target!r}. The " + "donor contract (see populace.build.uk_runtime.exposure_imputation) " + "requires an LFS/APS-derived frame with the SOC-level exposure " + f"score already attached; {_crosswalk_hint()}." + ) + missing = sorted(set(predictors) - columns) + if missing: + raise ValueError( + f"Donor is missing predictor column(s) {missing}. The donor " + "contract requires the shared LFS/APS-FRS covariates " + f"{list(predictors)}, harmonized to the target frame's coding " + "(see populace.build.uk_runtime.exposure_imputation)." + ) + + +def fit_exposure_imputer( + donor_frame: Frame | pd.DataFrame, + predictors: Sequence[str] = DEFAULT_PREDICTORS, + target: str = DEFAULT_EXPOSURE_COLUMN, + *, + weights: WeightSpec = DESIGN_WEIGHTS, + **model_kwargs, +) -> FittedModel: + """Fit the exposure imputer on an LFS/APS donor frame. + + A thin, contract-checking front door over :func:`populace.fit.fit`: the + single target is the numeric exposure score (see the module docstring for + why the score, not the SOC code, is the imputed quantity), and the fit is + weight-aware by construction — a Frame donor defaults to its typed design + weights (the LFS/APS person weight), a DataFrame donor must state its + weights explicitly, and ``weights="none"`` is the only unweighted path. + + The primary path conditions on the person's observed + :data:`SOC_MAJOR_GROUP_COLUMN` (refinement within the known major group). + Omitting it from ``predictors`` selects the documented *blind* fallback — + exposure from demographics alone — which still runs but emits a + :class:`UserWarning`, so a build log records the degradation. + + Args: + donor_frame: The LFS/APS-derived donor satisfying the DONOR CONTRACT + in the module docstring. + predictors: Conditioning covariates shared with the target frame. + Defaults to :data:`DEFAULT_PREDICTORS`. + target: The exposure column to learn. Defaults to + :data:`DEFAULT_EXPOSURE_COLUMN`. + weights: The fit's weight spec, per the + :class:`populace.fit.ConditionalModel` contract. + **model_kwargs: Forwarded to the canonical model (e.g. + ``n_estimators``, ``seed``). + + Returns: + A fitted :class:`populace.fit.FittedModel` whose single target is + ``target``. + + Raises: + ValueError: If the donor is missing the target or a predictor (the + message names the donor contract), or on any + :func:`populace.fit.fit` contract violation (non-finite target, + unresolvable weights, ...). + + Warns: + UserWarning: When ``predictors`` omit :data:`SOC_MAJOR_GROUP_COLUMN` + (the blind fallback path). + """ + predictors = list(predictors) + if SOC_MAJOR_GROUP_COLUMN not in predictors: + warnings.warn( + f"Fitting the exposure imputer without {SOC_MAJOR_GROUP_COLUMN!r}: " + "this is the blind fallback (exposure from demographics alone). " + "The FRS observes the 1-digit SOC 2020 major group in adult.tab; " + "merge it onto the person table upstream and include " + f"{SOC_MAJOR_GROUP_COLUMN!r} in predictors for the primary " + "within-group refinement path.", + UserWarning, + stacklevel=2, + ) + _require_donor_columns(donor_frame, predictors, target) + return fit(donor_frame, predictors, [target], weights=weights, **model_kwargs) + + +def impute_exposure(fitted: FittedModel, frame: Frame | pd.DataFrame) -> pd.Series: + """Draw one exposure score per person of ``frame`` from the fitted model. + + Args: + fitted: The model from :func:`fit_exposure_imputer`. + frame: The target :class:`~populace.frame.Frame` (drawing from its + person table) or a person-level DataFrame carrying the predictor + columns. + + Returns: + A :class:`pandas.Series` of exposure draws named after the fitted + target, index-aligned to the person rows. + + Raises: + ValueError: If ``fitted`` imputes anything but a single target (it + did not come from :func:`fit_exposure_imputer`), or a predictor + column is missing from ``frame``. + """ + drawn = fitted.predict(frame) + if len(drawn.columns) != 1: + raise ValueError( + "impute_exposure expects a model fitted on the single exposure " + f"target, but this model imputes {list(drawn.columns)}; fit it " + "with fit_exposure_imputer." + ) + return drawn[drawn.columns[0]] + + +def exposure_from_major_group( + soc_major_group: pd.Series | np.ndarray | Sequence[float], + crosswalk: pd.DataFrame, + *, + group_column: str = SOC_MAJOR_GROUP_COLUMN, + exposure_column: str = DEFAULT_EXPOSURE_COLUMN, + employment_column: str = "employment", +) -> pd.Series: + """Zero-imputation baseline: the major group's employment-weighted mean. + + No model at all: every person receives the employment-weighted mean + exposure of their observed 1-digit major group, computed straight from + the SOC->exposure crosswalk. This is the transparent lower bound reported + alongside the QRF refinement (:func:`fit_exposure_imputer` / + :func:`impute_exposure`); the difference between the two is exactly the + within-group refinement, and their sensitivity is a robustness check. + + Args: + soc_major_group: One major-group code per person (any consistent + numeric coding — ``1``-``9`` or the FRS's ``1000``-``9000`` — as + long as it matches ``crosswalk[group_column]`` exactly; codes are + matched, never normalized). + crosswalk: The SOC->exposure crosswalk at any SOC grain, one row per + occupation, carrying ``group_column`` (the occupation's major + group), ``exposure_column`` (its score), and + ``employment_column`` (its employment count/weight, the + aggregation weight). + group_column: The crosswalk's major-group column. Defaults to + :data:`SOC_MAJOR_GROUP_COLUMN`. + exposure_column: The crosswalk's score column. Defaults to + :data:`DEFAULT_EXPOSURE_COLUMN`. + employment_column: The crosswalk's employment-weight column. + + Returns: + A :class:`pandas.Series` named ``exposure_column``, one baseline + score per person, index-aligned to a Series input (positional for + array input). + + Raises: + ValueError: If the crosswalk is missing a column, carries non-finite + or negative employment, a group's total employment is zero, or a + person's code is absent from the crosswalk. Messages name the + culprits. + """ + missing = sorted( + {group_column, exposure_column, employment_column} - set(crosswalk.columns) + ) + if missing: + raise ValueError(f"crosswalk is missing column(s): {missing}.") + + employment = crosswalk[employment_column].to_numpy(dtype=np.float64) + if not np.isfinite(employment).all() or (employment < 0).any(): + raise ValueError( + f"crosswalk.{employment_column} must be finite and non-negative." + ) + exposure = crosswalk[exposure_column].to_numpy(dtype=np.float64) + if not np.isfinite(exposure).all(): + raise ValueError(f"crosswalk.{exposure_column} must be finite.") + + grouped = pd.DataFrame( + { + "group": crosswalk[group_column].to_numpy(), + "mass": employment * exposure, + "employment": employment, + } + ).groupby("group", sort=True) + totals = grouped[["mass", "employment"]].sum() + zero_groups = totals.index[totals["employment"] <= 0.0].tolist() + if zero_groups: + raise ValueError( + f"Major group(s) {zero_groups} have zero total employment in the " + "crosswalk; an employment-weighted mean is undefined there." + ) + means = totals["mass"] / totals["employment"] + + if isinstance(soc_major_group, pd.Series): + codes = soc_major_group + else: + codes = pd.Series(np.asarray(soc_major_group)) + unmatched = sorted(set(codes.unique()) - set(means.index)) + if unmatched: + raise ValueError( + f"soc_major_group code(s) {unmatched[:5]} are absent from the " + f"crosswalk's {group_column!r} values {means.index.tolist()}; " + "donor and target must use one consistent major-group coding " + "(codes are matched, never normalized)." + ) + out = codes.map(means) + out.name = exposure_column + return out + + +def attach_exposure( + frame: Frame, + values: pd.Series | np.ndarray | Sequence[float], + column: str = DEFAULT_EXPOSURE_COLUMN, +) -> Frame: + """Return a new frame with the exposure scores on the person table. + + Frames are immutable, so attachment reassembles: the person table gains + ``column`` and every other table, the typed weights, the strata, and the + mass log pass through untouched. Values are read **positionally** (a + Series' index is ignored), matching how :func:`impute_exposure` returns + draws aligned to the person rows. + + Args: + frame: The target frame. + values: One finite exposure score per person row, positionally + aligned to the person table. + column: The column name to attach under. Defaults to + :data:`DEFAULT_EXPOSURE_COLUMN`. + + Returns: + A new validated :class:`~populace.frame.Frame`. + + Raises: + ValueError: If ``column`` already exists on any entity table (the + stage must have one canonical producer, and silently overwriting + scores would hide a double run), or ``values`` is not 1-D, has + the wrong length, or contains non-finite entries. + """ + try: + owner = frame.column_entity(column) + except ValueError: + owner = None + if owner is not None: + raise ValueError( + f"Column {column!r} already exists on the {owner!r} table; " + "attach_exposure refuses to overwrite it. The exposure stage " + "should run exactly once — pass a different column name to " + "attach a second draw deliberately." + ) + + if isinstance(values, pd.Series): + array = values.to_numpy(dtype=np.float64) + else: + array = np.asarray(values, dtype=np.float64) + if array.ndim != 1: + raise ValueError(f"values must be 1-D, got shape {array.shape}.") + person_entity = frame.schema.person_entity + n_persons = frame.n(person_entity) + if len(array) != n_persons: + raise ValueError( + f"values has {len(array)} entries but the {person_entity!r} table " + f"has {n_persons} row(s); exposure draws must align positionally." + ) + non_finite = int((~np.isfinite(array)).sum()) + if non_finite: + raise ValueError( + f"values contains {non_finite} non-finite entr(ies); exposure " + "scores must be finite (the donor contract forbids NaN exposure)." + ) + + person = frame.person.copy() + person[column] = array + tables: dict[str, pd.DataFrame] = {person_entity: person} + for entity in frame.schema.group_entities: + tables[entity] = frame.table(entity) + for link in frame.links: + tables[link] = frame.link(link) + weights = {entity: frame.weights_for(entity) for entity in frame.weighted_entities} + return Frame( + tables, + frame.schema, + weights, + frame.strata, + mass_log=frame.mass_log, + ) + + +def exposure_imputation_stage( + fitted: FittedModel, + *, + column: str = DEFAULT_EXPOSURE_COLUMN, + donor: DonorSpec = LFS_APS_EXPOSURE_DONOR, +) -> Stage: + """Declare the exposure-imputation step of a UK build plan. + + The donor fit happens *before* plan assembly (the LFS/APS donor is a + licensed local artifact, not a frame column, so the plan's + consumes/produces bookkeeping cannot see it); the stage closes over the + fitted model and, when run, draws for the frame's persons and attaches + the scores. The stage consumes the model's predictors — on the primary + path that includes :data:`SOC_MAJOR_GROUP_COLUMN`, so the plan executor + refuses to run before the FRS adult.tab merge has put it on the frame. + Any failure inside — missing predictors, a double run — aborts the + build, per the plan executor's loudness rules. + + Args: + fitted: The model from :func:`fit_exposure_imputer`. + column: The produced person column. Defaults to + :data:`DEFAULT_EXPOSURE_COLUMN`. + donor: The donor declaration recorded on the stage. Defaults to + :data:`LFS_APS_EXPOSURE_DONOR`. + + Returns: + A :class:`populace.build.plan.Stage` consuming the model's predictors + and producing ``column``. + """ + + def transform(frame: Frame) -> Frame: + return attach_exposure(frame, impute_exposure(fitted, frame), column=column) + + return Stage( + name=UK_EXPOSURE_IMPUTATION_STAGE_NAME, + transform=transform, + produces=(column,), + consumes=tuple(fitted.predictors), + donor=donor, + ) diff --git a/packages/populace-build/src/populace/build/uk_runtime/frs_occupation.py b/packages/populace-build/src/populace/build/uk_runtime/frs_occupation.py new file mode 100644 index 00000000..8353ce1f --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/frs_occupation.py @@ -0,0 +1,316 @@ +"""Observed FRS occupation (SOC 2020 major group) onto the UK person table. + +Provenance +---------- +The UKDA Family Resources Survey (UK Data Service, **End User Licence**) +ships an ``adult.tab`` file with one row per adult respondent, keyed by +``SERNUM`` (household serial number), ``BENUNIT`` (benefit unit within the +household) and ``PERSON`` (person within the benefit unit). Its ``SOC2020`` +variable carries the respondent's Standard Occupational Classification 2020 +at **major-group** level, coded as thousands: ``1000`` (managers, directors +and senior officials) through ``9000`` (elementary occupations). Like all +UKDS EUL microdata (see :mod:`populace.build.uk_runtime.spi_support` and the +DONOR CONTRACT in :mod:`populace.build.uk_runtime.exposure_imputation`), the +file lives outside the repository and its path is supplied by the caller — +the raw microdata are **never committed**; tests exercise the contract with +synthetic tables only. + +Join keys +--------- +PolicyEngine-UK single-year person IDs are derived from the same FRS keys: +``person_id = SERNUM * 100 + BENUNIT * 10 + PERSON`` (with ``benunit_id = +SERNUM * 10 + BENUNIT`` and ``household_id = SERNUM``). This module rebuilds +that composite ID from ``adult.tab``'s key columns and joins it against the +person table's ``person_id`` (the first of +:data:`populace.build.uk_runtime.rowwise_dataset.PERSON_ID_COLUMNS`), the +same single-year ID surface :mod:`~populace.build.uk_runtime.spi_support` +clones from. + +Missingness contract +-------------------- +* Children have no ``adult.tab`` row, so they receive NaN. +* Adults whose ``SOC2020`` is missing or invalid (anything outside the + ``1000``-``9000`` major-group coding, e.g. FRS sentinel ``-1`` or ``0``) + keep NaN; the report counts them so a build log shows the gap. + +Contract with exposure imputation +--------------------------------- +The produced person column is +:data:`~populace.build.uk_runtime.exposure_imputation.SOC_MAJOR_GROUP_COLUMN` +(``soc_major_group``), kept in the **raw ``1000``-``9000`` coding**: +:mod:`~populace.build.uk_runtime.exposure_imputation` documents that any +consistent numeric coding works as long as donor and target match, and +:func:`populace.build.uk_runtime.ai_exposure.exposure_for_major_group` +accepts the coding via :func:`frs_major_group_to_digit` (or directly after +that normalisation). :data:`UK_FRS_OCCUPATION_STAGE_NAME` produces the +column, so in a :class:`~populace.build.plan.StagePlan` this stage slots +immediately before +:func:`~populace.build.uk_runtime.exposure_imputation.exposure_imputation_stage`, +whose primary path consumes it. +""" + +from __future__ import annotations + +from pathlib import Path +from typing import Any + +import numpy as np +import pandas as pd + +from populace.build.plan import DonorSpec, Stage +from populace.build.uk_runtime.exposure_imputation import SOC_MAJOR_GROUP_COLUMN +from populace.frame import Frame + +__all__ = [ + "FRS_ADULT_KEY_COLUMNS", + "FRS_ADULT_SOC_COLUMN", + "FRS_OCCUPATION_DONOR", + "SOC_MAJOR_GROUP_COLUMN", + "UK_FRS_OCCUPATION_STAGE_NAME", + "VALID_SOC_MAJOR_GROUPS", + "attach_soc_major_group", + "frs_major_group_to_digit", + "frs_occupation_stage", + "frs_person_ids", + "load_frs_adult_table", + "soc_major_group_for_persons", +] + +#: The ``adult.tab`` columns identifying one adult respondent. +FRS_ADULT_KEY_COLUMNS = ("SERNUM", "BENUNIT", "PERSON") + +#: The ``adult.tab`` SOC 2020 major-group variable (coded 1000-9000). +FRS_ADULT_SOC_COLUMN = "SOC2020" + +#: The valid FRS major-group codes: thousands 1000 (managers) to 9000 +#: (elementary occupations). Anything else (sentinels like -1, 0, or +#: missing) is treated as missing SOC. +VALID_SOC_MAJOR_GROUPS = tuple(range(1000, 10_000, 1000)) + +#: Stage name for the FRS occupation merge step of a UK build plan. +UK_FRS_OCCUPATION_STAGE_NAME = "frs_occupation" + +#: The donor declaration a UK build plan records for this stage. +FRS_OCCUPATION_DONOR = DonorSpec( + survey="DWP Family Resources Survey (UKDA, End User Licence)", + source="UK Data Service, Family Resources Survey adult.tab (SOC2020 " + "major-group variable, keyed SERNUM/BENUNIT/PERSON)", + notes="UKDS EUL microdata, held locally and never committed. This is an " + "observed merge, not an imputation: the adult's own 1-digit SOC 2020 " + "major group is joined onto the PolicyEngine-UK single-year person IDs. " + "Children and adults with missing/invalid SOC2020 stay NaN.", +) + +_MISSING_SOC_WARN_KEY = "adults_missing_soc" + + +def load_frs_adult_table( + path: str | Path, + *, + columns: tuple[str, ...] = (*FRS_ADULT_KEY_COLUMNS, FRS_ADULT_SOC_COLUMN), +) -> pd.DataFrame: + """Read the UKDA FRS ``adult.tab`` (tab-delimited), keeping ``columns``. + + The path points at locally held UKDS End User Licence microdata; it is + supplied by the caller and never committed. + + Raises: + ValueError: If the file lacks any of ``columns``. + """ + + table = pd.read_csv(path, sep="\t", usecols=lambda name: name in set(columns)) + missing = sorted(set(columns) - set(table.columns)) + if missing: + raise ValueError(f"FRS adult table at {path} is missing column(s): {missing}.") + return table + + +def frs_person_ids(adult: pd.DataFrame) -> np.ndarray: + """PolicyEngine-UK single-year person IDs from FRS adult key columns. + + ``person_id = SERNUM * 100 + BENUNIT * 10 + PERSON`` — the composite the + PolicyEngine-UK FRS build assigns, and the ``person_id`` the single-year + person table carries (see PERSON_ID_COLUMNS in + :mod:`populace.build.uk_runtime.rowwise_dataset`). + + Raises: + ValueError: On missing key columns or missing/non-numeric key values. + """ + + missing = sorted(set(FRS_ADULT_KEY_COLUMNS) - set(adult.columns)) + if missing: + raise ValueError(f"adult table is missing key column(s): {missing}.") + keys = {} + for column in FRS_ADULT_KEY_COLUMNS: + values = pd.to_numeric(adult[column], errors="raise") + if values.isna().any(): + raise ValueError(f"adult.{column} contains missing values.") + keys[column] = values.astype("int64").to_numpy() + return keys["SERNUM"] * 100 + keys["BENUNIT"] * 10 + keys["PERSON"] + + +def soc_major_group_for_persons( + person: pd.DataFrame, + adult: pd.DataFrame, + *, + person_id_column: str = "person_id", +) -> tuple[pd.Series, dict[str, int]]: + """Person-level ``soc_major_group`` series joined from ``adult.tab``. + + Joins the adult table onto ``person[person_id_column]`` via + :func:`frs_person_ids` and keeps the raw ``1000``-``9000`` coding + (:mod:`~populace.build.uk_runtime.exposure_imputation` documents that + coding, and :func:`~populace.build.uk_runtime.ai_exposure.exposure_for_major_group` + accepts it after :func:`frs_major_group_to_digit`). + + Returns: + ``(series, report)`` — a float series named + :data:`SOC_MAJOR_GROUP_COLUMN`, index-aligned to ``person``, NaN for + persons without an adult row (children) or with missing/invalid + ``SOC2020``; and a count report with keys ``n_persons``, + ``adults_matched``, ``adults_with_valid_soc``, + ``adults_missing_soc`` and ``persons_without_adult_row``. + + Raises: + ValueError: On missing columns or duplicated adult person IDs. + """ + + if person_id_column not in person.columns: + raise ValueError(f"person table is missing {person_id_column!r}.") + if FRS_ADULT_SOC_COLUMN not in adult.columns: + raise ValueError(f"adult table is missing {FRS_ADULT_SOC_COLUMN!r}.") + + adult_ids = frs_person_ids(adult) + if pd.Index(adult_ids).has_duplicates: + duplicates = pd.Index(adult_ids) + duplicates = duplicates[duplicates.duplicated()].unique() + raise ValueError( + "adult table has duplicated SERNUM/BENUNIT/PERSON keys; duplicate " + f"person ID(s): {list(map(str, duplicates[:5]))}." + ) + + soc = pd.to_numeric(adult[FRS_ADULT_SOC_COLUMN], errors="coerce").astype(float) + soc = soc.where(soc.isin(VALID_SOC_MAJOR_GROUPS)) + lookup = pd.Series(soc.to_numpy(), index=adult_ids) + + person_ids = pd.to_numeric(person[person_id_column], errors="raise").astype("int64") + values = person_ids.map(lookup) + values.name = SOC_MAJOR_GROUP_COLUMN + + matched = person_ids.isin(lookup.index) + report = { + "n_persons": int(len(person)), + "adults_matched": int(matched.sum()), + "adults_with_valid_soc": int(values.notna().sum()), + _MISSING_SOC_WARN_KEY: int((matched & values.isna()).sum()), + "persons_without_adult_row": int((~matched).sum()), + } + return values, report + + +def frs_major_group_to_digit(codes: Any) -> np.ndarray: + """Normalise FRS ``1000``-``9000`` major-group codes to ``"1"``-``"9"``. + + The 1-digit strings are what + :func:`populace.build.uk_runtime.ai_exposure.exposure_for_major_group` + indexes its major-group table by. Invalid/missing codes become ``None``. + """ + + values = pd.to_numeric(pd.Series(np.asarray(codes)), errors="coerce") + digits = values.where(values.isin(VALID_SOC_MAJOR_GROUPS)) / 1000 + return np.array( + [None if pd.isna(value) else str(int(value)) for value in digits], + dtype=object, + ) + + +def attach_soc_major_group( + frame: Frame, + adult: pd.DataFrame, + *, + column: str = SOC_MAJOR_GROUP_COLUMN, +) -> Frame: + """Return a new frame whose person table carries ``soc_major_group``. + + Mirrors :func:`~populace.build.uk_runtime.exposure_imputation.attach_exposure`'s + reassembly, but NaN is *allowed* here — it is the documented value for + children and for adults with missing/invalid SOC2020. + + Raises: + ValueError: If ``column`` already exists on any entity table (the + stage must run exactly once). + """ + + try: + owner = frame.column_entity(column) + except ValueError: + owner = None + if owner is not None: + raise ValueError( + f"Column {column!r} already exists on the {owner!r} table; " + "attach_soc_major_group refuses to overwrite it. The FRS " + "occupation stage should run exactly once." + ) + + values, _report = soc_major_group_for_persons(frame.person, adult) + person_entity = frame.schema.person_entity + person = frame.person.copy() + person[column] = values.to_numpy(dtype=float) + tables: dict[str, pd.DataFrame] = {person_entity: person} + for entity in frame.schema.group_entities: + tables[entity] = frame.table(entity) + for link in frame.links: + tables[link] = frame.link(link) + weights = {entity: frame.weights_for(entity) for entity in frame.weighted_entities} + return Frame( + tables, + frame.schema, + weights, + frame.strata, + mass_log=frame.mass_log, + ) + + +def frs_occupation_stage( + adult: pd.DataFrame | str | Path, + *, + column: str = SOC_MAJOR_GROUP_COLUMN, + donor: DonorSpec = FRS_OCCUPATION_DONOR, +) -> Stage: + """Declare the FRS occupation-merge step of a UK build plan. + + Modeled on the :mod:`~populace.build.uk_runtime.spi_support` / + :func:`~populace.build.uk_runtime.exposure_imputation.exposure_imputation_stage` + pattern: the licensed local artifact (the ``adult.tab`` path or a + pre-loaded frame) is closed over, and the stage, when run, joins the + observed major group onto the frame's person table. It produces + ``(column,)`` so it slots before ``exposure_imputation_stage`` — whose + primary path consumes :data:`SOC_MAJOR_GROUP_COLUMN` — in a + :class:`~populace.build.plan.StagePlan`. + + Args: + adult: The FRS ``adult.tab`` path (read lazily inside the transform, + so a bad path aborts the build loudly at run time) or an + already-loaded adult DataFrame. + column: The produced person column. Defaults to + :data:`SOC_MAJOR_GROUP_COLUMN`. + donor: The donor declaration recorded on the stage. + + Returns: + A :class:`populace.build.plan.Stage` producing ``column`` and + consuming ``person_id`` (the single-year ID the join keys on). + """ + + def transform(frame: Frame) -> Frame: + table = ( + adult if isinstance(adult, pd.DataFrame) else load_frs_adult_table(adult) + ) + return attach_soc_major_group(frame, table, column=column) + + return Stage( + name=UK_FRS_OCCUPATION_STAGE_NAME, + transform=transform, + produces=(column,), + consumes=("person_id",), + donor=donor, + ) diff --git a/packages/populace-build/src/populace/build/uk_runtime/occupation_targets.py b/packages/populace-build/src/populace/build/uk_runtime/occupation_targets.py new file mode 100644 index 00000000..0a24cfd5 --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/occupation_targets.py @@ -0,0 +1,477 @@ +"""UK occupation employment calibration targets from ONS ASHE and the APS. + +Two source families are supported, each parsed from a tidy CSV produced from +the published release (the raw workbooks/API pulls live outside the repo): + +* **ASHE Table 14** — employee-job counts by four-digit SOC2020 unit group, + from the Annual Survey of Hours and Earnings occupation tables + (Table 14.7a, "Annual pay - Gross"). Tidy columns: ``soc_code``, + ``soc_title``, ``employment_jobs``, ``median_annual_pay``. Only the job + counts become targets: a median is not a sum constraint, so + ``median_annual_pay`` is carried for provenance/diagnostics only. +* **Annual Population Survey via Nomis** — employment by SOC2020 sub-major + occupation group (dataset ``NM_17_1``, table ``T09b``) and, separately, + employment by age band on the scheme ``16-24``/``25-34``/``35-44``/ + ``45-54``/``55-64``/``65+`` (plus the ``16+`` roll-up), matching the + :mod:`~populace.build.uk_runtime.ai_shock_scenarios` reporting bands. + ``NM_17_1`` table ``T01`` only publishes 16-19/20-24/25-34/35-49/50-64/65+, + which cannot be reconciled with that scheme, so the age bands are derived + by summing the five-year bands of the companion APS dataset ``NM_170_1`` + (labour market status by age, ``C_ECOPUK11`` = Employed, all persons); + its marginals agree exactly with ``T01`` where the schemes coincide. The + APS publishes no occupation-by-age cross-tabulation, so the two margins + are declared as independent target families. Tidy columns: ``soc_code``, + ``soc_title``, ``employment``, ``period``, ``geography`` (occupation) and + ``age_band``, ``employment``, ``period``, ``geography`` (age). + +Every builder returns an :class:`OccupationTargetBuild`: the compiled +:class:`~populace.calibrate.TargetSet` plus the rows that could not become +targets, each with a reason — bad rows are skipped and reported, never +silently dropped. Structural problems (missing columns, empty files) raise. + +Measures are prepared indicator/count columns on the ``person`` entity: +``employee_jobs/occupation/`` for ASHE job counts (a person may hold +more than one job, so the column is a job count, not a 0/1 flag), +``employment/occupation_submajor/`` and ``employment/age/`` for +the APS person counts. +""" + +from __future__ import annotations + +import csv +import math +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from importlib.resources import files +from pathlib import Path + +from populace.calibrate import Target, TargetSet + +__all__ = [ + "ASHE_TABLE14_SOURCE", + "APS_NOMIS_AGE_SOURCE", + "APS_NOMIS_SOURCE", + "PACKAGED_APS_AGE_CSV", + "PACKAGED_APS_OCCUPATION_CSV", + "PACKAGED_ASHE_TABLE14_CSV", + "OccupationTargetBuild", + "SkippedTargetRow", + "aps_age_band_employment_targets", + "aps_occupation_employment_targets", + "ashe_occupation_employment_targets", + "packaged_occupation_csv_path", +] + +#: Provenance for the ASHE Table 14 vintage the tidy CSV was parsed from. +ASHE_TABLE14_SOURCE = ( + "ONS Annual Survey of Hours and Earnings (ASHE) Table 14.7a " + "'Annual pay - Gross', occupation by four-digit SOC2020, United Kingdom, " + "2025 provisional edition, released 23 October 2025, ons.gov.uk" +) + +#: Provenance for the APS occupation pulls from the Nomis API (NM_17_1). +APS_NOMIS_SOURCE = ( + "ONS Annual Population Survey via Nomis API dataset NM_17_1 (www.nomisweb.co.uk)" +) + +#: Provenance for the APS age-band pulls from the Nomis API (NM_170_1). +APS_NOMIS_AGE_SOURCE = ( + "ONS Annual Population Survey via Nomis API dataset NM_170_1 (www.nomisweb.co.uk)" +) + +#: The all-ages roll-up band published alongside the disjoint APS age bands. +APS_TOTAL_AGE_BAND = "16+" + +#: Filenames of the tidy CSVs committed under +#: ``populace/build/uk_runtime/occupation_targets_data``. +PACKAGED_ASHE_TABLE14_CSV = "ashe_table14_2025_soc4.csv" +PACKAGED_APS_OCCUPATION_CSV = "nomis_aps_occupation_employment.csv" +PACKAGED_APS_AGE_CSV = "nomis_aps_age_employment.csv" + + +def packaged_occupation_csv_path(filename: str) -> Path: + """Filesystem path of a tidy CSV committed with the package. + + Args: + filename: One of the ``PACKAGED_*_CSV`` constants. + + Returns: + The path of the packaged tidy CSV under + ``populace/build/uk_runtime/occupation_targets_data``. + + Raises: + FileNotFoundError: If ``filename`` is not a packaged tidy CSV. + """ + + path = Path( + str( + files("populace.build.uk_runtime").joinpath( + "occupation_targets_data", filename + ) + ) + ) + if not path.is_file(): + raise FileNotFoundError( + f"No packaged tidy CSV {filename!r} under " + "populace/build/uk_runtime/occupation_targets_data." + ) + return path + + +@dataclass(frozen=True) +class SkippedTargetRow: + """One tidy-CSV row that could not become a calibration target. + + Attributes: + row_number: 1-based data-row position in the CSV (header excluded). + reason: Human-readable explanation naming the culprit field. + row: The offending row, as read. + """ + + row_number: int + reason: str + row: Mapping[str, str] + + +@dataclass(frozen=True) +class OccupationTargetBuild: + """A compiled occupation target set plus its skipped-row report. + + Attributes: + target_set: The targets that compiled cleanly. + skipped: Rows excluded from ``target_set``, each with a reason. + Callers deciding whether a build is usable must consult this — + an empty tuple means every row became a target. + """ + + target_set: TargetSet + skipped: tuple[SkippedTargetRow, ...] + + +def ashe_occupation_employment_targets( + csv_path: str | Path, + *, + period: int | str = 2025, + source: str = ASHE_TABLE14_SOURCE, +) -> OccupationTargetBuild: + """Employee-job count targets by four-digit SOC2020 from ASHE Table 14. + + Rows with a suppressed or non-numeric ``employment_jobs`` (ASHE marks + suppressed cells ``x``; the tidy CSV leaves them blank) and rows whose + ``soc_code`` is not a four-digit code are skipped and reported. + + Args: + csv_path: Tidy CSV with columns ``soc_code``, ``soc_title``, + ``employment_jobs``, ``median_annual_pay``. + period: Period tag for the targets (the ASHE survey year). + source: Provenance string recorded on every target. + + Returns: + The compiled build: one ``person``-entity target per usable unit + group, measured by the prepared job-count column + ``employee_jobs/occupation/``. + + Raises: + ValueError: If the CSV is missing required columns or has no rows. + """ + + rows = _read_rows(csv_path, required=("soc_code", "soc_title", "employment_jobs")) + targets: list[Target] = [] + skipped: list[SkippedTargetRow] = [] + seen: dict[str, int] = {} + for row_number, row in rows: + soc_code = row["soc_code"].strip() + if not (len(soc_code) == 4 and soc_code.isdigit()): + skipped.append( + SkippedTargetRow( + row_number, + f"soc_code {soc_code!r} is not a four-digit SOC2020 " + "unit-group code", + row, + ) + ) + continue + if soc_code in seen: + skipped.append( + SkippedTargetRow( + row_number, + f"duplicate soc_code {soc_code!r} (first seen at data row " + f"{seen[soc_code]})", + row, + ) + ) + continue + jobs = _parse_count(row.get("employment_jobs", "")) + if jobs is None: + skipped.append( + SkippedTargetRow( + row_number, + "employment_jobs " + f"{row.get('employment_jobs', '')!r} is missing or not a " + "non-negative number (ASHE suppression is blank in the " + "tidy CSV)", + row, + ) + ) + continue + seen[soc_code] = row_number + targets.append( + Target( + name=f"ons/ashe14/occupation/{soc_code}/employment_jobs", + entity="person", + measure=f"employee_jobs/occupation/{soc_code}", + value=jobs, + period=period, + source=f"{source}; unit group {soc_code} {row['soc_title']}", + ) + ) + return _finish(csv_path, targets, skipped) + + +def aps_occupation_employment_targets( + csv_path: str | Path, + *, + source: str = APS_NOMIS_SOURCE, +) -> OccupationTargetBuild: + """Employment targets by SOC2020 sub-major group from the APS (Nomis). + + Args: + csv_path: Tidy CSV with columns ``soc_code`` (two-digit sub-major + group), ``soc_title``, ``employment``, ``period`` (e.g. + ``"Jan 2025-Dec 2025"``); an optional ``geography`` column is + appended to the provenance. + source: Provenance prefix recorded on every target. + + Returns: + The compiled build: one ``person``-entity target per usable + sub-major group, measured by the prepared indicator column + ``employment/occupation_submajor/``, with the row's APS + period string as the target period. + + Raises: + ValueError: If the CSV is missing required columns or has no rows. + """ + + rows = _read_rows( + csv_path, required=("soc_code", "soc_title", "employment", "period") + ) + targets: list[Target] = [] + skipped: list[SkippedTargetRow] = [] + seen: dict[str, int] = {} + for row_number, row in rows: + soc_code = row["soc_code"].strip() + if not (len(soc_code) == 2 and soc_code.isdigit()): + skipped.append( + SkippedTargetRow( + row_number, + f"soc_code {soc_code!r} is not a two-digit SOC2020 " + "sub-major group code", + row, + ) + ) + continue + if soc_code in seen: + skipped.append( + SkippedTargetRow( + row_number, + f"duplicate soc_code {soc_code!r} (first seen at data row " + f"{seen[soc_code]})", + row, + ) + ) + continue + employment = _parse_count(row.get("employment", "")) + if employment is None: + skipped.append( + SkippedTargetRow( + row_number, + f"employment {row.get('employment', '')!r} is missing or " + "not a non-negative number", + row, + ) + ) + continue + period = row["period"].strip() + if not period: + skipped.append(SkippedTargetRow(row_number, "period is blank", row)) + continue + seen[soc_code] = row_number + targets.append( + Target( + name=f"ons/aps/occupation_submajor/{soc_code}/employment", + entity="person", + measure=f"employment/occupation_submajor/{soc_code}", + value=employment, + period=period, + source=_aps_row_source(source, row, f"table T09b, {period}") + + f"; sub-major group {soc_code} {row['soc_title']}", + ) + ) + return _finish(csv_path, targets, skipped) + + +def aps_age_band_employment_targets( + csv_path: str | Path, + *, + include_total: bool = False, + source: str = APS_NOMIS_AGE_SOURCE, +) -> OccupationTargetBuild: + """In-employment person-count targets by age band from the APS (Nomis). + + The band scheme is ``16-24``/``25-34``/``35-44``/``45-54``/``55-64``/ + ``65+`` — the same reporting bands used by + :mod:`~populace.build.uk_runtime.ai_shock_scenarios` — derived by summing + the NM_170_1 five-year bands (see the module docstring for why NM_17_1 + table T01 cannot supply this scheme directly). The APS publishes no + occupation-by-age cross-tabulation, so this age margin complements + :func:`aps_occupation_employment_targets` as an independent family. The + published ``16+`` roll-up duplicates the sum of the disjoint bands (up to + the publication rounding to the nearest hundred); it is excluded by + default and reported as skipped so the exclusion is visible. + + Args: + csv_path: Tidy CSV with columns ``age_band`` (``16-24`` … ``65+``, + plus the ``16+`` total), ``employment``, ``period``; an optional + ``geography`` column is appended to the provenance. + include_total: Keep the redundant ``16+`` roll-up as a target. + source: Provenance prefix recorded on every target. + + Returns: + The compiled build: one ``person``-entity target per usable band, + measured by the prepared indicator column + ``employment/age/`` (band with ``-`` as ``_`` and ``+`` as + ``plus``, e.g. ``employment/age/16_19``, ``employment/age/65plus``). + + Raises: + ValueError: If the CSV is missing required columns or has no rows. + """ + + rows = _read_rows(csv_path, required=("age_band", "employment", "period")) + targets: list[Target] = [] + skipped: list[SkippedTargetRow] = [] + seen: dict[str, int] = {} + for row_number, row in rows: + age_band = row["age_band"].strip() + if not age_band: + skipped.append(SkippedTargetRow(row_number, "age_band is blank", row)) + continue + if age_band == APS_TOTAL_AGE_BAND and not include_total: + skipped.append( + SkippedTargetRow( + row_number, + f"age_band {APS_TOTAL_AGE_BAND!r} is the roll-up of the " + "disjoint bands; excluded by default (pass " + "include_total=True to keep it)", + row, + ) + ) + continue + if age_band in seen: + skipped.append( + SkippedTargetRow( + row_number, + f"duplicate age_band {age_band!r} (first seen at data row " + f"{seen[age_band]})", + row, + ) + ) + continue + employment = _parse_count(row.get("employment", "")) + if employment is None: + skipped.append( + SkippedTargetRow( + row_number, + f"employment {row.get('employment', '')!r} is missing or " + "not a non-negative number", + row, + ) + ) + continue + period = row["period"].strip() + if not period: + skipped.append(SkippedTargetRow(row_number, "period is blank", row)) + continue + seen[age_band] = row_number + band_token = age_band.replace("-", "_").replace("+", "plus") + targets.append( + Target( + name=f"ons/aps/employment/age/{band_token}", + entity="person", + measure=f"employment/age/{band_token}", + value=employment, + period=period, + source=_aps_row_source( + source, + row, + "employment by age, NM_170_1 five-year bands summed to " + f"the ai_shock_scenarios reporting bands, {period}", + ) + + f"; in employment, aged {age_band}", + ) + ) + return _finish(csv_path, targets, skipped) + + +def _read_rows( + csv_path: str | Path, *, required: Sequence[str] +) -> list[tuple[int, dict[str, str]]]: + """Read a tidy CSV, enforcing the structural contract. + + Returns ``(data_row_number, row)`` pairs; raises on missing columns or an + empty file — structural breakage is an error, not a skippable row. + """ + + csv_path = Path(csv_path) + with csv_path.open(newline="", encoding="utf-8") as handle: + reader = csv.DictReader(handle) + fieldnames = tuple(reader.fieldnames or ()) + missing = [name for name in required if name not in fieldnames] + if missing: + raise ValueError( + f"{csv_path}: missing required column(s) {missing}; " + f"found {list(fieldnames)}." + ) + rows = [ + (row_number, {key: (value or "") for key, value in row.items()}) + for row_number, row in enumerate(reader, start=1) + ] + if not rows: + raise ValueError(f"{csv_path}: no data rows.") + return rows + + +def _parse_count(text: str) -> float | None: + """Parse a non-negative count; ``None`` marks the row skippable.""" + + text = text.strip() + if not text: + return None + try: + value = float(text) + except ValueError: + return None + if not math.isfinite(value) or value < 0: + return None + return value + + +def _aps_row_source(source: str, row: Mapping[str, str], detail: str) -> str: + geography = row.get("geography", "").strip() + suffix = f", {geography}" if geography else "" + return f"{source}; {detail}{suffix}" + + +def _finish( + csv_path: str | Path, + targets: list[Target], + skipped: list[SkippedTargetRow], +) -> OccupationTargetBuild: + if not targets: + reasons = "; ".join( + f"row {entry.row_number}: {entry.reason}" for entry in skipped[:5] + ) + raise ValueError( + f"{csv_path}: every row was skipped, no targets built. " + f"First reasons: {reasons}" + ) + return OccupationTargetBuild(target_set=TargetSet(targets), skipped=tuple(skipped)) diff --git a/packages/populace-build/src/populace/build/uk_runtime/occupation_targets_data/ashe_table14_2025_soc4.csv b/packages/populace-build/src/populace/build/uk_runtime/occupation_targets_data/ashe_table14_2025_soc4.csv new file mode 100644 index 00000000..d668162e --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/occupation_targets_data/ashe_table14_2025_soc4.csv @@ -0,0 +1,413 @@ +soc_code,soc_title,employment_jobs,median_annual_pay +1111,Chief executives and senior officials,133000,89835.0 +1112,Elected officers and representatives,, +1121,Production managers and directors in manufacturing,479000,52885.0 +1122,Production managers and directors in construction,102000,54947.0 +1123,Production managers and directors in mining and energy,10000,63241.0 +1131,Financial managers and directors,430000,65336.0 +1132,"Marketing, sales and advertising directors",216000,90000.0 +1133,Public relations and communications directors,,72020.0 +1134,Purchasing managers and directors,58000,56779.0 +1135,Charitable organisation managers and directors,, +1136,Human resource managers and directors,171000,54474.0 +1137,Information technology directors,60000,90081.0 +1139,Functional managers and directors n.e.c.,136000,69996.0 +1140,"Directors in logistics, warehousing and transport",11000,80518.0 +1150,Managers and directors in retail and wholesale,336000,36006.0 +1161,Officers in armed forces,, +1162,Senior police officers,14000,66514.0 +1163,"Senior officers in fire, ambulance, prison and related services",, +1171,Health services and public health managers and directors,64000,55879.0 +1172,Social services managers and directors,13000,45155.0 +1211,Managers and proprietors in agriculture and horticulture,16000,34976.0 +1212,"Managers and proprietors in forestry, fishing and related services",,31126.0 +1221,Hotel and accommodation managers and proprietors,17000,33008.0 +1222,Restaurant and catering establishment managers and proprietors,46000,30513.0 +1223,Publicans and managers of licensed premises,,37427.0 +1224,Leisure and sports managers,32000,33342.0 +1225,Travel agency managers and proprietors,,34505.0 +1231,Health care practice managers,15000,38941.0 +1232,"Residential, day and domiciliary care managers and proprietors",58000,40661.0 +1233,Early education and childcare services proprietors,, +1241,Managers in transport and distribution,64000,46734.0 +1242,Managers in storage and warehousing,95000,36620.0 +1243,Managers in logistics,28000,45104.0 +1251,"Property, housing and estate managers",118000,41115.0 +1252,Garage managers and proprietors,, +1253,Hairdressing and beauty salon managers and proprietors,, +1254,Waste disposal and environmental services managers,8000,48927.0 +1255,Managers and directors in the creative industries,23000,50868.0 +1256,Betting shop and gambling establishment managers,, +1257,Hire services managers and proprietors,,31763.0 +1258,Directors in consultancy services,9000,73453.0 +1259,Managers and proprietors in other services n.e.c.,31000,43382.0 +2111,Chemical scientists,15000,39668.0 +2112,Biological scientists,13000,43781.0 +2113,Biochemists and biomedical scientists,46000,45269.0 +2114,Physical scientists,10000,53142.0 +2115,Social and humanities scientists,11000,38591.0 +2119,Natural and social science professionals n.e.c.,40000,41706.0 +2121,Civil engineers,63000,50602.0 +2122,Mechanical engineers,55000,50594.0 +2123,Electrical engineers,35000,59930.0 +2124,Electronics engineers,17000,51973.0 +2125,Production and process engineers,53000,47711.0 +2126,Aerospace engineers,,55817.0 +2127,Engineering project managers and project engineers,63000,52451.0 +2129,Engineering professionals n.e.c.,148000,47985.0 +2131,IT project managers,31000,58016.0 +2132,IT managers,213000,55502.0 +2133,"IT business analysts, architects and systems designers",168000,59593.0 +2134,Programmers and software development professionals,360000,55587.0 +2135,Cyber security professionals,27000,54816.0 +2136,IT quality and testing professionals,20000,44973.0 +2137,IT network professionals,38000,48294.0 +2139,Information technology professionals n.e.c.,62000,50459.0 +2141,Web design professionals,12000,46639.0 +2142,Graphic and multimedia designers,52000,31236.0 +2151,Conservation professionals,16000,37949.0 +2152,Environment professionals,41000,41555.0 +2161,Research and development (R&D) managers,81000,54857.0 +2162,"Other researchers, unspecified discipline",15000,42463.0 +2211,Generalist medical practitioners,88000,51756.0 +2212,Specialist medical practitioners,187000,88997.0 +2221,Physiotherapists,66000,37917.0 +2222,Occupational therapists,43000,37201.0 +2223,Speech and language therapists,17000,35002.0 +2224,Psychotherapists and cognitive behaviour therapists,,38230.0 +2225,Clinical psychologists,16000,45954.0 +2226,Other psychologists,17000,34250.0 +2229,Therapy professionals n.e.c.,22000,32287.0 +2231,Midwifery nurses,46000,39327.0 +2232,Community nurses,84000,33764.0 +2233,Specialist nurses,75000,41095.0 +2234,Nurse practitioners,114000,41392.0 +2235,Mental health nurses,20000,40028.0 +2236,Children's nurses,19000,34173.0 +2237,Other nursing professionals,720000,36775.0 +2240,Veterinarians,,45362.0 +2251,Pharmacists,60000,47508.0 +2252,Optometrists,14000,38743.0 +2253,Dental practitioners,, +2254,Medical radiographers,45000,44324.0 +2255,Paramedics,40000,50294.0 +2256,Podiatrists,,35920.0 +2259,Other health professionals n.e.c.,91000,38033.0 +2311,Higher education teaching professionals,246000,46494.0 +2312,Further education teaching professionals,55000,38642.0 +2313,Secondary education teaching professionals,466000,44246.0 +2314,Primary education teaching professionals,400000,42031.0 +2315,Nursery education teaching professionals,10000,31425.0 +2316,Special needs education teaching professionals,37000,40363.0 +2317,Teachers of English as a foreign language,, +2319,Teaching professionals n.e.c.,45000, +2321,Head teachers and principals,63000,70977.0 +2322,Education managers,51000,45043.0 +2323,Education advisers and school inspectors,22000,41535.0 +2324,Early education and childcare services managers,23000,28511.0 +2329,Other educational professionals n.e.c,22000,35079.0 +2411,Barristers and judges,28000,34253.0 +2412,Solicitors and lawyers,127000,53314.0 +2419,Legal professionals n.e.c.,71000,33822.0 +2421,Chartered and certified accountants,75000,45538.0 +2422,Finance and investment analysts and advisers,197000,47776.0 +2423,Taxation experts,30000,46280.0 +2431,Management consultants and business analysts,241000,51729.0 +2432,Marketing and commercial managers,100000,50589.0 +2433,"Actuaries, economists and statisticians",43000,51520.0 +2434,Business and related research professionals,99000,39941.0 +2435,Professional/Chartered company secretaries,, +2439,"Business, research and administrative professionals n.e.c.",83000,55106.0 +2440,Business and financial project management professionals,291000,57874.0 +2451,Architects,32000,45625.0 +2452,"Chartered architectural technologists, planning officers and consultants",20000,34951.0 +2453,Quantity surveyors,38000,51950.0 +2454,Chartered surveyors,58000,45673.0 +2455,Construction project managers and related professionals,35000,45613.0 +2461,Social workers,105000,42708.0 +2462,Probation officers,, +2463,Clergy,41000,30655.0 +2464,Youth work professionals,,34630.0 +2469,Welfare professionals n.e.c.,14000,33269.0 +2471,Librarians,13000, +2472,Archivists and curators,9000,33096.0 +2481,Quality control and planning engineers,50000,42511.0 +2482,Quality assurance and regulatory professionals,129000,47969.0 +2483,Environmental health professionals,10000,40044.0 +2491,Newspaper and periodical editors,26000,41583.0 +2492,Newspaper and periodical journalists and reporters,17000,42169.0 +2493,Public relations professionals,39000,36336.0 +2494,Advertising accounts managers and creative directors,31000,46356.0 +3111,Laboratory technicians,64000,26861.0 +3112,Electrical and electronics technicians,16000,35018.0 +3113,Engineering technicians,87000,44330.0 +3114,Building and civil engineering technicians,10000,36912.0 +3115,Quality assurance technicians,39000,33242.0 +3116,"Planning, process and production technicians",49000,36062.0 +3119,"Science, engineering and production technicians n.e.c.",146000,34475.0 +3120,"CAD, drawing and architectural technicians",41000,34465.0 +3131,IT operations technicians,69000,34656.0 +3132,IT user support technicians,152000,34314.0 +3133,Database administrators and web content technicians,39000,36015.0 +3211,Dispensing opticians,,27645.0 +3212,Pharmaceutical technicians,27000,28381.0 +3213,Medical and dental technicians,32000,29119.0 +3214,Complementary health associate professionals,, +3219,Health associate professionals n.e.c.,8000,25017.0 +3221,Youth and community workers,102000,27711.0 +3222,Child and early years officers,53000,29347.0 +3223,Housing officers,45000,32542.0 +3224,Counsellors,19000,27082.0 +3229,Welfare and housing associate professionals n.e.c.,146000,26640.0 +3231,Higher level teaching assistants,39000,22050.0 +3232,Early education and childcare practitioners,109000,19516.0 +3240,Veterinary nurses,14000,26666.0 +3311,Non-commissioned officers and other ranks,, +3312,Police officers (sergeant and below),274000, +3313,Fire service officers (watch manager and below),55000,40775.0 +3314,Prison service officers (below principal officer),9000,31603.0 +3319,Protective service associate professionals n.e.c.,36000,41592.0 +3411,Artists,, +3412,"Authors, writers and translators",20000,36865.0 +3413,"Actors, entertainers and presenters",, +3414,Dancers and choreographers,, +3415,Musicians,, +3416,"Arts officers, producers and directors",23000,39643.0 +3417,"Photographers, audio-visual and broadcasting equipment operators",21000,30396.0 +3421,Interior designers,18000,34962.0 +3422,"Clothing, fashion and accessories designers",,36731.0 +3429,Design occupations n.e.c.,23000,37017.0 +3431,Sports players,11000, +3432,"Sports coaches, instructors and officials",62000,12570.0 +3433,Fitness and wellbeing instructors,25000, +3511,Aircraft pilots and air traffic controllers,20000,107712.0 +3512,Ship and hovercraft officers,, +3520,Legal associate professionals,41000,32438.0 +3531,Brokers,19000,51026.0 +3532,Insurance underwriters,25000,38666.0 +3533,Financial and accounting technicians,29000,53265.0 +3534,Financial accounts managers,129000,45162.0 +3541,"Estimators, valuers and assessors",53000,37809.0 +3542,Importers and exporters,,34757.0 +3543,Project support officers,71000,34207.0 +3544,Data analysts,68000,38107.0 +3549,Business associate professionals n.e.c.,99000,33035.0 +3551,Buyers and procurement officers,61000,36230.0 +3552,Business sales executives,165000,36498.0 +3553,Merchandisers,19000,26554.0 +3554,Marketing associate professionals,133000,30479.0 +3555,Estate agents and auctioneers,25000,26988.0 +3556,Sales accounts and business development managers,415000,56021.0 +3557,Events managers and organisers,46000,29101.0 +3560,Public services associate professionals,82000,38454.0 +3571,Human resources and industrial relations officers,138000,33012.0 +3572,Careers advisers and vocational guidance specialists,22000,30045.0 +3573,Information technology trainers,,36621.0 +3574,Other vocational and industrial trainers,125000,33236.0 +3581,Inspectors of standards and regulations,20000,37236.0 +3582,Health and safety managers and officers,66000,44551.0 +4111,National government administrative occupations,131000,31363.0 +4112,Local government administrative occupations,59000,27642.0 +4113,Officers of non-governmental organisations,10000, +4121,Credit controllers,24000,26981.0 +4122,"Book-keepers, payroll managers and wages clerks",272000,27743.0 +4123,Bank and post office clerks,66000,27671.0 +4124,Finance officers,28000,28610.0 +4129,Financial administrative occupations n.e.c.,98000,25936.0 +4131,Records clerks and assistants,139000,26312.0 +4132,Pensions and insurance clerks and assistants,35000,29329.0 +4133,Stock control clerks and assistants,69000,28851.0 +4134,Transport and distribution clerks and assistants,55000,32060.0 +4135,Library clerks and assistants,18000,18659.0 +4136,Human resources administrative occupations,19000,25531.0 +4141,Office managers,195000,35000.0 +4142,Office supervisors,57000,32265.0 +4143,Customer service managers,66000,32983.0 +4151,Sales administrators,49000,27132.0 +4152,Data entry administrators,16000,26534.0 +4159,Other administrative occupations n.e.c.,728000,23385.0 +4211,Medical secretaries,43000,24071.0 +4212,Legal secretaries,22000,24263.0 +4213,School secretaries,23000,22155.0 +4214,Company secretaries and administrators,14000, +4215,Personal assistants and other secretaries,164000,25233.0 +4216,Receptionists,185000,18152.0 +4217,Typists and related keyboard occupations,, +5111,Farmers,6000,32728.0 +5112,Horticultural trades,,24613.0 +5113,Gardeners and landscape gardeners,33000,27057.0 +5114,Groundsmen and greenkeepers,38000,27519.0 +5119,Agricultural and fishing trades n.e.c.,16000,27676.0 +5211,Sheet metal workers,8000,31920.0 +5212,"Metal plate workers, smiths, moulders and related occupations",,37035.0 +5213,Welding trades,37000,34742.0 +5214,Pipe fitters,,42580.0 +5221,Metal machining setters and setter-operators,48000,35394.0 +5222,"Tool makers, tool fitters and markers-out",8000,38584.0 +5223,Metal working production and maintenance fitters,223000,40002.0 +5224,Precision instrument makers and repairers,10000,37031.0 +5225,Air-conditioning and refrigeration installers and repairers,,41166.0 +5231,"Vehicle technicians, mechanics and electricians",98000,36560.0 +5232,Vehicle body builders and repairers,16000,34848.0 +5233,Vehicle paint technicians,,34531.0 +5234,Aircraft maintenance and related trades,12000,44704.0 +5235,Boat and ship builders and repairers,,32600.0 +5236,Rail and rolling stock builders and repairers,,64322.0 +5241,Electricians and electrical fitters,97000,39187.0 +5242,Telecoms and related network installers and repairers,25000,39652.0 +5243,"TV, video and audio servicers and repairers",, +5244,Computer system and equipment installers and servicers,9000,34073.0 +5245,Security system installers and repairers,9000,37991.0 +5246,Electrical service and maintenance mechanics and repairers,51000,41111.0 +5249,Electrical and electronic trades n.e.c.,20000,48171.0 +5250,"Skilled metal, electrical and electronic trades supervisors",64000,44793.0 +5311,Steel erectors,,34782.0 +5312,Stonemasons and related trades,,33938.0 +5313,Bricklayers,8000,32480.0 +5314,"Roofers, roof tilers and slaters",12000,30961.0 +5315,Plumbers & heating and ventilating installers and repairers,58000,36563.0 +5316,Carpenters and joiners,56000,33797.0 +5317,"Glaziers, window fabricators and fitters",17000,28623.0 +5319,Construction and building trades n.e.c.,41000,34378.0 +5321,Plasterers,7000,33789.0 +5322,Floorers and wall tilers,9000,32663.0 +5323,Painters and decorators,22000,30889.0 +5330,Construction and building trades supervisors,42000,45000.0 +5411,Upholsterers,,26966.0 +5412,Footwear and leather working trades,,25116.0 +5413,Tailors and dressmakers,, +5419,"Textiles, garments and related trades n.e.c.",,26173.0 +5421,Pre-press technicians,,27496.0 +5422,Printers,8000,31367.0 +5423,Print finishing and binding workers,,25296.0 +5431,Butchers,24000,27929.0 +5432,Bakers and flour confectioners,10000,26983.0 +5433,Fishmongers and poultry dressers,, +5434,Chefs,152000,26531.0 +5435,Cooks,39000,17885.0 +5436,Catering and bar managers,37000,27888.0 +5441,"Glass and ceramics makers, decorators and finishers",, +5442,Furniture makers and other craft woodworkers,10000,30328.0 +5443,Florists,, +5449,Other skilled trades n.e.c.,12000,26800.0 +6111,Early education and childcare assistants,83000,19165.0 +6112,Teaching assistants,255000,18024.0 +6113,Educational support assistants,138000,17086.0 +6114,Childminders,, +6116,Nannies and au pairs,11000,22955.0 +6117,Playworkers,28000, +6121,Pest control officers,,27487.0 +6129,Animal care services occupations n.e.c.,25000,23345.0 +6131,Nursing auxiliaries and assistants,466000,24761.0 +6132,Ambulance staff (excluding paramedics),21000,31516.0 +6133,Dental nurses,53000,22615.0 +6134,Houseparents and residential wardens,12000,26499.0 +6135,Care workers and home carers,581000,21487.0 +6136,Senior care workers,101000,27417.0 +6137,Care escorts,14000,12175.0 +6138,"Undertakers, mortuary and crematorium assistants",17000,27020.0 +6211,Sports and leisure assistants,42000,14366.0 +6212,Travel agents,19000,26426.0 +6213,Air travel assistants,42000,28808.0 +6214,Rail travel assistants,19000,45240.0 +6219,Leisure and travel service occupations n.e.c.,9000, +6221,Hairdressers and barbers,38000,15064.0 +6222,Beauticians and related occupations,28000,15009.0 +6231,Housekeepers and related occupations,40000,16618.0 +6232,Caretakers,62000,25147.0 +6240,Cleaning and housekeeping managers and supervisors,48000,24931.0 +6250,Bed and breakfast and guest house owners and proprietors,, +6311,Police community support officers,11000,35189.0 +6312,Parking and civil enforcement occupations,14000,27766.0 +7111,Sales and retail assistants,895000,14491.0 +7112,Retail cashiers and check-out operators,51000,14018.0 +7113,Telephone salespersons,9000,26944.0 +7114,Pharmacy and optical dispensing assistants,47000,17993.0 +7115,Vehicle and parts salespersons and advisers,12000,31750.0 +7121,Collector salespersons and credit agents,, +7122,"Debt, rent and other cash collectors",10000,27454.0 +7123,Roundspersons and van salespersons,,26984.0 +7124,Market and street traders and assistants,, +7125,Visual merchandisers and related occupations,,25488.0 +7129,Sales related occupations n.e.c.,20000,28870.0 +7131,Shopkeepers and owners - retail and wholesale,6000,35083.0 +7132,Sales supervisors - retail and wholesale,85000,26112.0 +7211,Call and contact centre occupations,35000,25440.0 +7212,Telephonists,4000, +7213,Communication operators,15000,34934.0 +7214,Market research interviewers,, +7219,Customer service occupations n.e.c.,299000,24438.0 +7220,Customer service supervisors,18000,34033.0 +8111,"Food, drink and tobacco process operatives",147000,27267.0 +8112,Textile process operatives,9000,25572.0 +8113,Chemical and related process operatives,17000,33531.0 +8114,Plastics process operatives,12000,29644.0 +8115,Metal making and treating process operatives,12000,31893.0 +8119,Process operatives n.e.c.,14000,30843.0 +8120,Metal working machine operatives,14000,31344.0 +8131,Paper and wood machine operatives,13000,29640.0 +8132,Mining and quarry workers and related operatives,,38301.0 +8133,Energy plant operatives,, +8134,Water and sewerage plant operatives,11000,39057.0 +8135,Printing machine assistants,10000,29657.0 +8139,Plant and machine operatives n.e.c.,12000,29142.0 +8141,Assemblers (electrical and electronic products),18000,28241.0 +8142,Assemblers (vehicles and metal goods),40000,31041.0 +8143,Routine inspectors and testers,36000,33982.0 +8144,"Weighers, graders and sorters",,29141.0 +8145,"Tyre, exhaust and windscreen fitters",13000,30429.0 +8146,Sewing machinists,14000,22767.0 +8149,Assemblers and routine operatives n.e.c.,53000,26975.0 +8151,"Scaffolders, stagers and riggers",,40797.0 +8152,Road construction operatives,18000,38315.0 +8153,Rail construction and maintenance operatives,11000,44445.0 +8159,Construction operatives n.e.c.,82000,30237.0 +8160,"Production, factory and assembly supervisors",24000,35092.0 +8211,Large goods vehicle drivers,204000,39141.0 +8212,Bus and coach drivers,83000,33947.0 +8213,Taxi and cab drivers and chauffeurs,, +8214,Delivery drivers and couriers,126000,24627.0 +8215,Driving instructors,, +8219,Road transport drivers n.e.c.,112000,28725.0 +8221,Crane drivers,7000,46392.0 +8222,Fork-lift truck drivers,19000,31016.0 +8229,Mobile machine drivers and operatives n.e.c.,38000,36408.0 +8231,Train and tram drivers,26000,76176.0 +8232,Marine and waterways transport operatives,,39405.0 +8233,Air transport operatives,14000,32376.0 +8234,Rail transport operatives,15000,56925.0 +8239,Other drivers and transport operatives n.e.c.,7000,32066.0 +9111,Farm workers,25000, +9112,Forestry and related workers,, +9119,Fishing and other elementary agriculture occupations n.e.c.,, +9121,Groundworkers,13000,37849.0 +9129,Elementary construction occupations n.e.c.,45000,26723.0 +9131,Industrial cleaning process occupations,14000,26236.0 +9132,"Packers, bottlers, canners and fillers",61000,25087.0 +9139,Elementary process plant occupations n.e.c.,41000,28600.0 +9211,"Postal workers, mail sorters and messengers",116000,29761.0 +9219,Elementary administration occupations n.e.c.,23000,23005.0 +9221,Window cleaners,,25002.0 +9222,Street cleaners,4000,26330.0 +9223,Cleaners and domestics,411000,11852.0 +9224,"Launderers, dry cleaners and pressers",13000,20464.0 +9225,Refuse and salvage occupations,22000,27576.0 +9226,Vehicle valeters and cleaners,14000,24875.0 +9229,Elementary cleaning occupations n.e.c.,,25688.0 +9231,Security guards and related occupations,93000,30819.0 +9232,School midday and crossing patrol occupations,67000,4263.0 +9233,Exam invigilators,19000,1902.0 +9241,Shelf fillers,14000,15809.0 +9249,Elementary sales occupations n.e.c.,6000, +9251,Elementary storage supervisors,28000,30480.0 +9252,Warehouse operatives,449000,26574.0 +9253,Delivery operatives,19000,25541.0 +9259,Elementary storage occupations n.e.c.,6000,31589.0 +9261,Bar and catering supervisors,31000,22552.0 +9262,Hospital porters,6000,27988.0 +9263,Kitchen and catering assistants,322000,11840.0 +9264,Waiters and waitresses,146000,10000.0 +9265,Bar staff,84000,9166.0 +9266,Coffee shop workers,16000,12170.0 +9267,Leisure and theme park attendants,37000, +9269,Other elementary services occupations n.e.c.,11000, diff --git a/packages/populace-build/src/populace/build/uk_runtime/occupation_targets_data/nomis_aps_age_employment.csv b/packages/populace-build/src/populace/build/uk_runtime/occupation_targets_data/nomis_aps_age_employment.csv new file mode 100644 index 00000000..5a1694f0 --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/occupation_targets_data/nomis_aps_age_employment.csv @@ -0,0 +1,8 @@ +age_band,employment,period,geography +16+,33321600,Jan 2025-Dec 2025,United Kingdom +16-24,3582100,Jan 2025-Dec 2025,United Kingdom +25-34,7624500,Jan 2025-Dec 2025,United Kingdom +35-44,7393300,Jan 2025-Dec 2025,United Kingdom +45-54,7435300,Jan 2025-Dec 2025,United Kingdom +55-64,5695800,Jan 2025-Dec 2025,United Kingdom +65+,1590500,Jan 2025-Dec 2025,United Kingdom diff --git a/packages/populace-build/src/populace/build/uk_runtime/occupation_targets_data/nomis_aps_occupation_employment.csv b/packages/populace-build/src/populace/build/uk_runtime/occupation_targets_data/nomis_aps_occupation_employment.csv new file mode 100644 index 00000000..ffcd8c95 --- /dev/null +++ b/packages/populace-build/src/populace/build/uk_runtime/occupation_targets_data/nomis_aps_occupation_employment.csv @@ -0,0 +1,27 @@ +soc_code,soc_title,employment,period,geography +11,Corporate Managers and Directors,2619900,Jan 2025-Dec 2025,United Kingdom +12,Other Managers and Proprietors,1189900,Jan 2025-Dec 2025,United Kingdom +21,"Science, Research, Engineering and Technology Professionals",2657800,Jan 2025-Dec 2025,United Kingdom +22,Health Professionals,1733300,Jan 2025-Dec 2025,United Kingdom +23,Teaching and Other Educational Professionals,1750600,Jan 2025-Dec 2025,United Kingdom +24,"Business, Media and Public Service Professionals",2823600,Jan 2025-Dec 2025,United Kingdom +31,"Science, Engineering and Technology Associate Professionals",590600,Jan 2025-Dec 2025,United Kingdom +32,Health and Social Care Associate Professionals,765900,Jan 2025-Dec 2025,United Kingdom +33,Protective Service Occupations,415600,Jan 2025-Dec 2025,United Kingdom +34,"Culture, Media and Sports Occupations",773300,Jan 2025-Dec 2025,United Kingdom +35,Business and Public Service Associate Professionals,2407700,Jan 2025-Dec 2025,United Kingdom +41,Administrative Occupations,2598800,Jan 2025-Dec 2025,United Kingdom +42,Secretarial and Related Occupations,472900,Jan 2025-Dec 2025,United Kingdom +51,Skilled Agricultural and Related Trades,334400,Jan 2025-Dec 2025,United Kingdom +52,"Skilled Metal, Electrical and Electronic Trades",936700,Jan 2025-Dec 2025,United Kingdom +53,Skilled Construction and Building Trades,916400,Jan 2025-Dec 2025,United Kingdom +54,"Textiles, Printing and Other Skilled Trades",584100,Jan 2025-Dec 2025,United Kingdom +61,Caring Personal Service Occupations,2291400,Jan 2025-Dec 2025,United Kingdom +62,"Leisure, Travel and Related Personal Service Occupations",579300,Jan 2025-Dec 2025,United Kingdom +63,Community And Civil Enforcement Occupations,23200,Jan 2025-Dec 2025,United Kingdom +71,Sales Occupations,1457300,Jan 2025-Dec 2025,United Kingdom +72,Customer Service Occupations,447900,Jan 2025-Dec 2025,United Kingdom +81,"Process, Plant and Machine Operatives",676100,Jan 2025-Dec 2025,United Kingdom +82,Transport and Mobile Machine Drivers and Operatives,1171300,Jan 2025-Dec 2025,United Kingdom +91,Elementary Trades and Related Occupations,386100,Jan 2025-Dec 2025,United Kingdom +92,Elementary Administration and Service Occupations,2612800,Jan 2025-Dec 2025,United Kingdom diff --git a/packages/populace-build/tests/test_ai_exposure.py b/packages/populace-build/tests/test_ai_exposure.py new file mode 100644 index 00000000..b1fa8f48 --- /dev/null +++ b/packages/populace-build/tests/test_ai_exposure.py @@ -0,0 +1,129 @@ +from __future__ import annotations + +import numpy as np +import pytest + +from populace.build.uk_runtime import ( + MEASURE_TO_COLUMN, + exposure_for_major_group, + exposure_for_soc, + load_ai_exposure_table, + load_major_group_ai_exposure_table, +) + + +def test_unit_group_table_covers_all_soc2020_unit_groups() -> None: + table = load_ai_exposure_table() + + assert len(table) == 412 + assert table.index.str.fullmatch(r"\d{4}").all() + assert set(table["mapping_quality"]) <= { + "direct", + "chained", + "imputed-from-parent", + } + for column in MEASURE_TO_COLUMN.values(): + assert table[column].notna().all() + + +def test_known_codes_return_scores_in_expected_ranges() -> None: + theta = exposure_for_soc(["1111", "7113", "2211"], measure="complementarity") + assert ((theta > 0.0) & (theta < 1.0)).all() + # Medical practitioners are more AI-complementary than telephone + # salespersons (Pizzinelli et al. 2023's lawyers/telemarketers contrast). + assert theta[2] > theta[1] + + c_aioe = exposure_for_soc(["4112", "9267"], measure="c_aioe") + assert np.isfinite(c_aioe).all() + assert (np.abs(c_aioe) < 3.0).all() + # Administrative occupations are more exposed than elementary ones. + assert c_aioe[0] > c_aioe[1] + + eloundou = exposure_for_soc(["4112"], measure="eloundou_beta") + assert 0.0 <= eloundou[0] <= 1.0 + + # DSIT Annex 1 publishes telephone salespersons (SOC 7113, unchanged + # between SOC 2010 and SOC 2020) at 1.101573 (AI) and 1.966413 (LLM). + assert exposure_for_soc(["7113"], measure="dsit_aioe")[0] == pytest.approx( + 1.101573, abs=1e-4 + ) + assert exposure_for_soc(["7113"], measure="dsit_llm")[0] == pytest.approx( + 1.966413, abs=1e-4 + ) + + +def test_integer_codes_match_string_codes() -> None: + assert exposure_for_soc([4112], measure="felten_aioe")[0] == pytest.approx( + exposure_for_soc(["4112"], measure="felten_aioe")[0] + ) + + +def test_parent_code_fallback() -> None: + table = load_ai_exposure_table() + expected_411 = table.loc[table.index.str.startswith("411"), "c_aioe"].mean() + + # A 3-digit code resolves to the mean of its unit groups. + assert exposure_for_soc(["411"])[0] == pytest.approx(expected_411) + # An unknown 4-digit code falls back to its 3-digit parent. + assert "4119" not in table.index + assert exposure_for_soc(["4119"])[0] == pytest.approx(expected_411) + + # A code with no 3-digit parent falls back to the 2-digit parent. + expected_41 = table.loc[table.index.str.startswith("41"), "c_aioe"].mean() + assert not table.index.str.startswith("419").any() + assert exposure_for_soc(["4190"])[0] == pytest.approx(expected_41) + + +def test_unknown_code_returns_nan_with_warning() -> None: + with pytest.warns(UserWarning, match="0000"): + result = exposure_for_soc(["0000", "1111"]) + + assert np.isnan(result[0]) + assert np.isfinite(result[1]) + + with pytest.warns(UserWarning): + assert np.isnan(exposure_for_soc(["not-a-code"])[0]).all() + + +def test_invalid_measure_raises() -> None: + with pytest.raises(ValueError, match="Unknown measure"): + exposure_for_soc(["1111"], measure="webb_2020") + + +def test_major_group_lookup() -> None: + table = load_major_group_ai_exposure_table() + assert list(table.index) == [str(group) for group in range(1, 10)] + assert (table["weighting"] == "ashe_2025_table14_jobs").all() + + scores = exposure_for_major_group([1, 2, 9], measure="c_aioe") + assert np.isfinite(scores).all() + # Professional occupations are more exposed than elementary occupations. + assert scores[1] > scores[2] + + theta = exposure_for_major_group(["1", "9"], measure="complementarity") + assert ((theta > 0.0) & (theta < 1.0)).all() + # Managers have higher AI complementarity than elementary occupations. + assert theta[0] > theta[1] + + with pytest.warns(UserWarning, match="0"): + missing = exposure_for_major_group([0]) + assert np.isnan(missing[0]) + + +def test_major_group_accepts_frs_thousands_coding() -> None: + # FRS adult.tab codes SOC 2020 major groups as 1000-9000. + frs_codes = [group * 1000 for group in range(1, 10)] + plain_codes = list(range(1, 10)) + + for measure in ("c_aioe", "complementarity", "felten_aioe"): + np.testing.assert_allclose( + exposure_for_major_group(frs_codes, measure=measure), + exposure_for_major_group(plain_codes, measure=measure), + ) + + # String FRS coding works too, and non-multiples of 1000 stay unknown. + np.testing.assert_allclose( + exposure_for_major_group(["4000"]), exposure_for_major_group(["4"]) + ) + with pytest.warns(UserWarning, match="4100"): + assert np.isnan(exposure_for_major_group(["4100"])[0]) diff --git a/packages/populace-build/tests/test_ai_shock_runner.py b/packages/populace-build/tests/test_ai_shock_runner.py new file mode 100644 index 00000000..c0b33b3a --- /dev/null +++ b/packages/populace-build/tests/test_ai_shock_runner.py @@ -0,0 +1,248 @@ +"""AI shock runner: pure summary math, grid construction, dataclass I/O. + +Everything here runs without policyengine_uk installed or real data — the +runner's pure parts are the unit under test. The single engine-facing smoke +test is skipped when policyengine_uk is unavailable. +""" + +from __future__ import annotations + +import json + +import numpy as np +import pytest + +from populace.build.uk_runtime.ai_shock_runner import ( + GRID_CAPITAL_RETURN_INCREASE, + AgeBandDelta, + DecileDelta, + PopulationMetrics, + ScenarioResult, + age_band_deltas, + build_scenario_grid, + decile_deltas, + decile_ids, + gini, + scenario_result_from_dict, + weighted_mean, + write_grid_csv, + write_grid_json, +) + +# ---------------------------------------------------------------------- +# Grid construction +# ---------------------------------------------------------------------- + + +def test_grid_is_the_jr16_robustness_grid(): + grid = build_scenario_grid() + assert len(grid) == 50 # 10 displacement x 5 wage points + rates = sorted({scenario.displacement_rate for scenario in grid}) + uplifts = sorted({scenario.wage_uplift for scenario in grid}) + assert rates == pytest.approx([r / 100 for r in range(1, 11)]) + assert uplifts == pytest.approx([u / 100 for u in range(1, 6)]) + # The +0.4pp capital shock is always on. + assert all( + scenario.capital_return_increase == GRID_CAPITAL_RETURN_INCREASE + for scenario in grid + ) + assert len({scenario.name for scenario in grid}) == 50 + assert any(scenario.name == "grid_emp7pct_wage2pct" for scenario in grid) + + +def test_grid_custom_margins_and_seed(): + grid = build_scenario_grid( + displacement_rates=(0.02,), wage_uplifts=(0.01, 0.03), seed=7, n_draws=3 + ) + assert [scenario.name for scenario in grid] == [ + "grid_emp2pct_wage1pct", + "grid_emp2pct_wage3pct", + ] + assert all(s.seed == 7 and s.n_draws == 3 for s in grid) + + +# ---------------------------------------------------------------------- +# Pure summary math +# ---------------------------------------------------------------------- + + +def test_weighted_mean(): + assert weighted_mean(np.array([1.0, 3.0]), np.array([1.0, 3.0])) == 2.5 + assert weighted_mean(np.array([]), np.array([])) == 0.0 + + +def test_gini_known_values(): + # Perfect equality. + assert gini(np.full(5, 10.0), np.ones(5)) == pytest.approx(0.0) + # One person holds everything: G -> (n-1)/n with equal weights. + values = np.array([0.0, 0.0, 0.0, 100.0]) + assert gini(values, np.ones(4)) == pytest.approx(0.75) + # Weight-replication invariance: weight 2 equals a duplicated record. + a = gini(np.array([1.0, 2.0, 3.0]), np.array([1.0, 2.0, 1.0])) + b = gini(np.array([1.0, 2.0, 2.0, 3.0]), np.ones(4)) + assert a == pytest.approx(b) + + +def test_gini_rejects_bad_input(): + with pytest.raises(ValueError, match="finite"): + gini(np.array([1.0, np.nan]), np.ones(2)) + with pytest.raises(ValueError, match="non-negative"): + gini(np.array([1.0, 2.0]), np.array([1.0, -1.0])) + + +def test_decile_ids_equal_weights(): + income = np.arange(100, dtype=float) + deciles = decile_ids(income, np.ones(100)) + assert deciles.min() == 1 and deciles.max() == 10 + counts = np.bincount(deciles)[1:] + assert counts.tolist() == [10] * 10 + # Monotone in income. + assert (np.diff(deciles[np.argsort(income)]) >= 0).all() + + +def test_decile_ids_respects_weights(): + # One heavy low-income record should fill the bottom deciles. + income = np.array([1.0, 2.0, 3.0]) + deciles = decile_ids(income, np.array([8.0, 1.0, 1.0])) + assert deciles[0] == 1 and deciles[2] == 10 + + +def test_decile_deltas_fixed_at_baseline(): + baseline = np.arange(1.0, 101.0) + shocked = baseline.copy() + shocked[-10:] = 0.0 # top decile wiped out, JR16-style displacement + deltas = decile_deltas(baseline, shocked, np.ones(100)) + assert len(deltas) == 10 + assert all(d.mean_income_change == 0.0 for d in deltas[:-1]) + top = deltas[-1] + assert top.decile == 10 + # Reported within the *baseline* decile even though income collapsed. + assert top.baseline_mean_income == pytest.approx(95.5) + assert top.mean_income_change == pytest.approx(-95.5) + assert top.weighted_population == 10.0 + + +def test_age_band_deltas(): + age = np.array([20.0, 30.0, 70.0]) + baseline = np.array([100.0, 200.0, 300.0]) + shocked = np.array([50.0, 200.0, 330.0]) + deltas = age_band_deltas(age, baseline, shocked, np.ones(3)) + by_band = {d.age_band: d for d in deltas} + assert set(by_band) == {"16-24", "25-34", "35-44", "45-54", "55-64", "65+"} + assert by_band["16-24"].mean_income_change == pytest.approx(-50.0) + assert by_band["25-34"].mean_income_change == pytest.approx(0.0) + assert by_band["65+"].mean_income_change == pytest.approx(30.0) + assert by_band["35-44"].weighted_population == 0.0 + + +# ---------------------------------------------------------------------- +# Dataclass round-trip and writers +# ---------------------------------------------------------------------- + + +def _result(name: str = "central") -> ScenarioResult: + baseline = PopulationMetrics( + gov_balance=100.0, + poverty_rate_bhc=0.15, + poverty_rate_ahc=None, + gini_equivalised_income=0.32, + ) + shocked = PopulationMetrics( + gov_balance=90.0, + poverty_rate_bhc=0.17, + poverty_rate_ahc=None, + gini_equivalised_income=0.34, + ) + return ScenarioResult( + scenario=name, + displacement_rate=0.07, + wage_uplift=0.026, + capital_return_increase=0.004, + baseline=baseline, + shocked=shocked, + exchequer_cost=10.0, + poverty_rate_change_bhc=0.02, + poverty_rate_change_ahc=None, + gini_change=0.02, + by_decile=( + DecileDelta(1, 10.0, 9.0, -1.0, 100.0), + DecileDelta(2, 20.0, 21.0, 1.0, 100.0), + ), + by_age_band=(AgeBandDelta("16-24", 15.0, 12.0, -3.0, 50.0),), + employment_status_applied=True, + shock_summary={"scenario": name, "employment": {"displaced_count": 3.0}}, + ) + + +def test_scenario_result_json_round_trip(): + result = _result() + payload = json.loads(json.dumps(result.to_dict())) + assert scenario_result_from_dict(payload) == result + + +def test_write_grid_json_and_csv(tmp_path): + results = [_result("central"), _result("low")] + json_path = tmp_path / "out" / "results.json" + write_grid_json(results, json_path) + loaded = json.loads(json_path.read_text()) + assert [scenario_result_from_dict(item) for item in loaded] == results + + paths = write_grid_csv(results, tmp_path / "out") + assert set(paths) == {"overall", "by_decile", "by_age_band"} + import pandas as pd + + overall = pd.read_csv(paths["overall"]) + assert overall["scenario"].tolist() == ["central", "low"] + assert overall["exchequer_cost"].tolist() == [10.0, 10.0] + by_decile = pd.read_csv(paths["by_decile"]) + assert len(by_decile) == 4 # 2 scenarios x 2 deciles + by_age = pd.read_csv(paths["by_age_band"]) + assert by_age["age_band"].tolist() == ["16-24", "16-24"] + + +# ---------------------------------------------------------------------- +# Engine-facing path (skip without policyengine_uk) +# ---------------------------------------------------------------------- + + +def test_run_ai_shock_analysis_requires_exposure_or_major_group(): + import pandas as pd + + from populace.build.uk_runtime.ai_shock_runner import _attach_exposure_columns + + with pytest.raises(ValueError, match="soc_major_group"): + _attach_exposure_columns( + pd.DataFrame({"age": [30.0]}), + exposure=None, + soc_major_group=None, + complementarity=None, + ) + + +def test_attach_exposure_from_major_group_uses_crosswalk(): + import pandas as pd + + from populace.build.uk_runtime.ai_exposure import ( + load_major_group_ai_exposure_table, + ) + from populace.build.uk_runtime.ai_shock_runner import _attach_exposure_columns + + persons = _attach_exposure_columns( + pd.DataFrame({"age": [30.0, 40.0]}), + exposure=None, + soc_major_group=[2000, 9000], + complementarity=None, + ) + table = load_major_group_ai_exposure_table() + assert persons["ai_exposure"].iloc[0] == pytest.approx(table.loc["2", "c_aioe"]) + assert persons["ai_exposure"].iloc[1] == pytest.approx(table.loc["9", "c_aioe"]) + assert np.isfinite(persons["ai_complementarity"]).all() + + +def test_engine_smoke(): + pytest.importorskip("policyengine_uk") + # A full engine run needs a real UK H5 dataset (licensed, not in repo); + # the importable surface is asserted instead. + from populace.build.uk_runtime.ai_shock_runner import run_ai_shock_analysis + + assert callable(run_ai_shock_analysis) diff --git a/packages/populace-build/tests/test_ai_shock_scenarios.py b/packages/populace-build/tests/test_ai_shock_scenarios.py new file mode 100644 index 00000000..dca7046f --- /dev/null +++ b/packages/populace-build/tests/test_ai_shock_scenarios.py @@ -0,0 +1,451 @@ +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from populace.build.uk_runtime.ai_shock_scenarios import ( + AGE_BANDS, + COMPLEMENTARITY_COLUMN, + DISPLACED_COLUMN, + KLEIN_TEESELINK_YOUTH_MULTIPLIER, + PRE_SHOCK_EMPLOYMENT_INCOME_COLUMN, + PRESETS, + ShockScenario, + apply_capital_shock, + apply_employment_shock, + apply_wage_shock, + run_scenario, +) + +N_PERSONS = 12_000 + + +def synthetic_persons(seed: int = 7, n: int = N_PERSONS) -> pd.DataFrame: + """Person table with age, incomes, exposure/complementarity and weights.""" + + rng = np.random.default_rng(seed) + age = rng.integers(16, 86, size=n) + employed = rng.uniform(size=n) < np.where(age < 65, 0.75, 0.1) + occupation_group = rng.choice( + ["managers", "clerical", "trades", "care", "elementary"], size=n + ) + exposure_by_group = { + "managers": 0.9, + "clerical": 1.4, + "trades": 0.2, + "care": 0.4, + "elementary": 0.7, + } + exposure = np.array([exposure_by_group[g] for g in occupation_group]) + exposure = exposure + rng.normal(0.0, 0.05, size=n) + return pd.DataFrame( + { + "age": age, + "employment_income": np.where( + employed, rng.lognormal(10.0, 0.5, size=n), 0.0 + ), + "savings_interest_income": rng.exponential(120.0, size=n), + "dividend_income": rng.exponential(300.0, size=n), + "property_income": rng.exponential(500.0, size=n), + "ai_exposure": exposure, + "ai_complementarity": rng.uniform(0.1, 1.0, size=n), + "occupation_group": occupation_group, + "person_weight": rng.uniform(0.5, 3.0, size=n), + } + ) + + +def scenario(**overrides) -> ShockScenario: + parameters = { + "name": "test", + "displacement_rate": 0.07, + "wage_uplift": 0.026, + "n_draws": 10, + "seed": 11, + } + parameters.update(overrides) + return ShockScenario(**parameters) + + +def employed_mask(persons: pd.DataFrame) -> pd.Series: + return (persons["employment_income"] > 0) & (persons["age"] >= 16) + + +class TestEmploymentShock: + def test_aggregate_weighted_displacement_matches_rate(self): + persons = synthetic_persons() + shocked, summary = apply_employment_shock(persons, scenario()) + + employed = employed_mask(persons) + employed_weight = persons.loc[employed, "person_weight"].sum() + realised = ( + shocked.loc[shocked[DISPLACED_COLUMN], "person_weight"].sum() + / employed_weight + ) + # The realised draw honours the per-group weighted quotas up to the + # granularity of one person weight per group. + assert realised == pytest.approx(0.07, abs=0.005) + # The draw-averaged summary matches at least as tightly. + assert summary["displaced_weighted_share"] == pytest.approx(0.07, abs=0.005) + + def test_eq_3_4_group_totals_honoured(self): + persons = synthetic_persons() + shocked, summary = apply_employment_shock(persons, scenario()) + + employed = employed_mask(persons) + weights = persons["person_weight"] + total_job_loss = 0.07 * weights[employed].sum() + scores = {} + for label, block in persons[employed].groupby("occupation_group"): + emp = block["person_weight"].sum() + mean_exposure = (block["ai_exposure"] * block["person_weight"]).sum() / emp + scores[label] = emp * mean_exposure + total_score = sum(scores.values()) + + max_weight = weights.max() + for label, score in scores.items(): + expected_quota = total_job_loss * score / total_score + group = summary["by_occupation_group"][label] + assert group["job_loss_quota_weighted"] == pytest.approx(expected_quota) + # Every draw meets its quota up to one boundary person's weight, + # so the realised table and the draw average both honour eq 3.4. + assert group["displaced_weighted"] == pytest.approx( + expected_quota, abs=max_weight + ) + realised = shocked.loc[ + shocked[DISPLACED_COLUMN] & (shocked["occupation_group"] == label), + "person_weight", + ].sum() + assert realised == pytest.approx(expected_quota, abs=max_weight) + + def test_displacement_concentrates_in_high_exposure_group(self): + persons = synthetic_persons() + _, summary = apply_employment_shock(persons, scenario()) + by_group = summary["by_occupation_group"] + # clerical has the highest exposure, trades the lowest. + assert ( + by_group["clerical"]["displaced_weighted_share"] + > by_group["trades"]["displaced_weighted_share"] + ) + + def test_multi_draw_averaging_converges(self): + persons = synthetic_persons() + single = scenario(n_draws=1) + many = scenario(n_draws=50) + _, single_summary = apply_employment_shock(persons, single) + _, many_summary = apply_employment_shock(persons, many) + + # Averaging over draws must not move the aggregate away from the + # quota-implied share, and per-band averages stabilise: the + # youngest band's draw-averaged share over 50 draws sits within the + # spread of individual draws. + target = single_summary["displaced_weighted_share"] + assert many_summary["displaced_weighted_share"] == pytest.approx( + target, rel=0.02 + ) + band_shares = [] + for draw in range(20): + _, draw_summary = apply_employment_shock( + persons, scenario(n_draws=1, seed=100 + draw) + ) + band_shares.append( + draw_summary["by_age_band"]["16-24"]["displaced_weighted_share"] + ) + averaged = many_summary["by_age_band"]["16-24"]["displaced_weighted_share"] + assert min(band_shares) <= averaged <= max(band_shares) + + def test_youth_multiplier_shifts_displacement_to_16_24(self): + persons = synthetic_persons() + _, neutral = apply_employment_shock(persons, scenario()) + _, tilted = apply_employment_shock( + persons, scenario(youth_displacement_multiplier=3.0) + ) + + assert ( + tilted["by_age_band"]["16-24"]["displaced_weighted_share"] + > neutral["by_age_band"]["16-24"]["displaced_weighted_share"] + ) + # The group-level eq 3.4 totals are preserved under the tilt. + for label, group in tilted["by_occupation_group"].items(): + assert group["displaced_weighted"] == pytest.approx( + neutral["by_occupation_group"][label]["job_loss_quota_weighted"], + abs=persons["person_weight"].max(), + ) + # And the aggregate stays at the scenario rate. + assert tilted["displaced_weighted_share"] == pytest.approx(0.07, abs=0.005) + + def test_displaced_income_zeroed_and_flagged(self): + persons = synthetic_persons() + shocked, summary = apply_employment_shock(persons, scenario()) + + displaced = shocked[DISPLACED_COLUMN] + assert displaced.any() + assert (shocked.loc[displaced, "employment_income"] == 0.0).all() + assert (shocked.loc[displaced, PRE_SHOCK_EMPLOYMENT_INCOME_COLUMN] > 0.0).all() + # Non-displaced incomes are untouched by the employment shock. + untouched = ~displaced + pd.testing.assert_series_equal( + shocked.loc[untouched, "employment_income"], + persons.loc[untouched, "employment_income"], + ) + assert summary["returned_draw"] == 0 + + def test_fallback_exposure_groups_without_occupation_column(self): + persons = synthetic_persons().drop(columns=["occupation_group"]) + _, summary = apply_employment_shock(persons, scenario()) + assert set(summary["by_occupation_group"]) <= { + f"exposure_q{i}" for i in range(1, 6) + } + assert summary["displaced_weighted_share"] == pytest.approx(0.07, abs=0.005) + # Higher exposure quintiles bear a larger displaced share. + by_group = summary["by_occupation_group"] + assert ( + by_group["exposure_q5"]["displaced_weighted_share"] + > by_group["exposure_q1"]["displaced_weighted_share"] + ) + + def test_excluded_self_employed_weighted_in_summary(self): + persons = synthetic_persons() + # Make a slice of non-employees self-employed, plus one employee with + # a side self-employment income (not excluded) and one under-16 + # equivalent guard via employees only being 16+ already. + not_employed = ~employed_mask(persons) + self_employed = not_employed & (persons.index % 3 == 0) + persons["self_employment_income"] = 0.0 + persons.loc[self_employed, "self_employment_income"] = 15_000.0 + employee_with_side_income = persons.index[employed_mask(persons)][0] + persons.loc[employee_with_side_income, "self_employment_income"] = 2_000.0 + + _, summary = apply_employment_shock(persons, scenario()) + expected = persons.loc[self_employed, "person_weight"].sum() + assert summary["excluded_self_employed_weighted"] == pytest.approx(expected) + + def test_excluded_self_employed_zero_without_column(self): + persons = synthetic_persons() + _, summary = apply_employment_shock(persons, scenario()) + assert summary["excluded_self_employed_weighted"] == 0.0 + + +class TestWageShock: + def test_distributed_by_complementarity_with_target_mean(self): + persons = synthetic_persons() + shocked, summary = apply_wage_shock(persons, scenario()) + + recipients = employed_mask(persons) + uplift = ( + shocked.loc[recipients, "employment_income"] + / persons.loc[recipients, "employment_income"] + - 1.0 + ) + theta = persons.loc[recipients, COMPLEMENTARITY_COLUMN] + weights = persons.loc[recipients, "person_weight"] + mean_theta = (theta * weights).sum() / weights.sum() + + # Eq 3.5: person uplift proportional to theta, employment-weighted + # mean equal to the aggregate wage change. + np.testing.assert_allclose(uplift, 0.026 * theta / mean_theta) + weighted_mean = (uplift * weights).sum() / weights.sum() + assert weighted_mean == pytest.approx(0.026) + assert summary["weighted_mean_uplift"] == pytest.approx(0.026) + assert not summary["uniform_fallback"] + + def test_uniform_fallback_without_complementarity_column(self): + persons = synthetic_persons().drop(columns=[COMPLEMENTARITY_COLUMN]) + with pytest.warns(UserWarning, match="uniform wage uplift"): + shocked, summary = apply_wage_shock(persons, scenario()) + + recipients = employed_mask(persons) + uplift = ( + shocked.loc[recipients, "employment_income"] + / persons.loc[recipients, "employment_income"] + - 1.0 + ) + np.testing.assert_allclose(uplift, 0.026) + assert summary["uniform_fallback"] + + def test_wage_shock_leaves_displaced_at_zero(self): + persons = synthetic_persons() + shocked, _ = apply_employment_shock(persons, scenario()) + reshocked, _ = apply_wage_shock(shocked, scenario()) + + displaced = reshocked[DISPLACED_COLUMN] + assert displaced.any() + assert (reshocked.loc[displaced, "employment_income"] == 0.0).all() + + +class TestCapitalShock: + def test_scales_interest_and_dividends_by_return_ratio(self): + persons = synthetic_persons() + shocked, summary = apply_capital_shock(persons, scenario()) + + # ESRI anchor: 0.004 / 0.01005 ~= +39.8% capital income. + uplift = 0.004 / 0.01005 + assert uplift == pytest.approx(0.398, abs=0.001) + np.testing.assert_allclose( + shocked["savings_interest_income"], + persons["savings_interest_income"] * (1.0 + uplift), + ) + np.testing.assert_allclose( + shocked["dividend_income"], + persons["dividend_income"] * (1.0 + uplift), + ) + assert summary["capital_uplift"] == pytest.approx(uplift) + expected_change = ( + persons["person_weight"] + * (persons["savings_interest_income"] + persons["dividend_income"]) + * uplift + ).sum() + assert summary["weighted_capital_income_change"] == pytest.approx( + expected_change + ) + + def test_rent_excluded_by_default(self): + persons = synthetic_persons() + shocked, summary = apply_capital_shock(persons, scenario()) + pd.testing.assert_series_equal( + shocked["property_income"], persons["property_income"] + ) + assert "property_income" not in summary["capital_columns"] + + def test_errors_without_any_capital_column(self): + persons = synthetic_persons().drop( + columns=["savings_interest_income", "dividend_income"] + ) + with pytest.raises(ValueError, match="capital income column"): + apply_capital_shock(persons, scenario()) + + +class TestRunScenario: + def test_same_seed_identical_results(self): + persons = synthetic_persons() + first, first_summary = run_scenario(persons, scenario()) + second, second_summary = run_scenario(persons, scenario()) + + pd.testing.assert_frame_equal(first, second) + assert first_summary == second_summary + + def test_different_seed_differs(self): + persons = synthetic_persons() + first, _ = run_scenario(persons, scenario(seed=1)) + second, _ = run_scenario(persons, scenario(seed=2)) + assert not first[DISPLACED_COLUMN].equals(second[DISPLACED_COLUMN]) + + def test_preset_names_and_anchors(self): + assert set(PRESETS) == {"central", "low", "high", "central_youth_tilted"} + central = PRESETS["central"] + assert central.displacement_rate == pytest.approx(0.07) + assert central.wage_uplift == pytest.approx(0.026) + assert central.capital_uplift == pytest.approx(0.398, abs=0.001) + assert central.youth_displacement_multiplier == 1.0 + assert central.n_draws == 50 + + def test_klein_teeselink_youth_tilted_preset(self): + # Klein & Teeselink (2025): junior -5.8% vs total -4.5%. + assert KLEIN_TEESELINK_YOUTH_MULTIPLIER == pytest.approx(5.8 / 4.5) + assert KLEIN_TEESELINK_YOUTH_MULTIPLIER == pytest.approx(1.29, abs=0.01) + tilted = PRESETS["central_youth_tilted"] + central = PRESETS["central"] + # Central parameters, only the youth multiplier differs. + assert tilted.displacement_rate == central.displacement_rate + assert tilted.wage_uplift == central.wage_uplift + assert tilted.capital_return_increase == central.capital_return_increase + assert tilted.youth_displacement_multiplier == pytest.approx( + KLEIN_TEESELINK_YOUTH_MULTIPLIER + ) + # Default presets stay at pure ESRI replication (multiplier 1.0). + for name in ("central", "low", "high"): + assert PRESETS[name].youth_displacement_multiplier == 1.0 + + def test_composes_all_three_shocks(self): + persons = synthetic_persons() + shocked, summary = run_scenario(persons, scenario(), seed=99, draw_index=2) + + assert summary["parameters"]["seed"] == 99 + assert summary["employment"]["returned_draw"] == 2 + assert {"employment", "wage", "capital"} <= set(summary) + displaced = shocked[DISPLACED_COLUMN] + assert displaced.any() + assert (shocked.loc[displaced, "employment_income"] == 0.0).all() + survivors = employed_mask(persons) & ~displaced + assert ( + shocked.loc[survivors, "employment_income"] + > persons.loc[survivors, "employment_income"] + ).all() + assert (shocked["dividend_income"] > persons["dividend_income"]).all() + # Age-band summaries cover every band. + expected_labels = { + f"{lower}-{upper - 1}" if upper else f"{lower}+" + for lower, upper in AGE_BANDS + } + for shock in ("employment", "wage", "capital"): + assert set(summary[shock]["by_age_band"]) == expected_labels + + def test_age_bands_override_threads_through_all_summaries(self): + persons = synthetic_persons() + custom_bands = ((16, 30), (30, 50), (50, None)) + _, summary = run_scenario(persons, scenario(), age_bands=custom_bands) + + expected_labels = {"16-29", "30-49", "50+"} + for shock in ("employment", "wage", "capital"): + assert set(summary[shock]["by_age_band"]) == expected_labels + # Band totals partition the aggregate exactly. + total = sum( + band["displaced_weighted"] + for band in summary["employment"]["by_age_band"].values() + ) + assert total == pytest.approx(summary["employment"]["displaced_weighted"]) + + def test_age_bands_override_preserves_youth_band_semantics(self): + persons = synthetic_persons() + custom_bands = ((16, 40), (40, None)) + _, neutral = apply_employment_shock(persons, scenario(), age_bands=custom_bands) + _, tilted = apply_employment_shock( + persons, + scenario(youth_displacement_multiplier=3.0), + age_bands=custom_bands, + ) + # The multiplier still targets 16-24 by default, which shifts + # displacement into the coarse 16-39 reporting band. + assert ( + tilted["by_age_band"]["16-39"]["displaced_weighted_share"] + > neutral["by_age_band"]["16-39"]["displaced_weighted_share"] + ) + + def test_unknown_preset_raises(self): + with pytest.raises(ValueError, match="Unknown scenario preset"): + run_scenario(synthetic_persons(), "sideways") + + def test_input_table_not_mutated(self): + persons = synthetic_persons() + original = persons.copy() + run_scenario(persons, scenario()) + pd.testing.assert_frame_equal(persons, original) + + +class TestFrameInput: + def test_frame_person_table_is_extracted_with_weights(self): + pytest.importorskip("populace.frame") + from populace.frame import ( + EntitySchema, + Frame, + WeightKind, + Weights, + ) + + persons = synthetic_persons(n=400).drop(columns=["person_weight"]) + persons["person_id"] = np.arange(len(persons)) + persons["person_household_id"] = np.arange(len(persons)) + households = pd.DataFrame({"household_id": np.arange(len(persons))}) + schema = EntitySchema(group_entities=("household",)) + weights = Weights(np.full(len(persons), 1.5), WeightKind.CALIBRATED) + frame = Frame( + tables={"person": persons, "household": households}, + schema=schema, + weights={"household": weights}, + ) + + shocked, summary = apply_employment_shock(frame, scenario()) + assert isinstance(shocked, pd.DataFrame) + assert "person_weight" in shocked.columns + assert summary["displaced_weighted_share"] == pytest.approx(0.07, abs=0.01) diff --git a/packages/populace-build/tests/test_exposure_imputation.py b/packages/populace-build/tests/test_exposure_imputation.py new file mode 100644 index 00000000..f07ce5f7 --- /dev/null +++ b/packages/populace-build/tests/test_exposure_imputation.py @@ -0,0 +1,415 @@ +"""UK AI-exposure imputation: donor contract, draws, weights, planted signal. + +The LFS/APS donor and the FRS major-group merge are UKDS-licensed and never +committed, so every test builds small synthetic frames: persons carry an +observed ``soc_major_group`` (assigned independently of the demographics, so +demographics alone cannot recover it) and an exposure score with a known +structure — a per-major-group base level plus a within-group education signal +plus noise. The behavioral contracts asserted are the ones the build relies +on: fit+impute runs end to end, draws stay inside the donor's observed +support, the fit honors the frame's typed weights, the planted signals +survive imputation directionally, conditioning on the observed major group +beats the blind fallback, and the zero-model baseline reproduces the +crosswalk's employment-weighted group means exactly. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from populace.build.plan import Stage +from populace.build.uk_runtime.exposure_imputation import ( + DEFAULT_EXPOSURE_COLUMN, + SOC_MAJOR_GROUP_COLUMN, + UK_EXPOSURE_IMPUTATION_STAGE_NAME, + attach_exposure, + exposure_from_major_group, + exposure_imputation_stage, + fit_exposure_imputer, + impute_exposure, +) +from populace.frame import EntitySchema, Frame, WeightKind, Weights + +#: One-person-per-household schema, so person-level weights are unambiguous +#: and every covariate lives on the person entity. +SCHEMA = EntitySchema(group_entities=("household",)) + +#: Predictors carried by the synthetic donor and target frames: the observed +#: major group first (the primary path), then a subset of the production +#: DEFAULT_PREDICTORS — the contract under test is the fit machinery, not the +#: LFS/FRS harmonization. +PREDICTORS = [ + SOC_MAJOR_GROUP_COLUMN, + "age", + "gender", + "highest_education", + "employment_income", + "hours", +] + +#: The documented blind fallback: demographics only, no observed occupation. +BLIND_PREDICTORS = [p for p in PREDICTORS if p != SOC_MAJOR_GROUP_COLUMN] + +#: FRS adult.tab SOC2020 coding: major group as thousands. +MAJOR_GROUPS = (np.arange(1, 10) * 1000).astype("int64") + +#: Planted per-major-group base exposure. Deliberately non-monotone in the +#: code so a model interpolating the group code numerically could not fake it. +GROUP_BASE = dict( + zip( + MAJOR_GROUPS.tolist(), + [0.85, 0.75, 0.65, 0.55, 0.45, 0.35, 0.30, 0.20, 0.15], + strict=True, + ) +) + +#: Within-group education slope and noise scale of the planted exposure. +EDUCATION_SLOPE = 0.05 +NOISE_SCALE = 0.03 + +DONOR_N = 400 +TARGET_N = 250 + + +def _person_frame(columns: dict[str, np.ndarray], weights: np.ndarray) -> Frame: + """Assemble a one-person-per-household frame with person design weights.""" + n = len(weights) + person = pd.DataFrame( + { + "person_id": np.arange(n, dtype="int64"), + "person_household_id": np.arange(n, dtype="int64"), + **columns, + } + ) + household = pd.DataFrame({"household_id": np.arange(n, dtype="int64")}) + return Frame( + {"person": person, "household": household}, + SCHEMA, + {"person": Weights(values=weights, kind=WeightKind.DESIGN)}, + ) + + +def _covariates(n: int, rng: np.random.Generator) -> dict[str, np.ndarray]: + """Draw the shared covariates, with the major group independent of them.""" + education = rng.integers(0, 4, size=n).astype("int64") + income = np.exp(rng.normal(9.8, 0.5, size=n) + 0.25 * education) + return { + SOC_MAJOR_GROUP_COLUMN: rng.choice(MAJOR_GROUPS, size=n), + "age": rng.integers(18, 65, size=n).astype("int64"), + "gender": rng.integers(0, 2, size=n).astype("int64"), + "highest_education": education, + "employment_income": income, + "hours": rng.uniform(10.0, 45.0, size=n), + } + + +def _planted_exposure( + columns: dict[str, np.ndarray], rng: np.random.Generator +) -> np.ndarray: + """Exposure = major-group base + within-group education signal + noise.""" + base = np.vectorize(GROUP_BASE.get)(columns[SOC_MAJOR_GROUP_COLUMN]) + signal = EDUCATION_SLOPE * columns["highest_education"] + noise = rng.normal(0.0, NOISE_SCALE, size=len(signal)) + return np.clip(base + signal + noise, 0.0, 1.0) + + +def _synthetic_population( + n: int, seed: int, *, with_exposure: bool +) -> tuple[Frame, np.ndarray]: + """Build a frame and its true planted exposure (attached only if asked).""" + rng = np.random.default_rng(seed) + columns = _covariates(n, rng) + exposure = _planted_exposure(columns, rng) + if with_exposure: + columns = {**columns, DEFAULT_EXPOSURE_COLUMN: exposure} + return _person_frame(columns, np.full(n, 5.0)), exposure + + +@pytest.fixture +def donor_frame() -> tuple[Frame, np.ndarray]: + """A synthetic LFS-like donor: covariates, major group, exposure.""" + return _synthetic_population(DONOR_N, 1, with_exposure=True) + + +@pytest.fixture +def target_frame() -> tuple[Frame, np.ndarray]: + """A synthetic FRS-like target: covariates and major group, no exposure. + + The true (never-attached) exposure is returned alongside so error + comparisons can score the imputation against the planted ground truth. + """ + return _synthetic_population(TARGET_N, 2, with_exposure=False) + + +# ---------------------------------------------------------------------------- +# Fit + impute end to end, and draws stay in the donor's observed support +# ---------------------------------------------------------------------------- + + +def test_fit_and_impute_runs_and_draws_stay_in_donor_range( + donor_frame, target_frame +) -> None: + """One draw per target person, every draw inside the donor's support. + + The QRF draws by interpolating observed conditional quantiles, so no draw + can leave the donor's overall [min, max] — a value outside it would mean + the model extrapolated an exposure no donor row exhibits. + """ + donor, exposure = donor_frame + target, _ = target_frame + fitted = fit_exposure_imputer(donor, predictors=PREDICTORS, n_estimators=30, seed=0) + draws = impute_exposure(fitted, target) + + assert isinstance(draws, pd.Series) + assert draws.name == DEFAULT_EXPOSURE_COLUMN + assert len(draws) == TARGET_N + values = draws.to_numpy() + assert np.isfinite(values).all() + assert values.min() >= exposure.min() + assert values.max() <= exposure.max() + + +def test_higher_education_persons_draw_higher_exposure( + donor_frame, target_frame +) -> None: + """The planted within-group education signal survives directionally. + + Major groups are assigned independently of education, so across many + persons the group bases average out and the education slope must show: + target persons in the top education band draw a higher mean exposure + than those in the bottom band. + """ + donor, _ = donor_frame + target, _ = target_frame + fitted = fit_exposure_imputer(donor, predictors=PREDICTORS, n_estimators=30, seed=0) + draws = impute_exposure(fitted, target).to_numpy() + education = target.person["highest_education"].to_numpy() + + low = draws[education == 0].mean() + high = draws[education == 3].mean() + assert high > low + + +# ---------------------------------------------------------------------------- +# Primary vs blind path: the observed major group must earn its place +# ---------------------------------------------------------------------------- + + +def test_blind_fallback_warns_and_runs(donor_frame, target_frame) -> None: + """Omitting soc_major_group is allowed but loudly flagged.""" + donor, _ = donor_frame + target, _ = target_frame + with pytest.warns(UserWarning, match="blind fallback"): + fitted = fit_exposure_imputer( + donor, predictors=BLIND_PREDICTORS, n_estimators=20, seed=0 + ) + draws = impute_exposure(fitted, target) + assert len(draws) == TARGET_N + + +def test_major_group_conditioning_beats_the_blind_path( + donor_frame, target_frame +) -> None: + """Conditioning on the observed major group reduces imputation error. + + The major group carries most of the planted variance and is independent + of the demographics, so the blind path structurally cannot recover it: + its draws mix the group bases. Scored against the planted ground truth, + the within-group refinement's mean absolute error must be decisively + smaller than the blind fallback's. + """ + donor, _ = donor_frame + target, truth = target_frame + + primary = fit_exposure_imputer( + donor, predictors=PREDICTORS, n_estimators=30, seed=0 + ) + with pytest.warns(UserWarning, match="blind fallback"): + blind = fit_exposure_imputer( + donor, predictors=BLIND_PREDICTORS, n_estimators=30, seed=0 + ) + + primary_mae = np.abs(impute_exposure(primary, target).to_numpy() - truth).mean() + blind_mae = np.abs(impute_exposure(blind, target).to_numpy() - truth).mean() + + assert primary_mae < blind_mae - 0.05 + + +# ---------------------------------------------------------------------------- +# Weights: the fit reads the donor frame's typed weights and they move draws +# ---------------------------------------------------------------------------- + + +def test_donor_weights_shift_the_imputed_distribution(target_frame) -> None: + """A weighted fit reproduces the weighted, not unweighted, conditional. + + The donor mixes two exposure clusters with identical covariate + distributions: a low cluster at weight 1 and a high cluster at weight 20. + Per the populace-fit contract a Frame fit defaults to the typed design + weights, so the weighted draws' mean must land near the high cluster, + well above the unweighted (``weights="none"``) draws' mean. + """ + target, _ = target_frame + rng = np.random.default_rng(3) + n = DONOR_N + columns = _covariates(n, rng) + low_cluster = rng.uniform(0.1, 0.2, size=n) + high_cluster = rng.uniform(0.8, 0.9, size=n) + is_high = np.arange(n) % 2 == 0 + exposure = np.where(is_high, high_cluster, low_cluster) + weights = np.where(is_high, 20.0, 1.0) + donor = _person_frame({**columns, DEFAULT_EXPOSURE_COLUMN: exposure}, weights) + + weighted = fit_exposure_imputer( + donor, predictors=PREDICTORS, n_estimators=30, seed=0 + ) + unweighted = fit_exposure_imputer( + donor, predictors=PREDICTORS, weights="none", n_estimators=30, seed=0 + ) + weighted_mean = impute_exposure(weighted, target).mean() + unweighted_mean = impute_exposure(unweighted, target).mean() + + # Weighted population mean: (20*0.85 + 1*0.15) / 21 ~= 0.817; + # unweighted: (0.85 + 0.15) / 2 = 0.5. + assert weighted_mean > unweighted_mean + 0.15 + assert weighted.weight_kind == "design" + assert unweighted.weight_kind == "none" + + +# ---------------------------------------------------------------------------- +# Zero-model baseline: employment-weighted group means from the crosswalk +# ---------------------------------------------------------------------------- + + +def test_exposure_from_major_group_reproduces_weighted_means() -> None: + """The baseline is the crosswalk's employment-weighted mean per group.""" + crosswalk = pd.DataFrame( + { + SOC_MAJOR_GROUP_COLUMN: [1000, 1000, 2000, 2000], + DEFAULT_EXPOSURE_COLUMN: [0.9, 0.5, 0.3, 0.1], + "employment": [3.0, 1.0, 1.0, 1.0], + } + ) + codes = pd.Series([1000, 2000, 1000], name="whatever") + out = exposure_from_major_group(codes, crosswalk) + + # group 1000: (3*0.9 + 1*0.5) / 4 = 0.8; group 2000: (0.3 + 0.1) / 2 = 0.2. + assert out.tolist() == pytest.approx([0.8, 0.2, 0.8]) + assert out.name == DEFAULT_EXPOSURE_COLUMN + assert out.index.equals(codes.index) + + +def test_exposure_from_major_group_refuses_unknown_code() -> None: + """A person code absent from the crosswalk fails loudly, not NaN.""" + crosswalk = pd.DataFrame( + { + SOC_MAJOR_GROUP_COLUMN: [1000], + DEFAULT_EXPOSURE_COLUMN: [0.5], + "employment": [1.0], + } + ) + with pytest.raises(ValueError, match="absent from the crosswalk"): + exposure_from_major_group(np.array([1000, 9000]), crosswalk) + with pytest.raises(ValueError, match="missing column"): + exposure_from_major_group( + np.array([1000]), crosswalk.drop(columns=["employment"]) + ) + + +# ---------------------------------------------------------------------------- +# attach_exposure: immutability, pass-through, and explicit failures +# ---------------------------------------------------------------------------- + + +def test_attach_exposure_returns_new_frame_with_column( + donor_frame, target_frame +) -> None: + """The column lands on the person table; everything else passes through.""" + donor, _ = donor_frame + target, _ = target_frame + fitted = fit_exposure_imputer(donor, predictors=PREDICTORS, n_estimators=20, seed=0) + draws = impute_exposure(fitted, target) + attached = attach_exposure(target, draws) + + assert DEFAULT_EXPOSURE_COLUMN in attached.person.columns + assert attached.column_entity(DEFAULT_EXPOSURE_COLUMN) == "person" + np.testing.assert_array_equal( + attached.person[DEFAULT_EXPOSURE_COLUMN].to_numpy(), draws.to_numpy() + ) + # The input frame is untouched (frames are immutable). + assert DEFAULT_EXPOSURE_COLUMN not in target.person.columns + # Weights and strata pass through. + np.testing.assert_array_equal( + attached.weights_for("person").values, + target.weights_for("person").values, + ) + assert attached.weights_for("person").kind is WeightKind.DESIGN + assert attached.strata.tolist() == target.strata.tolist() + + +def test_attach_exposure_refuses_existing_column(donor_frame) -> None: + """Attaching onto a frame that already carries the column is refused.""" + donor, exposure = donor_frame + with pytest.raises(ValueError, match="already exists"): + attach_exposure(donor, exposure) + + +def test_attach_exposure_refuses_misaligned_or_nonfinite_values( + target_frame, +) -> None: + """Length and finiteness violations fail loudly, naming the problem.""" + target, _ = target_frame + with pytest.raises(ValueError, match="align positionally"): + attach_exposure(target, np.zeros(TARGET_N - 1)) + bad = np.full(TARGET_N, 0.5) + bad[0] = np.nan + with pytest.raises(ValueError, match="non-finite"): + attach_exposure(target, bad) + + +# ---------------------------------------------------------------------------- +# Donor contract: missing columns fail with the actionable contract message +# ---------------------------------------------------------------------------- + + +def test_fit_refuses_donor_without_exposure_column(target_frame) -> None: + """A donor missing the exposure score names the crosswalk contract.""" + target, _ = target_frame + with pytest.raises(ValueError, match="SOC->exposure"): + fit_exposure_imputer(target, predictors=PREDICTORS) + + +def test_fit_refuses_donor_missing_predictors(donor_frame) -> None: + """A donor missing a shared covariate names it and the contract.""" + donor, _ = donor_frame + with pytest.raises(ValueError, match=r"\['sic_industry_division'\]"): + fit_exposure_imputer(donor, predictors=[*PREDICTORS, "sic_industry_division"]) + + +# ---------------------------------------------------------------------------- +# Stage declaration: the plan-facing wrapper draws and attaches in one step +# ---------------------------------------------------------------------------- + + +def test_exposure_imputation_stage_produces_the_column( + donor_frame, target_frame +) -> None: + """The stage transform attaches the exposure draws to the frame.""" + donor, _ = donor_frame + target, _ = target_frame + fitted = fit_exposure_imputer(donor, predictors=PREDICTORS, n_estimators=20, seed=0) + stage = exposure_imputation_stage(fitted) + + assert isinstance(stage, Stage) + assert stage.name == UK_EXPOSURE_IMPUTATION_STAGE_NAME + assert stage.produces == (DEFAULT_EXPOSURE_COLUMN,) + # The primary path consumes the observed major group, so the plan + # executor refuses to run before the FRS adult.tab merge lands it. + assert stage.consumes == tuple(PREDICTORS) + assert SOC_MAJOR_GROUP_COLUMN in stage.consumes + assert stage.donor is not None + + out = stage.transform(target) + assert DEFAULT_EXPOSURE_COLUMN in out.person.columns diff --git a/packages/populace-build/tests/test_frs_occupation.py b/packages/populace-build/tests/test_frs_occupation.py new file mode 100644 index 00000000..a6beb5f0 --- /dev/null +++ b/packages/populace-build/tests/test_frs_occupation.py @@ -0,0 +1,166 @@ +"""FRS occupation merge: synthetic adult.tab join, missingness, stage contract. + +The real FRS adult.tab is UKDS EUL microdata and never committed; every test +builds a tiny fake adult table plus a matching person table using the +PolicyEngine-UK single-year composite-ID convention +(``person_id = SERNUM*100 + BENUNIT*10 + PERSON``). +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from populace.build.plan import Stage, StagePlan +from populace.build.uk_runtime.frs_occupation import ( + FRS_OCCUPATION_DONOR, + SOC_MAJOR_GROUP_COLUMN, + UK_FRS_OCCUPATION_STAGE_NAME, + attach_soc_major_group, + frs_major_group_to_digit, + frs_occupation_stage, + frs_person_ids, + load_frs_adult_table, + soc_major_group_for_persons, +) +from populace.frame import EntitySchema, Frame, WeightKind, Weights + +SCHEMA = EntitySchema(group_entities=("household",)) + + +def _fake_adult() -> pd.DataFrame: + """Three adults in two households; one adult with an invalid SOC.""" + + return pd.DataFrame( + { + "SERNUM": [10, 10, 20], + "BENUNIT": [1, 1, 1], + "PERSON": [1, 2, 1], + "SOC2020": [2000, 9000, -1], + } + ) + + +def _fake_person() -> pd.DataFrame: + """The two households' persons: three adults plus one child (no adult row).""" + + return pd.DataFrame( + { + "person_id": [1011, 1012, 2011, 2012], + "person_household_id": [10, 10, 20, 20], + "age": [40, 38, 30, 5], + } + ) + + +def _frame(person: pd.DataFrame) -> Frame: + household = pd.DataFrame( + {"household_id": sorted(person["person_household_id"].unique())} + ) + return Frame( + {"person": person, "household": household}, + SCHEMA, + {"household": Weights(values=np.ones(len(household)), kind=WeightKind.DESIGN)}, + ) + + +def test_frs_person_ids_composite_convention(): + ids = frs_person_ids(_fake_adult()) + assert ids.tolist() == [1011, 1012, 2011] + + +def test_frs_person_ids_requires_key_columns(): + with pytest.raises(ValueError, match="key column"): + frs_person_ids(pd.DataFrame({"SERNUM": [1], "BENUNIT": [1]})) + + +def test_join_correctness_and_missingness_report(): + values, report = soc_major_group_for_persons(_fake_person(), _fake_adult()) + assert values.name == SOC_MAJOR_GROUP_COLUMN + # Adults with valid SOC keep the raw 1000-9000 coding. + assert values.iloc[0] == 2000.0 + assert values.iloc[1] == 9000.0 + # Adult with invalid SOC (-1) and the child both stay NaN. + assert np.isnan(values.iloc[2]) + assert np.isnan(values.iloc[3]) + assert report == { + "n_persons": 4, + "adults_matched": 3, + "adults_with_valid_soc": 2, + "adults_missing_soc": 1, + "persons_without_adult_row": 1, + } + + +def test_duplicate_adult_keys_rejected(): + adult = pd.concat([_fake_adult(), _fake_adult().iloc[[0]]], ignore_index=True) + with pytest.raises(ValueError, match="duplicated SERNUM/BENUNIT/PERSON"): + soc_major_group_for_persons(_fake_person(), adult) + + +def test_load_frs_adult_table_reads_tab_delimited(tmp_path): + path = tmp_path / "adult.tab" + _fake_adult().assign(EXTRA=1).to_csv(path, sep="\t", index=False) + table = load_frs_adult_table(path) + assert sorted(table.columns) == ["BENUNIT", "PERSON", "SERNUM", "SOC2020"] + assert len(table) == 3 + + +def test_load_frs_adult_table_missing_column(tmp_path): + path = tmp_path / "adult.tab" + _fake_adult().drop(columns=["SOC2020"]).to_csv(path, sep="\t", index=False) + with pytest.raises(ValueError, match="SOC2020"): + load_frs_adult_table(path) + + +def test_frs_major_group_to_digit(): + digits = frs_major_group_to_digit([1000, 9000, -1, np.nan, 4000]) + assert digits.tolist() == ["1", "9", None, None, "4"] + + +def test_attach_soc_major_group_on_frame(): + frame = attach_soc_major_group(_frame(_fake_person()), _fake_adult()) + column = frame.person[SOC_MAJOR_GROUP_COLUMN] + assert column.tolist()[:2] == [2000.0, 9000.0] + assert column.isna().sum() == 2 + + +def test_attach_refuses_double_run(): + frame = attach_soc_major_group(_frame(_fake_person()), _fake_adult()) + with pytest.raises(ValueError, match="already exists"): + attach_soc_major_group(frame, _fake_adult()) + + +def test_stage_contract(): + stage = frs_occupation_stage(_fake_adult()) + assert isinstance(stage, Stage) + assert stage.name == UK_FRS_OCCUPATION_STAGE_NAME + assert stage.produces == (SOC_MAJOR_GROUP_COLUMN,) + assert stage.consumes == ("person_id",) + assert stage.donor == FRS_OCCUPATION_DONOR + out = stage.transform(_frame(_fake_person())) + assert SOC_MAJOR_GROUP_COLUMN in out.person.columns + + +def test_stage_reads_adult_tab_path_lazily(tmp_path): + path = tmp_path / "adult.tab" + stage = frs_occupation_stage(path) # not written yet: read at run time + _fake_adult().to_csv(path, sep="\t", index=False) + out = stage.transform(_frame(_fake_person())) + assert out.person[SOC_MAJOR_GROUP_COLUMN].iloc[0] == 2000.0 + + +def test_stage_slots_before_exposure_imputation_in_a_plan(): + """StagePlan accepts frs_occupation producing what a consumer needs.""" + + consumer = Stage( + name="ai_exposure_imputation", + transform=lambda frame: frame, + consumes=(SOC_MAJOR_GROUP_COLUMN,), + ) + plan = StagePlan([frs_occupation_stage(_fake_adult()), consumer]) + assert [stage.name for stage in plan.stages] == [ + UK_FRS_OCCUPATION_STAGE_NAME, + "ai_exposure_imputation", + ] diff --git a/packages/populace-build/tests/test_occupation_targets.py b/packages/populace-build/tests/test_occupation_targets.py new file mode 100644 index 00000000..c6f4bf09 --- /dev/null +++ b/packages/populace-build/tests/test_occupation_targets.py @@ -0,0 +1,171 @@ +from __future__ import annotations + +import pytest + +from populace.build.uk_runtime import ( + aps_age_band_employment_targets, + aps_occupation_employment_targets, + ashe_occupation_employment_targets, +) +from populace.build.uk_runtime.occupation_targets import ( + PACKAGED_APS_AGE_CSV, + packaged_occupation_csv_path, +) + +ASHE_FIXTURE = """soc_code,soc_title,employment_jobs,median_annual_pay +1111,Chief executives and senior officials,133000,89835 +1112,Elected officers and representatives,, +2211,Generalist medical practitioners,110000,76512 +221,Health professionals aggregate,505000,52000 +2211,Generalist medical practitioners,999,1 +6135,Care workers and home carers,714000,23456 +""" + +APS_OCC_FIXTURE = """soc_code,soc_title,employment,period,geography +11,Corporate Managers and Directors,2619900,Jan 2025-Dec 2025,United Kingdom +12,Other Managers and Proprietors,1189900,Jan 2025-Dec 2025,United Kingdom +92,Elementary Administration and Service Occupations,not_a_number,Jan 2025-Dec 2025,United Kingdom +""" + +APS_AGE_FIXTURE = """age_band,employment,period,geography +16+,33321600,Jan 2025-Dec 2025,United Kingdom +16-24,3582100,Jan 2025-Dec 2025,United Kingdom +65+,1590500,Jan 2025-Dec 2025,United Kingdom +""" + +#: The ai_shock_scenarios reporting bands the packaged age CSV must match. +SCENARIO_AGE_BANDS = {"16-24", "25-34", "35-44", "45-54", "55-64", "65+"} + + +def _write(tmp_path, name, text): + path = tmp_path / name + path.write_text(text) + return path + + +class TestAsheOccupationEmploymentTargets: + def test_builds_targets_matching_fixture_values(self, tmp_path): + build = ashe_occupation_employment_targets( + _write(tmp_path, "ashe.csv", ASHE_FIXTURE) + ) + by_name = {target.name: target for target in build.target_set} + assert len(build.target_set) == 3 + assert by_name["ons/ashe14/occupation/1111/employment_jobs"].value == 133000 + assert by_name["ons/ashe14/occupation/2211/employment_jobs"].value == 110000 + assert by_name["ons/ashe14/occupation/6135/employment_jobs"].value == 714000 + for target in build.target_set: + assert target.entity == "person" + assert target.period == 2025 + assert target.measure.startswith("employee_jobs/occupation/") + assert "ASHE" in target.source + assert "2025 provisional" in target.source + + def test_bad_rows_are_skipped_and_reported_not_silently_dropped(self, tmp_path): + build = ashe_occupation_employment_targets( + _write(tmp_path, "ashe.csv", ASHE_FIXTURE) + ) + reasons = {entry.row_number: entry.reason for entry in build.skipped} + # Row 2: suppressed (blank) job count; row 4: 3-digit aggregate code; + # row 5: duplicate unit group. + assert set(reasons) == {2, 4, 5} + assert "employment_jobs" in reasons[2] + assert "four-digit" in reasons[4] + assert "duplicate" in reasons[5] + # The offending rows travel with the report. + skipped_codes = {entry.row["soc_code"] for entry in build.skipped} + assert skipped_codes == {"1112", "221", "2211"} + + def test_missing_required_column_raises(self, tmp_path): + path = _write( + tmp_path, "bad.csv", "soc_code,soc_title\n1111,Chief executives\n" + ) + with pytest.raises(ValueError, match="employment_jobs"): + ashe_occupation_employment_targets(path) + + def test_all_rows_skipped_raises_rather_than_empty_set(self, tmp_path): + path = _write( + tmp_path, + "empty.csv", + "soc_code,soc_title,employment_jobs,median_annual_pay\n" + "1111,Chief executives,,\n", + ) + with pytest.raises(ValueError, match="every row was skipped"): + ashe_occupation_employment_targets(path) + + +class TestApsOccupationEmploymentTargets: + def test_builds_targets_matching_fixture_values(self, tmp_path): + build = aps_occupation_employment_targets( + _write(tmp_path, "aps_occ.csv", APS_OCC_FIXTURE) + ) + by_name = {target.name: target for target in build.target_set} + assert len(build.target_set) == 2 + target = by_name["ons/aps/occupation_submajor/11/employment"] + assert target.value == 2619900 + assert target.entity == "person" + assert target.measure == "employment/occupation_submajor/11" + assert target.period == "Jan 2025-Dec 2025" + assert "NM_17_1" in target.source + assert "United Kingdom" in target.source + assert by_name["ons/aps/occupation_submajor/12/employment"].value == 1189900 + + def test_bad_employment_is_skipped_and_reported(self, tmp_path): + build = aps_occupation_employment_targets( + _write(tmp_path, "aps_occ.csv", APS_OCC_FIXTURE) + ) + assert len(build.skipped) == 1 + (entry,) = build.skipped + assert entry.row_number == 3 + assert "not_a_number" in entry.reason + assert entry.row["soc_code"] == "92" + + +class TestApsAgeBandEmploymentTargets: + def test_total_band_is_excluded_by_default_and_reported(self, tmp_path): + build = aps_age_band_employment_targets( + _write(tmp_path, "aps_age.csv", APS_AGE_FIXTURE) + ) + names = [target.name for target in build.target_set] + assert names == [ + "ons/aps/employment/age/16_24", + "ons/aps/employment/age/65plus", + ] + by_name = {target.name: target for target in build.target_set} + assert by_name["ons/aps/employment/age/16_24"].value == 3582100 + assert by_name["ons/aps/employment/age/65plus"].value == 1590500 + assert by_name["ons/aps/employment/age/65plus"].measure == ( + "employment/age/65plus" + ) + (entry,) = build.skipped + assert entry.row["age_band"] == "16+" + assert "roll-up" in entry.reason + + def test_include_total_keeps_the_roll_up(self, tmp_path): + build = aps_age_band_employment_targets( + _write(tmp_path, "aps_age.csv", APS_AGE_FIXTURE), + include_total=True, + ) + by_name = {target.name: target for target in build.target_set} + assert by_name["ons/aps/employment/age/16plus"].value == 33321600 + assert build.skipped == () + + +class TestPackagedApsAgeCsv: + def test_band_set_equals_ai_shock_scenario_reporting_bands(self): + build = aps_age_band_employment_targets( + packaged_occupation_csv_path(PACKAGED_APS_AGE_CSV) + ) + bands = {target.source.rsplit("aged ", 1)[1] for target in build.target_set} + assert bands == SCENARIO_AGE_BANDS + + def test_bands_sum_to_the_16_plus_total_within_rounding(self): + build = aps_age_band_employment_targets( + packaged_occupation_csv_path(PACKAGED_APS_AGE_CSV), + include_total=True, + ) + by_name = {target.name: target for target in build.target_set} + total = by_name.pop("ons/aps/employment/age/16plus").value + residual = total - sum(target.value for target in by_name.values()) + # Published values are rounded to the nearest hundred; six bands can + # drift from the roll-up by at most half that per band. + assert abs(residual) <= 6 * 50