diff --git a/packages/populace-build/src/populace/build/us/country_package.json b/packages/populace-build/src/populace/build/us/country_package.json index 07708026..1cf8d83c 100644 --- a/packages/populace-build/src/populace/build/us/country_package.json +++ b/packages/populace-build/src/populace/build/us/country_package.json @@ -10,6 +10,7 @@ "obbba_reforms.json", "puf_aggregate_record_disaggregation.json", "release_input_coverage_manifest.json", + "snap_fy2022_participation_rates.json", "soca_capital_gain_distribution_shares.json", "soi_baseline_levels.json", "source_stages.json", diff --git a/packages/populace-build/src/populace/build/us/snap_fy2022_participation_rates.json b/packages/populace-build/src/populace/build/us/snap_fy2022_participation_rates.json new file mode 100644 index 00000000..2738360e --- /dev/null +++ b/packages/populace-build/src/populace/build/us/snap_fy2022_participation_rates.json @@ -0,0 +1,72 @@ +{ + "schema_version": 1, + "classification": "validation_reference", + "title": "Reaching Those in Need: Estimates of State SNAP Participation Rates in 2022", + "publisher": "USDA Food and Nutrition Service", + "landing_page": "https://www.fns.usda.gov/research/snap/state-participation-rates/2022", + "report": "https://www.fns.usda.gov/sites/default/files/resource-files/ear-snap-Reaching-Those-in-Need-2022.pdf", + "table": "Estimates of participation rates (percentage)", + "fiscal_year": 2022, + "measure": "share_of_eligible_people_participating_in_average_month", + "national_rate": 0.88, + "source_precision": "whole_percentage_point", + "caveats": [ + "These eligible-person rates are an independent validator, not the FY2024 household-caseload seeding or calibration basis.", + "FNS estimates have substantial sampling and model uncertainty and are published as whole percentages.", + "FNS applies Federal income and resource rules and excludes people eligible solely through State categorical eligibility policies, so its eligibility denominator can differ from PolicyEngine current-law eligibility.", + "FNS caps estimates above 100 percent at 100 percent; a published value of 100 percent does not establish universal participation." + ], + "state_rates": { + "01": 0.90, + "02": 0.73, + "04": 0.77, + "05": 0.59, + "06": 0.81, + "08": 1.00, + "09": 0.98, + "10": 0.91, + "11": 1.00, + "12": 0.81, + "13": 0.92, + "15": 0.81, + "16": 0.73, + "17": 1.00, + "18": 0.89, + "19": 0.98, + "20": 0.79, + "21": 0.75, + "22": 0.99, + "23": 0.94, + "24": 0.85, + "25": 1.00, + "26": 1.00, + "27": 0.93, + "28": 0.74, + "29": 0.92, + "30": 0.75, + "31": 0.93, + "32": 0.98, + "33": 0.82, + "34": 0.91, + "35": 1.00, + "36": 0.91, + "37": 0.95, + "38": 0.81, + "39": 0.99, + "40": 0.98, + "41": 1.00, + "42": 1.00, + "44": 1.00, + "45": 0.76, + "46": 0.84, + "47": 0.84, + "48": 0.74, + "49": 0.76, + "50": 0.99, + "51": 0.83, + "53": 1.00, + "54": 0.98, + "55": 1.00, + "56": 0.63 + } +} diff --git a/packages/populace-build/src/populace/build/us/target_parity_manifest.json b/packages/populace-build/src/populace/build/us/target_parity_manifest.json index 6ae1d839..c905179e 100644 --- a/packages/populace-build/src/populace/build/us/target_parity_manifest.json +++ b/packages/populace-build/src/populace/build/us/target_parity_manifest.json @@ -536,15 +536,7 @@ } }, "irs_soi.form_w2_social_security_tips": { - "status": "reviewed_exclusion", - "classification": "not_modeled", - "reason": "IRS W-2 Social Security tips \u2014 a real us-data target (PR #220) the model cannot yet satisfy: PolicyEngine-US produces a structural zero for tip_income in the populace base microdata (no tip source column), so the target is unsatisfiable until the tip-imputation source stage is ported.", - "evidence": "US_FISCAL_TARGET_SUPPORT_EXCLUSIONS entry irs_soi.ty2023.form_w2_social_security_tips.box_7_social_security_tips.return_count", - "fence": { - "origin": "us-data PR #220 (Impute tips, fixes #215)", - "purpose": "\"Proposals such as 'No Tax on Tips' require a clean tip_income field distinct from regular wages ... Tipped workers skew lower-income; omitting tips biases poverty, EITC/CTC, and payroll-tax results\" (issue #215)", - "verdict_basis": "deferred (not source-absent \u2014 the feed carries a tip amount): PolicyEngine-US yields a structural zero for tip_income in the populace base microdata (US_FISCAL_TARGET_SUPPORT_EXCLUSIONS), so the target is unsatisfiable until the SIPP/ORG tip-imputation source stage (us-data #220) is ported. Wiring it now would ship a 0-vs-target gap, not a fit." - } + "status": "compiled" }, "irs_soi.historic_table_2": { "status": "compiled" diff --git a/packages/populace-build/src/populace/build/us_runtime/fiscal_targets.py b/packages/populace-build/src/populace/build/us_runtime/fiscal_targets.py index 0090b4d2..bb9b51cf 100644 --- a/packages/populace-build/src/populace/build/us_runtime/fiscal_targets.py +++ b/packages/populace-build/src/populace/build/us_runtime/fiscal_targets.py @@ -562,11 +562,6 @@ "support in the under-$1 AGI slice; this narrow offset-income cell needs " "richer state/tail support before it can be calibrated." ), - "irs_soi.ty2023.form_w2_social_security_tips.box_7_social_security_tips.return_count": ( - "Current US support does not yet materialize a positive tip_income source " - "column; W-2 Social Security tip return counts need the SIPP/ORG tip " - "source stage wired into the fiscal refresh before calibration." - ), "hhs_acf_tanf.fy2024.cash_assistance.ar.basic_assistance_excluding_relative_foster_care_and_adoption_guardianship.all_funds": ( "Current 2024 base microdata have zero positive TANF benefit support " "in Arkansas under PolicyEngine-US state TANF formulas." diff --git a/packages/populace-build/src/populace/build/us_runtime/snap_release_acceptance.py b/packages/populace-build/src/populace/build/us_runtime/snap_release_acceptance.py new file mode 100644 index 00000000..1ab499f9 --- /dev/null +++ b/packages/populace-build/src/populace/build/us_runtime/snap_release_acceptance.py @@ -0,0 +1,702 @@ +"""Post-build SNAP acceptance checks for a certified US release candidate. + +The release builder already fails closed while constructing the SNAP take-up +surface. This module checks the persisted candidate and its diagnostics as a +separate acceptance step: + +* every state household-caseload and benefit target is present and fitted; +* the ARTIFACT's simulated caseloads (engine ``snap > 0`` from the shipped H5) + also fit the FNS household targets — the calibrated column alone can hit + its targets while the shipped dataset diverges (populace#419); +* the stage-time state take-up gate still reports a complete, anchor-preserving + surface, with California saturation made explicit; +* county FIPS coverage is complete and internally consistent, including a + non-collapsed Alaska borough/census-area distribution; and +* modeled eligible-person participation is compared with USDA FNS FY2022 state + estimates as an advisory validator only. + +The participation comparison deliberately does not seed take-up and does not +enter the release pass/fail verdict. It has a different denominator and +vintage from the FY2024 average-monthly household targets used by calibration. +""" + +from __future__ import annotations + +import json +import math +from collections import Counter +from collections.abc import Mapping +from importlib.resources import files +from pathlib import Path +from typing import Any + +import pandas as pd + +from populace.build.gates import GateResult +from populace.build.us_runtime.snap_state_take_up import us_snap_state_take_up_gate + +SNAP_PARTICIPATION_REFERENCE_RESOURCE = "snap_fy2022_participation_rates.json" +SNAP_STATE_FIPS = frozenset( + { + "01", + "02", + "04", + "05", + "06", + "08", + "09", + "10", + "11", + "12", + "13", + "15", + "16", + "17", + "18", + "19", + "20", + "21", + "22", + "23", + "24", + "25", + "26", + "27", + "28", + "29", + "30", + "31", + "32", + "33", + "34", + "35", + "36", + "37", + "38", + "39", + "40", + "41", + "42", + "44", + "45", + "46", + "47", + "48", + "49", + "50", + "51", + "53", + "54", + "55", + "56", + } +) + +__all__ = [ + "SNAP_PARTICIPATION_REFERENCE_RESOURCE", + "SNAP_STATE_FIPS", + "assemble_us_snap_release_acceptance", + "load_snap_fy2022_participation_reference", + "us_snap_artifact_caseload_gate", + "us_snap_simulated_state_caseloads", + "us_snap_core_release_gate", + "us_snap_county_coverage_gate", + "us_snap_participation_validation", + "us_snap_state_target_fit_gate", +] + + +def load_snap_fy2022_participation_reference() -> dict[str, Any]: + """Load and validate the packaged FNS FY2022 eligible-person rates.""" + payload = json.loads( + files("populace.build.us") + .joinpath(SNAP_PARTICIPATION_REFERENCE_RESOURCE) + .read_text() + ) + rates = payload.get("state_rates") + if not isinstance(rates, dict): + raise ValueError("SNAP participation reference needs a state_rates mapping.") + normalized = {str(state).zfill(2): float(rate) for state, rate in rates.items()} + if set(normalized) != SNAP_STATE_FIPS: + missing = sorted(SNAP_STATE_FIPS - set(normalized)) + unexpected = sorted(set(normalized) - SNAP_STATE_FIPS) + raise ValueError( + "SNAP participation reference must cover 50 states plus DC: " + f"missing={missing}, unexpected={unexpected}." + ) + invalid = { + state: rate for state, rate in normalized.items() if not 0.0 < rate <= 1.0 + } + if invalid: + raise ValueError(f"SNAP participation reference has invalid rates: {invalid}.") + payload["state_rates"] = normalized + return payload + + +def _gate_payload(result: GateResult, *, required: bool = True) -> dict[str, Any]: + return { + "required": required, + "passed": bool(result.passed), + "failures": list(result.failures), + "details": result.details, + } + + +def us_snap_core_release_gate(build_manifest: dict[str, Any]) -> GateResult: + """Require every explicit top-level release gate to have passed.""" + gates = build_manifest.get("gates") + failures: list[str] = [] + checked: list[str] = [] + if not isinstance(gates, dict): + failures.append("build_manifest.json has no gates object.") + gates = {} + for name, payload in sorted(gates.items()): + if not isinstance(payload, dict) or "passed" not in payload: + continue + checked.append(name) + if not bool(payload["passed"]): + failures.append(f"build gate {name!r} did not pass.") + if not checked: + failures.append("build_manifest.json records no explicit passed release gates.") + return GateResult( + name="snap_core_release", + passed=not failures, + failures=tuple(failures), + details={"checked_gates": checked}, + ) + + +def us_snap_state_target_fit_gate( + calibration_diagnostics: dict[str, Any], + *, + target_role: str, + label: str, + relative_tolerance: float = 0.10, +) -> GateResult: + """Require 51 state target rows and bound every absolute relative miss.""" + if not 0.0 < relative_tolerance < 1.0: + raise ValueError("relative_tolerance must be in (0, 1).") + targets = calibration_diagnostics.get("targets") + if not isinstance(targets, list): + targets = [] + selected = [ + row + for row in targets + if isinstance(row, dict) + and isinstance(row.get("metadata"), dict) + and row["metadata"].get("target_role") == target_role + and row["metadata"].get("state_fips") is not None + ] + states = [str(row["metadata"]["state_fips"]).zfill(2) for row in selected] + counts = Counter(states) + duplicates = sorted(state for state, count in counts.items() if count != 1) + missing = sorted(SNAP_STATE_FIPS - set(states)) + unexpected = sorted(set(states) - SNAP_STATE_FIPS) + failures: list[str] = [] + if missing: + failures.append(f"{label}: missing state target rows {missing}.") + if unexpected: + failures.append(f"{label}: unexpected state target rows {unexpected}.") + if duplicates: + failures.append(f"{label}: duplicate state target rows {duplicates}.") + + rows: list[dict[str, Any]] = [] + ordered = sorted(zip(states, selected, strict=True), key=lambda item: item[0]) + for state, row in ordered: + target = float(row.get("target", 0.0)) + estimate = float(row.get("final_estimate", math.nan)) + relative_error = ( + (estimate - target) / target + if target > 0 and math.isfinite(estimate) + else math.nan + ) + within = math.isfinite(relative_error) and ( + abs(relative_error) <= relative_tolerance + ) + rows.append( + { + "state_fips": state, + "target_name": row.get("target_name", row.get("name")), + "target": target, + "final_estimate": estimate if math.isfinite(estimate) else None, + "relative_error": ( + relative_error if math.isfinite(relative_error) else None + ), + "within_tolerance": within, + } + ) + if not within: + rendered_error = ( + f"{relative_error:+.1%}" + if math.isfinite(relative_error) + else "not finite" + ) + failures.append( + f"{label}: state {state} relative miss {rendered_error} exceeds " + f"{relative_tolerance:.1%}." + ) + + finite_errors = [ + abs(row["relative_error"]) for row in rows if row["relative_error"] is not None + ] + return GateResult( + name=f"snap_{target_role}_state_fit", + passed=not failures, + failures=tuple(failures), + details={ + "label": label, + "target_role": target_role, + "relative_tolerance": relative_tolerance, + "state_rows": len(rows), + "max_absolute_relative_error": ( + max(finite_errors) if finite_errors else None + ), + "states_outside_tolerance": [ + row["state_fips"] for row in rows if not row["within_tolerance"] + ], + "rows": rows, + }, + ) + + +def _normalized_fips(series: pd.Series, *, width: int) -> pd.Series: + return series.astype("string").str.replace(r"\.0$", "", regex=True).str.zfill(width) + + +def us_snap_county_coverage_gate(h5_path: Path | str) -> GateResult: + """Require complete state-consistent county FIPS and non-collapsed Alaska.""" + h5_path = Path(h5_path) + household = pd.read_hdf( + h5_path, + "household", + columns=["state_fips", "county_fips"], + ) + missing_state = int(household["state_fips"].isna().sum()) + missing_county = int(household["county_fips"].isna().sum()) + state = _normalized_fips(household["state_fips"], width=2) + county = _normalized_fips(household["county_fips"], width=5) + invalid_state = int((state.str.fullmatch(r"\d{2}") != True).sum()) # noqa: E712 + invalid_county = int((county.str.fullmatch(r"\d{5}") != True).sum()) # noqa: E712 + prefix_mismatch = int((county.str[:2] != state).fillna(True).sum()) + represented_states = set(state.dropna().astype(str)) + missing_states = sorted(SNAP_STATE_FIPS - represented_states) + alaska = county[state == "02"].dropna().astype(str) + alaska_counties = sorted(alaska.unique().tolist()) + failures: list[str] = [] + if missing_state or missing_county: + failures.append( + "household geography has missing values: " + f"state_fips={missing_state}, county_fips={missing_county}." + ) + if invalid_state or invalid_county: + failures.append( + "household geography has invalid FIPS formatting: " + f"state_fips={invalid_state}, county_fips={invalid_county}." + ) + if prefix_mismatch: + failures.append( + f"{prefix_mismatch} household county FIPS values disagree with state_fips." + ) + if missing_states: + failures.append(f"household geography is missing states {missing_states}.") + if len(alaska_counties) < 2: + failures.append( + "Alaska county/borough geography collapsed to fewer than two FIPS values." + ) + return GateResult( + name="snap_county_coverage", + passed=not failures, + failures=tuple(failures), + details={ + "household_records": int(len(household)), + "missing_state_fips": missing_state, + "missing_county_fips": missing_county, + "invalid_state_fips": invalid_state, + "invalid_county_fips": invalid_county, + "state_county_prefix_mismatches": prefix_mismatch, + "represented_state_count": len(represented_states & SNAP_STATE_FIPS), + "alaska_household_records": int(len(alaska)), + "alaska_unique_county_fips": len(alaska_counties), + "alaska_county_fips": alaska_counties, + "geography_semantics": ( + "Calibrated allocation from the Populace geography ladder, not " + "measured household location." + ), + }, + ) + + +def _spm_unit_state_fips(h5_path: Path) -> pd.Series: + spm_unit = pd.read_hdf(h5_path, "spm_unit", columns=["spm_unit_id"]) + person = pd.read_hdf( + h5_path, + "person", + columns=["person_spm_unit_id", "person_household_id"], + ) + household = pd.read_hdf( + h5_path, + "household", + columns=["household_id", "state_fips"], + ) + household_counts = person.groupby("person_spm_unit_id")[ + "person_household_id" + ].nunique() + if (household_counts != 1).any(): + examples = household_counts[household_counts != 1].index[:5].tolist() + raise ValueError( + "SNAP participation validation requires each SPM unit to occupy one " + f"household; invalid SPM unit ids include {examples}." + ) + spm_to_household = person.drop_duplicates("person_spm_unit_id").set_index( + "person_spm_unit_id" + )["person_household_id"] + household_state = household.set_index("household_id")["state_fips"] + aligned = spm_unit["spm_unit_id"].map(spm_to_household).map(household_state) + if aligned.isna().any(): + raise ValueError( + "SNAP participation validation could not map every SPM unit to a state." + ) + return _normalized_fips(aligned, width=2) + + +def us_snap_participation_validation( + h5_path: Path | str, + *, + period: int = 2024, + tolerance_percentage_points: float = 10.0, +) -> dict[str, Any]: + """Compare post-calibration modeled participation with FNS FY2022 rates. + + PolicyEngine ``MicroSeries`` objects retain the release's calibrated + weights through multiplication, filtering, and ``sum``; no weight vector + is loaded or applied manually here. + """ + if not 0.0 < tolerance_percentage_points <= 100.0: + raise ValueError("tolerance_percentage_points must be in (0, 100].") + from policyengine_us import Microsimulation + from policyengine_us.data import USSingleYearDataset + + h5_path = Path(h5_path) + reference = load_snap_fy2022_participation_reference() + dataset = USSingleYearDataset(file_path=str(h5_path)) + simulation = Microsimulation(dataset=dataset) + eligible = simulation.calc("is_snap_eligible", period=period, map_to="spm_unit") + takes_up = simulation.calc( + "takes_up_snap_if_eligible", period=period, map_to="spm_unit" + ) + snap_unit_size = simulation.calc("snap_unit_size", period=period, map_to="spm_unit") + eligible_people = eligible * snap_unit_size + participating_people = eligible_people * takes_up + state_fips = _spm_unit_state_fips(h5_path) + if len(state_fips) != len(eligible_people): + raise ValueError( + "SNAP participation validation state and simulation rows do not align: " + f"{len(state_fips)} states, {len(eligible_people)} simulation rows." + ) + state_fips.index = eligible_people.index + + rows: list[dict[str, Any]] = [] + target_participating_total = 0.0 + for state, target_rate in sorted(reference["state_rates"].items()): + mask = state_fips == state + eligible_total = float(eligible_people[mask].sum()) + participating_total = float(participating_people[mask].sum()) + modeled_rate = ( + participating_total / eligible_total if eligible_total > 0 else None + ) + difference_pp = ( + 100.0 * (modeled_rate - target_rate) if modeled_rate is not None else None + ) + within = difference_pp is not None and ( + abs(difference_pp) <= tolerance_percentage_points + ) + target_participating_total += eligible_total * target_rate + rows.append( + { + "state_fips": state, + "fns_fy2022_rate": target_rate, + "modeled_rate": modeled_rate, + "difference_percentage_points": difference_pp, + "within_tolerance": within, + "fns_rate_capped_at_100": target_rate == 1.0, + "modeled_eligible_people": eligible_total, + "modeled_participating_people": participating_total, + } + ) + + eligible_national = float(eligible_people.sum()) + participating_national = float(participating_people.sum()) + modeled_national_rate = ( + participating_national / eligible_national if eligible_national > 0 else None + ) + state_mix_reference_rate = ( + target_participating_total / eligible_national + if eligible_national > 0 + else None + ) + misses = [row for row in rows if not row["within_tolerance"]] + finite_differences = [ + abs(row["difference_percentage_points"]) + for row in rows + if row["difference_percentage_points"] is not None + ] + return { + "required": False, + "passed_within_advisory_tolerance": not misses, + "advisory_tolerance_percentage_points": tolerance_percentage_points, + "reason_not_release_blocking": ( + "FNS FY2022 estimates measure eligible people and carry sampling/model " + "uncertainty; release calibration uses FY2024 average-monthly households." + ), + "source": { + key: reference[key] + for key in ( + "title", + "publisher", + "landing_page", + "report", + "table", + "fiscal_year", + "measure", + "national_rate", + "source_precision", + "caveats", + ) + }, + "national": { + "modeled_eligible_people": eligible_national, + "modeled_participating_people": participating_national, + "modeled_rate": modeled_national_rate, + "published_fns_rate": reference["national_rate"], + "state_mix_weighted_fns_rate": state_mix_reference_rate, + }, + "states_outside_advisory_tolerance": [row["state_fips"] for row in misses], + "max_absolute_difference_percentage_points": ( + max(finite_differences) if finite_differences else None + ), + "rows": rows, + } + + +def _diagnostics_state_targets( + calibration_diagnostics: dict[str, Any], *, target_role: str +) -> dict[str, float]: + """Per-state target values the calibrated release recorded for a role.""" + targets = calibration_diagnostics.get("targets") + if not isinstance(targets, list): + targets = [] + out: dict[str, float] = {} + for row in targets: + if not isinstance(row, dict): + continue + metadata = row.get("metadata") + if not isinstance(metadata, dict): + continue + if metadata.get("target_role") != target_role: + continue + state = metadata.get("state_fips") + if state is None: + continue + out[str(state).zfill(2)] = float(row.get("target", 0.0)) + return out + + +def us_snap_simulated_state_caseloads( + h5_path: Path | str, *, period: int = 2024 +) -> dict[str, float]: + """Weighted SNAP taker-household count per state, simulated from the artifact. + + This is the caseload a downstream PolicyEngine-US consumer computes from + the release: SPM units with engine ``snap > 0`` under the shipped weights + and take-up flags. ``MicroSeries`` retains the release's calibrated + weights through comparison, filtering, and ``sum``. + """ + from policyengine_us import Microsimulation + from policyengine_us.data import USSingleYearDataset + + h5_path = Path(h5_path) + dataset = USSingleYearDataset(file_path=str(h5_path)) + simulation = Microsimulation(dataset=dataset) + snap = simulation.calc("snap", period=period, map_to="spm_unit") + takers = snap > 0 + state_fips = _spm_unit_state_fips(h5_path) + if len(state_fips) != len(takers): + raise ValueError( + "SNAP artifact caseload state and simulation rows do not align: " + f"{len(state_fips)} states, {len(takers)} simulation rows." + ) + state_fips.index = takers.index + return { + str(state): float(takers[state_fips == state].sum()) + for state in sorted(set(state_fips)) + } + + +def us_snap_artifact_caseload_gate( + h5_path: Path | str, + calibration_diagnostics: dict[str, Any], + *, + period: int = 2024, + relative_tolerance: float = 0.10, + simulated_caseloads: Mapping[str, float] | None = None, +) -> GateResult: + """Require the ARTIFACT's simulated caseloads to fit the FNS targets. + + The calibrated-column fit gate (``us_snap_state_target_fit_gate``) proves + the solver hit the materialized caseload measure; this gate proves the + thing a downstream consumer actually computes from the shipped H5 agrees. + The two can diverge when the take-up stage's assignment basis or the + export's input surface differs from the build frame — populace#419 + measured a release with a 52/52 calibrated fit whose artifact-simulated + caseloads missed 23/52 states (AK +563%). This gate fails closed on that + class. + + Semantics note (recorded in details): the FNS targets are fiscal-year + average-monthly household stocks; the simulated measure is annual + engine participation (``snap > 0``) under the shipped take-up flags. + + Args: + h5_path: The release H5. + calibration_diagnostics: The release's calibration diagnostics; the + per-state ``snap_households`` target values are read from it so + the gate compares against exactly what the build calibrated to. + period: Simulation period. + relative_tolerance: Maximum absolute relative miss per state. + simulated_caseloads: Injectable precomputed caseloads (tests); when + ``None`` the artifact is simulated. + """ + if not 0.0 < relative_tolerance < 1.0: + raise ValueError("relative_tolerance must be in (0, 1).") + targets = _diagnostics_state_targets( + calibration_diagnostics, target_role="snap_households" + ) + failures: list[str] = [] + missing_targets = sorted(SNAP_STATE_FIPS - set(targets)) + if missing_targets: + failures.append( + "artifact caseload fit: calibration diagnostics carry no " + f"snap_households target for state(s) {missing_targets}; the " + "release predates the FNS caseload surface or dropped it." + ) + if simulated_caseloads is None: + simulated_caseloads = us_snap_simulated_state_caseloads(h5_path, period=period) + rows: list[dict[str, Any]] = [] + for state in sorted(SNAP_STATE_FIPS & set(targets)): + target = targets[state] + modeled = simulated_caseloads.get(state) + relative_error = ( + (modeled - target) / target if modeled is not None and target > 0 else None + ) + within = relative_error is not None and ( + abs(relative_error) <= relative_tolerance + ) + if modeled is None: + failures.append( + f"artifact caseload fit: no simulated caseload for state {state}." + ) + elif not within: + failures.append( + f"artifact caseload fit: state {state} simulated " + f"{modeled:,.0f} vs FNS target {target:,.0f} " + f"({relative_error:+.1%} > ±{relative_tolerance:.0%})." + ) + rows.append( + { + "state_fips": state, + "target": target, + "simulated_taker_households": modeled, + "relative_error": relative_error, + "within_tolerance": within, + } + ) + finite_errors = [ + abs(row["relative_error"]) for row in rows if row["relative_error"] is not None + ] + return GateResult( + name="snap_artifact_caseload_fit", + passed=not failures, + failures=tuple(failures), + details={ + "semantics": ( + "FNS targets are fiscal-year average-monthly household stocks; " + "the simulated measure is annual engine participation " + "(snap > 0) under the shipped take-up flags (populace#419)." + ), + "relative_tolerance": relative_tolerance, + "state_rows": len(rows), + "max_absolute_relative_error": ( + max(finite_errors) if finite_errors else None + ), + "states_outside_tolerance": [ + row["state_fips"] for row in rows if not row["within_tolerance"] + ], + "rows": rows, + }, + ) + + +def assemble_us_snap_release_acceptance( + *, + release_id: str, + build_manifest: dict[str, Any], + snap_state_take_up: dict[str, Any], + calibration_diagnostics: dict[str, Any], + h5_path: Path | str, + participation_validation: dict[str, Any], + target_relative_tolerance: float = 0.10, + simulated_caseloads: Mapping[str, float] | None = None, +) -> dict[str, Any]: + """Assemble the final required-gate verdict and advisory validator.""" + checks = { + "core_release": _gate_payload(us_snap_core_release_gate(build_manifest)), + "state_take_up": _gate_payload(us_snap_state_take_up_gate(snap_state_take_up)), + "state_household_caseload_fit": _gate_payload( + us_snap_state_target_fit_gate( + calibration_diagnostics, + target_role="snap_households", + label="FNS FY2024 average-monthly SNAP households", + relative_tolerance=target_relative_tolerance, + ) + ), + "state_benefit_fit": _gate_payload( + us_snap_state_target_fit_gate( + calibration_diagnostics, + target_role="snap_total", + label="FNS FY2024 SNAP benefits", + relative_tolerance=target_relative_tolerance, + ) + ), + "county_coverage": _gate_payload(us_snap_county_coverage_gate(h5_path)), + "artifact_household_caseload_fit": _gate_payload( + us_snap_artifact_caseload_gate( + h5_path, + calibration_diagnostics, + relative_tolerance=target_relative_tolerance, + simulated_caseloads=simulated_caseloads, + ) + ), + } + required_failures = [ + name + for name, check in checks.items() + if check["required"] and not check["passed"] + ] + saturated_states = snap_state_take_up.get("saturated_states", []) + return { + "schema_version": 1, + "classification": "release_acceptance", + "release_id": release_id, + "passed": not required_failures, + "required_failures": required_failures, + "checks": checks, + "california_saturation": { + "saturated": "06" in saturated_states, + "saturated_states": saturated_states, + "interpretation": ( + "Saturation means the FNS household count meets or exceeds modeled " + "eligible SPM-unit weight; it is an eligibility-undercount diagnostic, " + "not evidence of literal universal participation." + ), + }, + "fy2022_eligible_person_participation": participation_validation, + } diff --git a/packages/populace-build/tests/test_release_target_parity.py b/packages/populace-build/tests/test_release_target_parity.py index 3c47ab48..4bc722c6 100644 --- a/packages/populace-build/tests/test_release_target_parity.py +++ b/packages/populace-build/tests/test_release_target_parity.py @@ -291,6 +291,13 @@ def test_wired_nipa_and_liheap_families_are_compiled(self) -> None: ): assert manifest.by_name[family].status == COMPILED_STATUS + def test_wired_social_security_tips_family_is_compiled(self) -> None: + manifest = load_target_parity_manifest() + assert ( + manifest.by_name["irs_soi.form_w2_social_security_tips"].status + == COMPILED_STATUS + ) + def test_deferred_state_wages_carry_a_fence(self) -> None: manifest = load_target_parity_manifest() state_wages = manifest.by_name["bea_regional.state_wages_salaries"] diff --git a/packages/populace-build/tests/test_us_fiscal_targets.py b/packages/populace-build/tests/test_us_fiscal_targets.py index 406a1ee6..24f5de91 100644 --- a/packages/populace-build/tests/test_us_fiscal_targets.py +++ b/packages/populace-build/tests/test_us_fiscal_targets.py @@ -535,7 +535,7 @@ def test_acs_congressional_district_age_targets_are_opt_in() -> None: def test_zero_support_ledger_facts_are_reviewed_exclusions() -> None: - assert len(US_FISCAL_TARGET_SUPPORT_EXCLUSIONS) == 42 + assert len(US_FISCAL_TARGET_SUPPORT_EXCLUSIONS) == 41 assert all( source_record_id.startswith(("census_stc.", "hhs_acf_tanf.", "irs_soi.")) for source_record_id in US_FISCAL_TARGET_SUPPORT_EXCLUSIONS @@ -1985,6 +1985,7 @@ def test_soi_eitc_child_record_set_metadata_reaches_compiled_target() -> None: specs = {spec.name: spec for spec in registry.specs} spec = specs[source_record_id] + assert source_record_id not in US_FISCAL_TARGET_SUPPORT_EXCLUSIONS assert spec.family == "irs_soi" assert spec.metadata["variable"] == "eitc" assert spec.metadata["agi_lower_bound"] == "25000.0" diff --git a/packages/populace-build/tests/test_us_snap_release_acceptance.py b/packages/populace-build/tests/test_us_snap_release_acceptance.py new file mode 100644 index 00000000..a1eb60a7 --- /dev/null +++ b/packages/populace-build/tests/test_us_snap_release_acceptance.py @@ -0,0 +1,203 @@ +"""SNAP release acceptance tests.""" + +from __future__ import annotations + +import pandas as pd +import pytest + +from populace.build.us_runtime.snap_release_acceptance import ( + SNAP_STATE_FIPS, + assemble_us_snap_release_acceptance, + load_snap_fy2022_participation_reference, + us_snap_artifact_caseload_gate, + us_snap_core_release_gate, + us_snap_county_coverage_gate, + us_snap_state_target_fit_gate, +) + + +def _target_rows(*, relative_error: float = 0.0) -> list[dict]: + rows = [] + for role in ("snap_households", "snap_total"): + for state in sorted(SNAP_STATE_FIPS): + target = 100.0 + rows.append( + { + "name": f"{role}.{state}", + "target_name": f"{role}.{state}", + "target": target, + "final_estimate": target * (1.0 + relative_error), + "metadata": {"target_role": role, "state_fips": state}, + } + ) + return rows + + +def _snap_take_up_diagnostics() -> dict: + return { + "states_without_targets": [], + "saturated_states": ["06"], + "states": [ + { + "state_fips": state, + "target": 100.0, + "eligible_weight": 200.0, + "anchored_eligible_weight": 20.0, + "caseload_weight": 100.0, + "saturated": state == "06", + "anchored_not_taking_up_count": 0, + "max_unit_weight": 1.0, + } + for state in sorted(SNAP_STATE_FIPS) + ], + } + + +def _write_households(path, *, collapse_alaska: bool = False) -> None: + rows = [ + {"state_fips": state, "county_fips": f"{state}001"} + for state in sorted(SNAP_STATE_FIPS) + ] + rows.append( + { + "state_fips": "02", + "county_fips": "02001" if collapse_alaska else "02003", + } + ) + pd.DataFrame(rows).to_hdf(path, key="household", format="table") + + +def test_fy2022_participation_reference_covers_every_state() -> None: + reference = load_snap_fy2022_participation_reference() + assert set(reference["state_rates"]) == SNAP_STATE_FIPS + assert reference["national_rate"] == 0.88 + assert reference["state_rates"]["06"] == 0.81 + assert reference["state_rates"]["11"] == 1.0 + + +def test_core_release_gate_requires_explicit_passing_gate() -> None: + assert us_snap_core_release_gate( + {"gates": {"calibration": {"passed": True}}} + ).passed + failed = us_snap_core_release_gate({"gates": {"calibration": {"passed": False}}}) + assert not failed.passed + + +def test_state_target_fit_gate_requires_all_states_within_tolerance() -> None: + diagnostics = {"targets": _target_rows(relative_error=0.05)} + assert us_snap_state_target_fit_gate( + diagnostics, + target_role="snap_households", + label="households", + ).passed + + diagnostics["targets"][0]["final_estimate"] = 130.0 + failed = us_snap_state_target_fit_gate( + diagnostics, + target_role="snap_households", + label="households", + ) + assert not failed.passed + assert failed.details["states_outside_tolerance"] == ["01"] + + +def test_state_target_fit_gate_rejects_missing_state() -> None: + rows = _target_rows() + rows = [ + row + for row in rows + if not ( + row["metadata"]["target_role"] == "snap_total" + and row["metadata"]["state_fips"] == "56" + ) + ] + failed = us_snap_state_target_fit_gate( + {"targets": rows}, + target_role="snap_total", + label="benefits", + ) + assert not failed.passed + assert any("56" in failure for failure in failed.failures) + + +def test_county_gate_checks_alaska_and_state_prefixes(tmp_path) -> None: + pytest.importorskip("tables") # pandas HDF backend + good = tmp_path / "good.h5" + _write_households(good) + result = us_snap_county_coverage_gate(good) + assert result.passed + assert result.details["alaska_unique_county_fips"] == 2 + + collapsed = tmp_path / "collapsed.h5" + _write_households(collapsed, collapse_alaska=True) + failed = us_snap_county_coverage_gate(collapsed) + assert not failed.passed + assert any("Alaska" in failure for failure in failed.failures) + + +def test_assembled_acceptance_keeps_participation_advisory(tmp_path) -> None: + pytest.importorskip("tables") # pandas HDF backend + h5 = tmp_path / "candidate.h5" + _write_households(h5) + participation = { + "required": False, + "passed_within_advisory_tolerance": False, + "states_outside_advisory_tolerance": ["06"], + } + payload = assemble_us_snap_release_acceptance( + release_id="populace-us-test", + build_manifest={ + "build_id": "populace-us-test", + "gates": {"calibration": {"passed": True}}, + }, + snap_state_take_up=_snap_take_up_diagnostics(), + calibration_diagnostics={"targets": _target_rows()}, + h5_path=h5, + participation_validation=participation, + simulated_caseloads={state: 100.0 for state in sorted(SNAP_STATE_FIPS)}, + ) + assert payload["passed"] + assert payload["required_failures"] == [] + assert payload["california_saturation"]["saturated"] is True + assert payload["fy2022_eligible_person_participation"] is participation + + +def test_artifact_caseload_gate_passes_when_simulated_matches_targets(tmp_path) -> None: + diagnostics = {"targets": _target_rows()} + result = us_snap_artifact_caseload_gate( + tmp_path / "unused.h5", + diagnostics, + simulated_caseloads={state: 100.0 for state in sorted(SNAP_STATE_FIPS)}, + ) + assert result.passed + assert result.details["states_outside_tolerance"] == [] + + +def test_artifact_caseload_gate_fails_when_artifact_diverges_from_calibration( + tmp_path, +) -> None: + """The populace#419 class: calibrated fit green, shipped artifact 5x off.""" + diagnostics = {"targets": _target_rows(relative_error=0.0)} + simulated = {state: 100.0 for state in sorted(SNAP_STATE_FIPS)} + simulated["02"] = 663.0 # AK-style divergence + result = us_snap_artifact_caseload_gate( + tmp_path / "unused.h5", + diagnostics, + simulated_caseloads=simulated, + ) + assert not result.passed + assert result.details["states_outside_tolerance"] == ["02"] + assert any("state 02" in failure for failure in result.failures) + + +def test_artifact_caseload_gate_fails_when_release_lacks_caseload_targets( + tmp_path, +) -> None: + diagnostics = {"targets": []} + result = us_snap_artifact_caseload_gate( + tmp_path / "unused.h5", + diagnostics, + simulated_caseloads={}, + ) + assert not result.passed + assert any("predates" in failure for failure in result.failures) diff --git a/tools/validate_us_snap_release.py b/tools/validate_us_snap_release.py new file mode 100644 index 00000000..95627afc --- /dev/null +++ b/tools/validate_us_snap_release.py @@ -0,0 +1,123 @@ +#!/usr/bin/env python3 +"""Run SNAP-specific post-build acceptance checks on a US release candidate.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +from populace.build.us_runtime.snap_release_acceptance import ( + assemble_us_snap_release_acceptance, + us_snap_participation_validation, +) + +DATASET_FILENAME = "populace_us_2024.h5" + + +def _parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--release-dir", + type=Path, + required=True, + help="Candidate releases/ directory.", + ) + parser.add_argument( + "--h5", + type=Path, + help=( + f"Candidate H5. Defaults to /artifacts/{DATASET_FILENAME}." + ), + ) + parser.add_argument( + "--target-relative-tolerance", + type=float, + default=0.10, + help="Maximum absolute relative error for every state SNAP target.", + ) + parser.add_argument( + "--participation-tolerance-pp", + type=float, + default=10.0, + help=( + "Advisory absolute percentage-point tolerance for FNS FY2022 " + "eligible-person participation rates." + ), + ) + parser.add_argument( + "--output", + type=Path, + help=( + "Output JSON. Defaults to /us_snap_release_acceptance.json." + ), + ) + return parser.parse_args() + + +def _load_json(path: Path) -> dict: + payload = json.loads(path.read_text()) + if not isinstance(payload, dict): + raise ValueError(f"{path} must contain a JSON object.") + return payload + + +def main() -> int: + args = _parse_args() + release_dir = args.release_dir.resolve() + h5_path = ( + args.h5.resolve() + if args.h5 is not None + else release_dir.parent.parent / "artifacts" / DATASET_FILENAME + ) + output = ( + args.output.resolve() + if args.output is not None + else release_dir / "us_snap_release_acceptance.json" + ) + required = { + "build_manifest": release_dir / "build_manifest.json", + "calibration_diagnostics": release_dir / "calibration_diagnostics.json", + "snap_state_take_up": release_dir / "us_snap_state_take_up.json", + "h5": h5_path, + } + missing = [str(path) for path in required.values() if not path.is_file()] + if missing: + raise FileNotFoundError(f"Missing SNAP release acceptance input(s): {missing}") + + build_manifest = _load_json(required["build_manifest"]) + release_id = str(build_manifest.get("build_id") or release_dir.name) + participation = us_snap_participation_validation( + h5_path, + tolerance_percentage_points=args.participation_tolerance_pp, + ) + payload = assemble_us_snap_release_acceptance( + release_id=release_id, + build_manifest=build_manifest, + snap_state_take_up=_load_json(required["snap_state_take_up"]), + calibration_diagnostics=_load_json(required["calibration_diagnostics"]), + h5_path=h5_path, + participation_validation=participation, + target_relative_tolerance=args.target_relative_tolerance, + ) + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text(json.dumps(payload, indent=2, allow_nan=False) + "\n") + print( + json.dumps( + { + "output": str(output), + "release_id": release_id, + "passed": payload["passed"], + "required_failures": payload["required_failures"], + "participation_states_outside_advisory_tolerance": participation[ + "states_outside_advisory_tolerance" + ], + }, + indent=2, + ) + ) + return 0 if payload["passed"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main())