From e518fb86c216c64de75463aadac4e2a5d7a4c4fc Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 27 Sep 2026 23:07:10 -0400 Subject: [PATCH 1/5] Pin the 2025 ASEC and default the US pool to the newest three income years Pin census_cps_2025.h5 (ASEC 2026, income year 2025), built by the archived loader revision that reproduces the pinned 2024 file exactly, at the Hugging Face revision that added it; each processed file now resolves at its own upload revision, so the 2022-2024 URLs do not move. ASEC_DEFAULT_POOL_INCOME_YEARS is the newest three pinned years, (2023, 2024, 2025), derived from the pins. Income year 2022 leaves the default (oldest vintage; 2 of 18 NOW_* recodes, #720) and stays pinned and selectable. The education, PAW_TYP and SPM-role registries pin the ASEC 2026 archive; their loaders and the fetch tool default to the default pool; the BuildP SPM-role enrichment stays bound to its own 2022-2024 years. Co-Authored-By: Claude Opus 5.5 --- .../us-asec-income-2025-default-pool.added.md | 1 + docs/us-asec-census-person-columns.md | 6 + docs/us-asec-source-pins.md | 243 +- experiments/us-asec-2025-source/README.md | 193 ++ .../us-asec-2025-source/census_cps_2025.patch | 318 ++ .../receipts/build_2025.json | 20 + .../receipts/loader_controls_2024.json | 316 ++ .../receipts/measurements_2025.json | 75 + .../receipts/pipeline_smoke_2023_2025.json | 800 +++++ .../receipts/reproduction_2025.json | 20 + .../receipts/schema_diff_2024_2025.json | 2826 +++++++++++++++++ .../scripts/compare_stores.py | 40 + .../scripts/measure_pool.py | 93 + .../us-asec-2025-source/scripts/run_loader.py | 94 + .../scripts/schema_diff.py | 156 + .../us_runtime/asec_census_person_columns.py | 5 +- .../build/us_runtime/asec_sources.py | 88 +- .../us_runtime/education_assistance_source.py | 29 +- .../public_assistance_type_source.py | 12 +- .../build/us_runtime/spm_independence_role.py | 4 +- .../build/us_runtime/spm_role_source.py | 13 +- .../us/test_us_asec_census_person_columns.py | 2 +- .../engine_free/us/test_us_asec_sources.py | 192 +- .../us/test_us_education_assistance_source.py | 2 +- .../engine_free/us/test_us_spine_blindness.py | 14 +- .../us/test_us_spm_independence_role.py | 2 +- .../engine_free/us/test_us_spm_role_source.py | 25 +- tools/build_us_spm_role_enrichment.py | 5 +- tools/fetch_us_asec_sources.py | 39 +- 29 files changed, 5525 insertions(+), 108 deletions(-) create mode 100644 changelog.d/us-asec-income-2025-default-pool.added.md create mode 100644 experiments/us-asec-2025-source/README.md create mode 100644 experiments/us-asec-2025-source/census_cps_2025.patch create mode 100644 experiments/us-asec-2025-source/receipts/build_2025.json create mode 100644 experiments/us-asec-2025-source/receipts/loader_controls_2024.json create mode 100644 experiments/us-asec-2025-source/receipts/measurements_2025.json create mode 100644 experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json create mode 100644 experiments/us-asec-2025-source/receipts/reproduction_2025.json create mode 100644 experiments/us-asec-2025-source/receipts/schema_diff_2024_2025.json create mode 100644 experiments/us-asec-2025-source/scripts/compare_stores.py create mode 100644 experiments/us-asec-2025-source/scripts/measure_pool.py create mode 100644 experiments/us-asec-2025-source/scripts/run_loader.py create mode 100644 experiments/us-asec-2025-source/scripts/schema_diff.py diff --git a/changelog.d/us-asec-income-2025-default-pool.added.md b/changelog.d/us-asec-income-2025-default-pool.added.md new file mode 100644 index 000000000..3ccecff58 --- /dev/null +++ b/changelog.d/us-asec-income-2025-default-pool.added.md @@ -0,0 +1 @@ +Pin the income-year 2025 CPS ASEC input (`census_cps_2025.h5`, built from the ASEC 2026 public-use file) and make the default ASEC pool the newest three pinned income years, 2023-2025. Income year 2022 stays pinned and selectable for byte-reproducible historical builds. Each processed file now resolves at the Hugging Face revision that added it, and the education, PAW_TYP and SPM-role registries pin the ASEC 2026 archive. diff --git a/docs/us-asec-census-person-columns.md b/docs/us-asec-census-person-columns.md index 73592bac7..9eef04b40 100644 --- a/docs/us-asec-census-person-columns.md +++ b/docs/us-asec-census-person-columns.md @@ -14,6 +14,12 @@ The base pools three processed CPS ASEC inputs, pinned in | 2023 | `census_cps_2023.h5` | `cb578173…` | 144,265 | 2 of 18 | | 2024 | `census_cps_2024.h5` | `ec36604c…` | 142,125 | 18 of 18 | +Added 2026-09-27: income year 2025 (`census_cps_2025.h5`, `4c5a3218…`, +134,729 person rows) carries all 18 recodes and every reviewed column below; +in a default-pool (2023-2025) source construction the restore added nothing +for 2025 and verified all 11 reviewed columns equal to `pppub26.csv`. Income +year 2022 has left the default pool (`docs/us-asec-source-pins.md`). + The 2022 and 2023 files were extracted with an older column list. Besides 16 `NOW_*` at-interview coverage recodes they lack `A_EXPRRP`, `PTOTVAL`, `A_ENRLW`, `A_FTPT`, `A_FAMREL`, `A_FAMTYP` and `PECOHAB`; the 2022 file also diff --git a/docs/us-asec-source-pins.md b/docs/us-asec-source-pins.md index 8be681372..3a661bc95 100644 --- a/docs/us-asec-source-pins.md +++ b/docs/us-asec-source-pins.md @@ -1,48 +1,163 @@ # US base build: pinned CPS ASEC inputs -The base stage of `tools/build_us_puf_support_base.py` reads three processed -CPS ASEC files, one per income year 2022, 2023 and 2024, through -`--asec-h5 YEAR=PATH`. Until 18 September 2026 those files existed in one -untracked directory on one build machine, and the build recorded their -digests without comparing them to anything. +The base stage of `tools/build_us_puf_support_base.py` reads processed CPS +ASEC files, one per income year, through `--asec-h5 YEAR=PATH`. Four income +years are pinned: 2022, 2023, 2024 and 2025. A new build pools the newest three +(the default pool). Until 18 September 2026 the 2022-2024 files existed in one +untracked directory on one build machine, and the build recorded their digests +without comparing them to anything. + +Files are named by income year. The survey year is the income year plus one: +`census_cps_2025.h5` is the ASEC 2026 (`asecpub26csv.zip`). ## Where the files live -They are mirrored, unchanged, to the public Hugging Face dataset repository +The files are in the public Hugging Face dataset repository [`policyengine/microcosm-us-sources`](https://huggingface.co/datasets/policyengine/microcosm-us-sources) (CC BY 4.0; they derive from public Census Bureau files). Microcosm addresses -them by upload revision, never by branch, so a later upload cannot move what -a build reads. +each file at the revision of the upload that added it, never by branch. A later +upload therefore cannot move what a build reads, and adding a year does not +change the URL an earlier year resolves to. `microcosm.build.us_runtime.asec_sources` owns the coordinates: -| Income year | File | SHA-256 | Bytes | -|---|---|---|---| -| 2022 | `census_cps_2022.h5` | `7ccca976284bb47815d84460cc4f75a0a65d26d7754ab0a0f417de351b3d474e` | 301,129,278 | -| 2023 | `census_cps_2023.h5` | `cb57817327799f42b741caed5f9be94d04021c2e6809c1ad7bd0686da5428d88` | 299,036,610 | -| 2024 | `census_cps_2024.h5` | `ec36604cb735a660b51b0b2f90be27d803b5878f3464fb30d0eacead59c1260d` | 323,994,739 | +| Income year | File | Revision | SHA-256 | Bytes | +|---|---|---|---|---| +| 2022 | `census_cps_2022.h5` | `78acc83e…` | `7ccca976284bb47815d84460cc4f75a0a65d26d7754ab0a0f417de351b3d474e` | 301,129,278 | +| 2023 | `census_cps_2023.h5` | `78acc83e…` | `cb57817327799f42b741caed5f9be94d04021c2e6809c1ad7bd0686da5428d88` | 299,036,610 | +| 2024 | `census_cps_2024.h5` | `78acc83e…` | `ec36604cb735a660b51b0b2f90be27d803b5878f3464fb30d0eacead59c1260d` | 323,994,739 | +| 2025 | `census_cps_2025.h5` | `efe5e210…` | `4c5a32188b6acfbcfbeb9f5d719d3847873887b72ebdbb19bc402c16e0a64e58` | 304,753,967 | + +The revisions are `78acc83ea8b099a97cb0d658bbed91ea75aae8b0` (2026-09-18: the +2022-2024 files, mirrored unchanged) and +`efe5e2107b252506ea9b868071967cf73cc410bc` (2026-09-27: the 2025 file; the +three earlier files' bytes are unchanged at this revision). + +The 2022-2024 digests are also recorded as hermetic-input evidence in +`ecps_parity_known_gaps.json`, which describes the historical Build J inputs. +`test_us_asec_sources.py` asserts that the evidence and the pins agree for +every year the evidence names. + +## The default pool + +`asec_sources.ASEC_DEFAULT_POOL_INCOME_YEARS` is `(2023, 2024, 2025)`. It is +derived as the newest three pinned income years, not written out, so pinning a +2026 file moves it to `(2024, 2025, 2026)`. + +Income year 2022 left the default on 27 September 2026, for two reasons: + +- It is the oldest vintage. +- Its processed file carries only 2 of the 18 `NOW_*` at-interview coverage + recodes (`NOW_GRP`, `NOW_MRK`; microcosm #720). Its reported-coverage inputs + depend on `asec_census_person_columns` restoring reviewed recodes from the + Census archive. The 2025 file carries all 18. Income year 2023 has the same + gap and stays in the default until a newer year displaces it. + +2022 stays pinned, fetchable and accepted by every registry, so a historical +income-2022..2024 build remains byte-reproducible. Pass its years explicitly. + +A default-pool base build passes the three processed files with their pins +and, for offline runs, the three Census archives: + +```bash +uv run python tools/fetch_us_asec_sources.py # prints the --asec-h5 / --asec-h5-sha256 lines +PYTHONHASHSEED=0 uv run python tools/build_us_puf_support_base.py \ + --asec-h5 2023=… --asec-h5-sha256 2023=cb578173… \ + --asec-h5 2024=… --asec-h5-sha256 2024=ec36604c… \ + --asec-h5 2025=… --asec-h5-sha256 2025=4c5a3218… \ + --asec-education-source 2023=asecpub24csv.zip \ + --asec-education-source 2024=asecpub25csv.zip \ + --asec-education-source 2025=asecpub26csv.zip \ + --asec-2023-weeks-unemployed-source asecpub23csv.zip \ + … +``` + +`--asec-education-source` years without a mapping are fetched from Census and +verified against the same pins. The last flag is explained under "What a +2022-free pool still reads". + +### Every pooled year needs four registries + +A pooled income year needs its processed H5 and its Census survey archive. The +archive is used for four things: the `ED_VAL` and `PAW_TYP` sidecars, the SPM +independence role, and the #720 person-column restore. Each of these is pinned +per income year: -Revision `78acc83ea8b099a97cb0d658bbed91ea75aae8b0`. The same digests are -recorded independently as hermetic-input evidence in -`ecps_parity_known_gaps.json`, and `test_us_asec_sources.py` asserts the two -rosters agree, so there is one source of truth. +| Registry | Module | 2025 value (measured from `asecpub26csv.zip`) | +|---|---|---| +| `ASEC_SOURCE_ARTIFACTS` | `asec_sources` | the H5 above | +| `ASEC_EDUCATION_ASSISTANCE_ARCHIVES` | `education_assistance_source` | archive `fe819d0d…` (139,103,894 bytes); `pppub26.csv` `f7892069…` (259,887,437 bytes, crc32 `477a9e19`); 134,729 rows; 2,574 `ED_VAL` > 0 | +| `ASEC_SPM_ROLE_SOURCES` | `spm_role_source` | 55,401 SPM units | +| `ASEC_PUBLIC_ASSISTANCE_TYPE_AUDIT_PINS` | `public_assistance_type_source` | `PAW_TYP` counts (134,163, 317, 223, 26); 566 PAW-positive; 343 TANF-typed | + +`test_every_pinned_year_is_pinned_in_every_year_keyed_registry` requires the +four key sets to be equal. A year missing from any of them therefore fails in +CI, not partway through a build. Each loader re-derives its values from the +archive and refuses a mismatch. Each one accepted the 2025 archive when run +against the local copy on 27 September 2026. + +The loaders' default `income_years` is the default pool. Every caller in the +build passes the pooled years explicitly, so the default applies only to a +bare call. + +## Income year 2025 + +`census_cps_2025.h5` was built on 27 September 2026 from +`asecpub26csv.zip` (139,103,894 bytes, SHA-256 `fe819d0d…`, Last-Modified 15 +September 2026). It was built by the loader that produced the pinned 2024 file: +the archived US data package's `CensusCPS` at commit `40314a75`. That package +is archived, so it cannot take a PR. Instead, one commit on Max's fork of it +(`9f177ddd`, branch `census-cps-2025`) adds the income-year 2025 URL and stops +requesting `SPM_BBSUBVAL`. The experiment README links both commits. + +The recipe, receipts and schema diff are in +[`experiments/us-asec-2025-source/`](../experiments/us-asec-2025-source/README.md). +In brief: + +- **Loader choice.** Rebuilt from `asecpub25csv.zip`, `40314a75` reproduces + every table of the pinned `census_cps_2024.h5` exactly. The later archived + head `42ed5d45` adds seven person columns, among them `ED_VAL`, which + `fill_asec_education_assistance_source` refuses to overwrite. +- **Schema.** Relative to 2024, only the Census-side changes appear: + - `person` (161 columns) and `spm_unit` (38) lack `SPM_BBSUBVAL`, which the + ASEC 2026 file no longer publishes. It is omitted, not zero-filled. + - `household` (153) drops six broadband and child-care-problem columns and + adds `HTCC_PROBLOSS` and `HXCC_PROBLOSS`. + - No Microcosm reader uses any of these columns. + - All 18 `NOW_*` recodes are present, with codes {1, 2}. +- **Size.** 53,344 households and 134,729 persons, against 55,762 and 142,125 + for 2024. The default pool holds 165,357 households and 421,119 persons; the + old pool held 168,852 and 432,523. +- **Rotation.** 16,601 of the 53,344 households (31.1%; 31.5% weighted by + `HSUP_WGT`) share an `H_IDNUM` with the income-2024 file. The links are real: + the reference person's sex agrees on 98.3% of linked pairs, and their age + advances by 0-2 years on 96.4%, against 49% and 5% for random pairs. 39,927 + persons (29.6%) share a `PERIDNUM`. +- **County identification.** 41.7% of households (45.4% weighted) have a + nonzero `GTCO`, against 41.3% (45.3%) for 2024. The file codes 307 distinct + counties; the income-2024 file codes 280. ## Fetching -`fetch_asec_source(year)` returns a verified local path. It checks the archived -US data repository checkout's storage directory and -`~/.cache/microcosm/asec/` first, to spare a 300 MB transfer on machines that -already hold the file, then downloads the pinned revision through -`huggingface_hub`. Every candidate, the download included, must match the -pinned byte length and SHA-256 before it is returned; a download that does not -raises `ValueError`. Passing `cache_dir` skips the convenience copies and +`fetch_asec_source(year)` returns a verified local path. It first checks the +archived US data repository checkout's storage directory and +`~/.cache/microcosm/asec/`, to spare a 300 MB transfer on machines that already +hold the file. Otherwise it downloads the file at its own pinned revision +through `huggingface_hub`. Every candidate, the download included, must match +the pinned byte length and SHA-256 before it is returned; a download that does +not raises `ValueError`. Passing `cache_dir` skips the convenience copies and downloads into that `huggingface_hub` cache. -`tools/fetch_us_asec_sources.py` resolves all three and prints the builder's +`tools/fetch_us_asec_sources.py` resolves files and prints the builder's arguments, one year per line: +- With no arguments it resolves the default pool. +- `--all-pinned` resolves every pinned year. +- Explicit years resolve exactly those years. + ```bash -uv run python tools/fetch_us_asec_sources.py +uv run python tools/fetch_us_asec_sources.py # 2023 2024 2025 +uv run python tools/fetch_us_asec_sources.py 2022 2023 2024 # historical pool ``` ## Verifying in the build @@ -50,21 +165,65 @@ uv run python tools/fetch_us_asec_sources.py `--asec-h5-sha256 YEAR=SHA256`, passed once per year, makes the builder refuse a `--asec-h5` input whose bytes differ from the declared digest. The check runs in `main()` before any stage, so a wrong input refuses in seconds rather than -surfacing as drift hours later. Every pin is parsed and checked before any -file is read: a pin without `--asec-h5`, a value that is not `YEAR=SHA256`, a -year that is not an integer, a digest that is not 64 hexadecimal characters, a -year named twice, a year no `--asec-h5` mapping provides, and, for a year with -a canonical pin, a declared digest that differs from it (the flag can narrow -nothing and re-pin nothing) all refuse with a message naming the value. Once -any year is pinned, every `--asec-h5` year must be pinned, so a build is either -fully pinned or not pinned at all. Then each pinned file is hashed in year -order; a missing file and a digest mismatch refuse the same way. The pins are -locked into the checkpoint `run_config` (`asec_h5_sha256`, a `{year: digest}` -mapping with the digest lower-cased) so a resume with different pins refuses -and an equivalently spelled pin resumes, and the raw values are forwarded to -every staged child process. +surfacing as drift hours later. + +Every pin is parsed and checked before any file is read. Each of the following +refuses with a message naming the value: + +- a pin without `--asec-h5`; +- a value that is not `YEAR=SHA256`; +- a year that is not an integer; +- a digest that is not 64 hexadecimal characters; +- a year named twice; +- a year that no `--asec-h5` mapping provides; +- for a year with a canonical pin, a declared digest that differs from it. The + flag can narrow nothing and re-pin nothing, and 2025 now has a canonical pin. + +Once any year is pinned, every `--asec-h5` year must be pinned, so a build is +either fully pinned or not pinned at all. Then each pinned file is hashed in +year order; a missing file and a digest mismatch refuse the same way. + +The pins are locked into the checkpoint `run_config` (`asec_h5_sha256`, a +`{year: digest}` mapping with the digest lower-cased). A resume with different +pins refuses, and an equivalently spelled pin resumes. The raw values are +forwarded to every staged child process. The flag is opt-in. Fixture-driven tests build from synthetic -`census_cps_*.h5` files and are not held to the production digests. A -certified from-scratch build passes it for all three years; nothing enforces -that yet, and the US release build rule is where it belongs. +`census_cps_*.h5` files and are not held to the production digests. A certified +from-scratch build passes it for every pooled year; nothing enforces that yet, +and the US release build rule is where it belongs. + +## What a 2022-free pool still reads + +The `source_construction` and `pre_clone_enrichment` stages ran on the real +2023/2024/2025 files, fully pinned, on 27 September 2026 (see the experiment +README). Things that still name income year 2022 or the old pool: + +- **The LKWEEKS sidecar.** The builder loads the income-2022 sidecar + `asecpub23csv.zip` in every mode. It fetches the file when + `--asec-2023-weeks-unemployed-source` is omitted, and records it as the + `LKWEEKS` source pin. For a pool without 2022 the repair fills no rows, + because every other file carries `LKWEEKS`. The raw-stage checkpoint + validator requires a non-empty `LKWEEKS` pin list, so the sidecar cannot + simply be dropped. Pass the local archive when building offline. +- **The generated spec.** `us/spec/sources.yaml` `asec_raw_stage` still pins + the income-2022..2024 raw-stage checkpoint (`51e9fafc…`) and its + `asec_2022/2023/2024` authorities. That checkpoint is the certified lineage + and is unchanged. The first default-pool base build produces the checkpoint + to pin next. +- **The support-spine spec.** `us/support_spine.json` (spec mode, not the + production path) declares two sources, the target year and the year before. +- **Money.** The pooled person tables carry nominal dollars of each income + year, as they did for 2022-2024; no base-build stage restates them by source + year. `target_aging` ages calibration targets, not survey records. The mean + income year of the default pool is 2024, the target year; for the old pool + it was 2023. +- **Prior-year income.** The cohort with no prior-year link inside the pool + moves from 2022 to 2023. +- **Arrival year.** `immigration._ARRIVAL_YEAR_MIDPOINTS` is shared across + vintages. Census re-bins `PEINUSYR`'s top codes every year: + - ASEC 2025: code 28 = 2022-2025. + - ASEC 2026: code 28 = 2022-2023, and a new code 29 = 2024-2026. + + 28 → 2023 stays inside both intervals. 29 → 2024 clamps the 1,121 most + recent arrivals of 2025 to the 2024 target year (zero years in the US). diff --git a/experiments/us-asec-2025-source/README.md b/experiments/us-asec-2025-source/README.md new file mode 100644 index 000000000..92c707580 --- /dev/null +++ b/experiments/us-asec-2025-source/README.md @@ -0,0 +1,193 @@ +# Income year 2025 ASEC source (`census_cps_2025.h5`) + +Built and pinned on 27 September 2026. It is the processed ASEC 2026 file that +`asec_sources.ASEC_SOURCE_ARTIFACTS[2025]` names, and one third of the default +pool (2023, 2024, 2025). See `docs/us-asec-source-pins.md` for how builds use +it. + +| | | +|---|---| +| Census archive | `https://www2.census.gov/programs-surveys/cps/datasets/2026/march/asecpub26csv.zip`, 139,103,894 bytes, SHA-256 `fe819d0d2fc4470c282e76c247b8aa8b6054811619730307e9f2fe6186707d93`, Last-Modified 15 September 2026 | +| Output | `census_cps_2025.h5`, 304,753,967 bytes, SHA-256 `4c5a32188b6acfbcfbeb9f5d719d3847873887b72ebdbb19bc402c16e0a64e58` | +| Hosted at | `policyengine/microcosm-us-sources` revision `efe5e2107b252506ea9b868071967cf73cc410bc` | +| Loader | `PolicyEngine/policyengine-us-data@40314a75` plus [`census_cps_2025.patch`](census_cps_2025.patch) (`MaxGhenis/policyengine-us-data@9f177ddd`, branch `census-cps-2025`) | + +## Why this loader + +The 2022-2024 files were made by the archived US data package's `CensusCPS` +loader, but nobody recorded which commit. The package is archived, so it takes +no PRs. To build 2025 "the same way as 2024", the loader revision that produced +the pinned 2024 file had to be identified first. + +Three revisions rebuilt `census_cps_2024.h5` from a live download of +`asecpub25csv.zip`. The download's SHA-256 was `318845a2…`, equal to the +archive pinned in `education_assistance_source`. Each rebuild was compared with +the pinned file, table by table, with `assert_frame_equal(check_exact=True)` +([`receipts/loader_controls_2024.json`](receipts/loader_controls_2024.json)): + +| Loader | Tables equal to pinned 2024 | Note | +|---|---|---| +| `42ed5d45` (archived head) | family, household, spm_unit, tax_unit | person has 7 extra columns: `PERRP`, `ED_VAL`, `FIN_VAL`, `NOW_GRPFTYP`, `NOW_HIPAID`, `NOW_OWNGRP`, `SRVS_VAL`; the 162 shared columns are equal | +| `40314a75` (#824, "Construct CPS tax units from household records") | all five | chosen | +| `cabe62d2` (the same change before merge) | all five | | + +The archived head was not used, for two reasons. A 2025 file with its extra +columns would differ in shape from 2024. And its `ED_VAL` would make +`fill_asec_education_assistance_source` refuse ("found a preexisting ED_VAL +column with values"). + +Rebuilds reproduce tables, not bytes. All three rebuilds have the pinned file's +exact byte length but different SHA-256s, because HDF5 object metadata is not +byte-deterministic. That is why files are pinned by the uploaded bytes. A second +2025 build from the same archive, with the current +[`scripts/run_loader.py`](scripts/run_loader.py), again gave five identical +tables, the same length and a different digest +([`receipts/reproduction_2025.json`](receipts/reproduction_2025.json)). + +## The loader patch + +`pppub26.csv` no longer carries `SPM_BBSUBVAL`, the SPM unit's broadband +subsidy. The loader requested it for every income year after 2020, so an +unpatched run refuses with `KeyError: Missing required CPS person columns: +SPM_BBSUBVAL`. The patch makes three changes: + +1. Add `CPS_URL_BY_YEAR[2025]` (`…/2026/march/asecpub26csv.zip`) and + `CensusCPS_2025`. +2. Request `SPM_BBSUBVAL` only for income years 2021-2024 + (`SPM_BBSUBVAL_INCOME_YEARS`), the files that publish it. Before, the three + places that built the SPM column list each had their own `<= 2020` test; one + helper, `spm_unit_columns_for_year`, replaces them. +3. Add `tests/unit/datasets/test_census_cps_2025.py` (125 tests, all passing + under the archived package's environment). Two of them are exhaustive over + income years 1990-2100: every URL is the following survey year, and + `SPM_BBSUBVAL` is requested in exactly 2021-2024. Two more run `generate()` + end to end on a synthetic ASEC 2026 archive with the network mocked: 2025 + writes all five tables without `SPM_BBSUBVAL`, and 2024 still refuses + without it. Run against the unpatched loader, the file fails to import. + +The column is omitted rather than zero-filled. Census did not publish it for +2025, and nothing in Microcosm reads it. Microcosm's pooled reader loads only +the `person` and `household` tables, and `SPM_BBSUBVAL` does not match +microunit's `_VAL` pass-through rule. + +## Reproduce + +```bash +git -C policyengine-us-data fetch https://github.com/MaxGhenis/policyengine-us-data census-cps-2025 +git -C policyengine-us-data checkout 9f177ddd6197dd0e296ad05d080865a3e5c2cdbf +cd policyengine-us-data && uv sync --locked --no-dev +PYTHONPATH=$PWD .venv/bin/python /experiments/us-asec-2025-source/scripts/run_loader.py 2025 /tmp/census_cps_2025.h5 +uv run python /experiments/us-asec-2025-source/scripts/compare_stores.py ~/.cache/microcosm/asec/census_cps_2025.h5 /tmp/census_cps_2025.h5 +``` + +Equivalently, apply `census_cps_2025.patch` to `40314a75`. Write to an +explicit path, never the package's `storage/` folder. Other builds hash the +files there. + +## Schema diff against `census_cps_2024.h5` + +Full output: [`receipts/schema_diff_2024_2025.json`](receipts/schema_diff_2024_2025.json) +(from [`scripts/schema_diff.py`](scripts/schema_diff.py)). No shared column +changed dtype. + +| Table | Rows 2024 → 2025 | Columns | Dropped | Added | +|---|---|---|---|---| +| person | 142,125 → 134,729 | 162 → 161 | `SPM_BBSUBVAL` | none | +| household | 55,762 → 53,344 | 157 → 153 | `HBBSUB_MNTH`, `HBBSUB_YN`, `I_HBBSUBMNTH`, `I_HBBSUBYN`, `HECC_PROBTIME`, `HXCC_PROBTIME` | `HTCC_PROBLOSS`, `HXCC_PROBLOSS` | +| family | 62,479 → 59,262 | 85 → 85 | none | none | +| spm_unit | 58,147 → 55,401 | 39 → 38 | `SPM_BBSUBVAL` | none | +| tax_unit | 74,697 → 71,085 | 1 → 1 | none | none | + +- **Renames.** None. Census's ASEC 2026 change list shows `HECC_PROBTIME` + ("time lost from work due to child care problems", 0-999) as removed. It + shows `HTCC_PROBLOSS` ("days of work lost due to child care issues", + 0-260) as added. That is a re-specified question, not a renamed column. The + broadband-subsidy columns go with the end of the subsidy fields. +- **Raw-file changes the processed file never carried.** `pppub26.csv` also + drops the twelve 5-year-migration fields (`M5G*`, `I_M5G1-3`). The loader + never requested them, and no pinned H5 has them. +- **Column order.** Person, family, SPM unit and tax unit keep 2024's order. + The household table's shared columns follow `hhpub26.csv`'s own order, which + differs. Microcosm reads columns by name. +- **Value ranges.** Top codes and maxima move as they do every year: + - 39 person, 24 household, 26 family and 15 SPM-unit numeric columns reach + outside their 2024 range, among them income top codes (`WSAL_VAL`, + `PTOTVAL`, `DIV_VAL`), `SPM_WEIGHT`/`HSUP_WGT` and SPM thresholds. + - `H_YEAR`, `FILEDATE` and `YYYYMM` move by one survey year. +- **Code sets** (columns with at most 60 values). + - `PEINUSYR` gains code 29, Census's yearly re-binning (ASEC 2025: 28 = + 2022-2025; ASEC 2026: 28 = 2022-2023, 29 = 2024-2026). + - `GTCSA` gains 4 combined statistical areas, and `GTCBSASZ` gains a code. + - The rest are household-size and line-number codes present in one year and + not the other. +- **`NOW_*` coverage recodes (#720).** All 18 that the 2024 file carries are + present, with codes {1, 2}. Weighted under-65 "yes" counts, 2024 → 2025, in + millions: + - `NOW_MCAID` 53.8 → 50.9 + - `NOW_CAID` 51.5 → 48.5 + - `NOW_GRP` 166.3 → 165.5 + - `NOW_MRK` 14.6 → 14.3 + - `NOW_PRIV` 194.3 → 193.1 + - `NOW_PUB` 60.7 → 58.0 + - `NOW_COV` 248.3 → 244.4 + + The 2022 and 2023 files carry 2 of them. +- **Geography.** + - `GESTFIPS`: all 51 state codes in both years. + - `GTCO` nonzero: 41.7% of households, 45.4% weighted (2024: 41.3%, 45.3%). + - `GTCBSA` nonzero: 74.2%, 82.0% weighted (2024: 75.4%, 82.5%); 280 + distinct codes (2024: 261). + - `GTCSA` nonzero: 41.5% (2024: 42.3%). +- **`H_IDNUM`.** A 20-character string, unique per household, equal to the + first 20 characters of every member's `PERIDNUM`. + +## Measurements + +[`receipts/measurements_2025.json`](receipts/measurements_2025.json), from +[`scripts/measure_pool.py`](scripts/measure_pool.py). + +- **Households.** 53,344 (134,729 persons; 137.1 million weighted), against + 55,762 (142,125; 135.0 million) for income 2024. The default pool + (2023/2024/2025) holds 165,357 households and 421,119 persons; the old pool + (2022/2023/2024) held 168,852 and 432,523. +- **Rotation overlap with income year 2024.** + - 16,601 households share an `H_IDNUM` with the income-2024 file: 31.1% of + 2025 households, 31.5% weighted by `HSUP_WGT`. 29.8% of 2024 households + reappear. + - The links are the same households. The reference person's sex agrees on + 98.3% of linked pairs, and their age advances by 0-2 years on 96.4%. For + random pairs the figures are 49.0% and 4.6%. + - Most linked pairs carry the same `H_MIS` value in both files, so `H_MIS` + does not predict linkability here. + - 39,927 persons (29.6% of 2025 persons) share a `PERIDNUM`. +- **County identification.** + - 22,267 households have a nonzero `GTCO`: 41.7%, 45.4% weighted. + - They code 307 distinct counties (income 2024: 280). + - Against the ASEC 2026 identified-county list (`cpsmar26.pdf` List 4, 361 + counties), packaged on the location-v1 branch: 287 are coded and listed, + 20 coded but not listed, and 74 listed but not coded. The 20 include the + four old Connecticut county codes, which List 4 no longer names. + - The list is keyed by survey year, and income year 2025 selects survey year + 2026. The file carries every household column the location code reads: + `H_SEQ`, `GESTFIPS`, `GTCO`, `GTCBSA`, `HSUP_WGT`, `H_NUMPER`. + +## Pipeline smoke on the default pool + +[`receipts/pipeline_smoke_2023_2025.json`](receipts/pipeline_smoke_2023_2025.json). +The run used `tools/build_us_puf_support_base.py` with `--stage +source_construction`, then `--stage pre_clone_enrichment`. It passed +`--asec-h5` and `--asec-h5-sha256` for 2023, 2024 and 2025, the three local +Census archives, and `PYTHONHASHSEED=0`. Both stages succeeded: 64 s and 11.0 +GB peak, then 38 s and 13.6 GB. + +- **Pins.** All three `--asec-h5-sha256` pins were verified before either stage + ran; the 2025 pin now has a canonical value to match. +- **#720 restore.** It joined every person one-to-one: 144,265, 142,125 and + 134,729. For 2023 it added the 11 reviewed columns. For 2024 and 2025 every + reviewed column the H5 carries equals the Census member. +- **Pool shares.** Equal thirds, anchored to the 2024 weighted population. +- **Signals.** Every pre-clone signal passed: + - SPM independence role share 0.628; minor role share 0.018; composition + PASS. + - Relationship, eligibility, housing, Medicare take-up, pregnancy and WIC + claim. diff --git a/experiments/us-asec-2025-source/census_cps_2025.patch b/experiments/us-asec-2025-source/census_cps_2025.patch new file mode 100644 index 000000000..93601d5d3 --- /dev/null +++ b/experiments/us-asec-2025-source/census_cps_2025.patch @@ -0,0 +1,318 @@ +From 9f177ddd6197dd0e296ad05d080865a3e5c2cdbf Mon Sep 17 00:00:00 2001 +From: Max Ghenis +Date: Sun, 27 Sep 2026 21:24:03 -0400 +Subject: [PATCH] Load income year 2025 from the ASEC 2026 public-use archive + +Add CensusCPS_2025 and its asecpub26csv.zip URL. The ASEC 2026 person file +no longer carries SPM_BBSUBVAL, which the loader required for every income +year after 2020; request it only for income years 2021-2024, the files that +publish it, through spm_unit_columns_for_year(). + +Based on 40314a75 (#824), the loader revision that reproduces every table +of the pinned census_cps_2024.h5 exactly. + +Co-Authored-By: Claude Opus 5.5 +--- + .../datasets/cps/census_cps.py | 40 ++-- + tests/unit/datasets/test_census_cps_2025.py | 207 ++++++++++++++++++ + 2 files changed, 232 insertions(+), 15 deletions(-) + create mode 100644 tests/unit/datasets/test_census_cps_2025.py + +diff --git a/policyengine_us_data/datasets/cps/census_cps.py b/policyengine_us_data/datasets/cps/census_cps.py +index a62aeeed..5ad4cdef 100644 +--- a/policyengine_us_data/datasets/cps/census_cps.py ++++ b/policyengine_us_data/datasets/cps/census_cps.py +@@ -50,6 +50,19 @@ def _resolve_person_usecols( + return [column for column in requested_columns if column in available_columns] + + ++# SPM_BBSUBVAL (the SPM unit's broadband subsidy value) is published only in ++# the ASEC 2022-2025 public-use person files, i.e. income years 2021-2024. ++# The ASEC 2026 file (income year 2025) no longer carries it. ++SPM_BBSUBVAL_INCOME_YEARS = range(2021, 2025) ++ ++ ++def spm_unit_columns_for_year(time_period: int) -> list[str]: ++ """SPM unit columns the Census person file carries for an income year.""" ++ if int(time_period) in SPM_BBSUBVAL_INCOME_YEARS: ++ return list(SPM_UNIT_COLUMNS) ++ return [col for col in SPM_UNIT_COLUMNS if col != "SPM_BBSUBVAL"] ++ ++ + def _fill_missing_optional_person_columns(person: pd.DataFrame) -> pd.DataFrame: + for column in OPTIONAL_PERSON_COLUMNS: + if column not in person.columns: +@@ -72,11 +85,7 @@ class CensusCPS(Dataset): + + url = self._cps_download_url + +- spm_unit_columns = SPM_UNIT_COLUMNS +- if self.time_period <= 2020: +- spm_unit_columns = [ +- col for col in spm_unit_columns if col != "SPM_BBSUBVAL" +- ] ++ spm_unit_columns = spm_unit_columns_for_year(self.time_period) + + response = requests.get(url, stream=True) + total_size_in_bytes = int(response.headers.get("content-length", 200e6)) +@@ -98,11 +107,7 @@ class CensusCPS(Dataset): + progress_bar.total = content_length_actual + progress_bar.close() + zipfile = ZipFile(file) +- spm_unit_columns = SPM_UNIT_COLUMNS +- if self.time_period <= 2020: +- spm_unit_columns = [ +- col for col in spm_unit_columns if col != "SPM_BBSUBVAL" +- ] ++ spm_unit_columns = spm_unit_columns_for_year(self.time_period) + with pd.HDFStore(self.file_path, mode="w") as storage: + file_year = int(self.time_period) + 1 + file_year_code = str(file_year)[-2:] +@@ -165,14 +170,18 @@ class CensusCPS(Dataset): + def _create_spm_unit_table( + self, person: pd.DataFrame, time_period: int + ) -> pd.DataFrame: +- spm_unit_columns = SPM_UNIT_COLUMNS +- if time_period <= 2020: +- spm_unit_columns = [ +- col for col in spm_unit_columns if col != "SPM_BBSUBVAL" +- ] ++ spm_unit_columns = spm_unit_columns_for_year(time_period) + return person[spm_unit_columns].groupby(person.SPM_ID).first() + + ++class CensusCPS_2025(CensusCPS): ++ time_period = 2025 ++ label = "Census CPS (2025)" ++ name = "census_cps_2025" ++ file_path = STORAGE_FOLDER / "census_cps_2025.h5" ++ data_format = Dataset.TABLES ++ ++ + class CensusCPS_2024(CensusCPS): + time_period = 2024 + label = "Census CPS (2024)" +@@ -237,6 +246,7 @@ CPS_URL_BY_YEAR = { + 2022: "https://www2.census.gov/programs-surveys/cps/datasets/2023/march/asecpub23csv.zip", + 2023: "https://www2.census.gov/programs-surveys/cps/datasets/2024/march/asecpub24csv.zip", + 2024: "https://www2.census.gov/programs-surveys/cps/datasets/2025/march/asecpub25csv.zip", ++ 2025: "https://www2.census.gov/programs-surveys/cps/datasets/2026/march/asecpub26csv.zip", + } + + +diff --git a/tests/unit/datasets/test_census_cps_2025.py b/tests/unit/datasets/test_census_cps_2025.py +new file mode 100644 +index 00000000..28d7e315 +--- /dev/null ++++ b/tests/unit/datasets/test_census_cps_2025.py +@@ -0,0 +1,207 @@ ++"""The income-year 2025 (ASEC 2026) Census CPS loader. ++ ++The ASEC 2026 public-use person file drops SPM_BBSUBVAL, which the loader ++required for every income year after 2020. These tests pin the URL map, the ++year-keyed SPM column list, and an end-to-end ``generate()`` over a synthetic ++ASEC 2026 archive with no network access. ++""" ++ ++import io ++import re ++import zipfile ++ ++import pandas as pd ++import pytest ++ ++from policyengine_us_data.datasets.cps import census_cps ++from policyengine_us_data.datasets.cps.census_cps import ( ++ CPS_URL_BY_YEAR, ++ PERSON_COLUMNS, ++ SPM_BBSUBVAL_INCOME_YEARS, ++ SPM_UNIT_COLUMNS, ++ TAX_UNIT_COLUMNS, ++ CensusCPS_2024, ++ CensusCPS_2025, ++ _resolve_person_usecols, ++ spm_unit_columns_for_year, ++) ++ ++URL_PATTERN = re.compile( ++ r"^https://www2\.census\.gov/programs-surveys/cps/datasets/" ++ r"(?P\d{4})/march/asecpub(?P\d{2})csv\.zip$" ++) ++ ++ ++@pytest.mark.parametrize("income_year", sorted(CPS_URL_BY_YEAR)) ++def test_every_url_is_the_following_survey_year(income_year): ++ match = URL_PATTERN.match(CPS_URL_BY_YEAR[income_year]) ++ assert match is not None ++ assert int(match["survey"]) == income_year + 1 ++ assert match["code"] == str(income_year + 1)[-2:] ++ ++ ++def test_income_year_2025_reads_the_asec_2026_archive(): ++ assert CPS_URL_BY_YEAR[2025] == ( ++ "https://www2.census.gov/programs-surveys/cps/datasets/2026/march/" ++ "asecpub26csv.zip" ++ ) ++ dataset = object.__new__(CensusCPS_2025) ++ assert dataset.time_period == 2025 ++ assert dataset.name == "census_cps_2025" ++ assert dataset.file_path.name == "census_cps_2025.h5" ++ assert dataset._cps_download_url == CPS_URL_BY_YEAR[2025] ++ ++ ++@pytest.mark.parametrize("income_year", range(1990, 2101)) ++def test_spm_columns_carry_bbsubval_exactly_in_income_years_2021_to_2024( ++ income_year, ++): ++ columns = spm_unit_columns_for_year(income_year) ++ assert ("SPM_BBSUBVAL" in columns) == (2021 <= income_year <= 2024) ++ # Every other SPM column is always requested, in the original order. ++ assert [c for c in SPM_UNIT_COLUMNS if c != "SPM_BBSUBVAL"] == [ ++ c for c in columns if c != "SPM_BBSUBVAL" ++ ] ++ assert columns == [c for c in SPM_UNIT_COLUMNS if c in columns] ++ ++ ++def test_spm_bbsubval_window_matches_the_published_files(): ++ # ASEC 2022-2025 (income 2021-2024) carry SPM_BBSUBVAL; ASEC 2021 and 2026 ++ # do not. Earlier loader versions encoded the lower edge as `<= 2020`. ++ assert list(SPM_BBSUBVAL_INCOME_YEARS) == [2021, 2022, 2023, 2024] ++ assert "SPM_BBSUBVAL" in SPM_UNIT_COLUMNS ++ ++ ++def test_spm_columns_for_a_year_do_not_alias_the_module_list(): ++ columns = spm_unit_columns_for_year(2024) ++ columns.append("SENTINEL") ++ assert "SENTINEL" not in SPM_UNIT_COLUMNS ++ ++ ++def _asec_2026_person_header() -> list[str]: ++ return [ ++ c ++ for c in dict.fromkeys(PERSON_COLUMNS + SPM_UNIT_COLUMNS + TAX_UNIT_COLUMNS) ++ if c != "SPM_BBSUBVAL" ++ ] ++ ++ ++def test_an_asec_2026_header_resolves_for_2025_and_refuses_for_2024(): ++ header = _asec_2026_person_header() ++ usecols = _resolve_person_usecols(header, spm_unit_columns_for_year(2025)) ++ assert "SPM_BBSUBVAL" not in usecols ++ assert "SPM_ID" in usecols ++ with pytest.raises(KeyError, match="SPM_BBSUBVAL"): ++ _resolve_person_usecols(header, spm_unit_columns_for_year(2024)) ++ ++ ++def _synthetic_asec_2026_archive() -> bytes: ++ header = _asec_2026_person_header() ++ rows = [] ++ for line, (age, wages, expr) in enumerate( ++ ((40, 50_000, 1), (38, 20_000, 3), (10, 0, 5)), start=1 ++ ): ++ row = dict.fromkeys(header, 0) ++ row.update( ++ PH_SEQ=1, ++ PF_SEQ=1, ++ P_SEQ=line, ++ A_LINENO=line, ++ A_AGE=age, ++ A_SEX=1 + line % 2, ++ A_MARITL=1 if age >= 18 else 7, ++ A_SPOUSE=(2 if line == 1 else 1) if age >= 18 else 0, ++ A_EXPRRP=expr, ++ PEPAR1=1 if age < 18 else -1, ++ PEPAR2=2 if age < 18 else -1, ++ PECOHAB=-1, ++ WSAL_VAL=wages, ++ PTOTVAL=wages, ++ A_FNLWGT=150_000, ++ TAX_ID=7, ++ SPM_ID=9, ++ SPM_WEIGHT=150_000, ++ SPM_NUMPER=3, ++ ) ++ rows.append(row) ++ person = pd.DataFrame(rows, columns=header) ++ family = pd.DataFrame({"FH_SEQ": [1, 2], "FFPOS": [1, 1], "FTOTVAL": [70_000, 0]}) ++ household = pd.DataFrame( ++ {"H_SEQ": [1, 2], "H_IDNUM": ["000000000000001", "000000000000002"]} ++ ) ++ buffer = io.BytesIO() ++ with zipfile.ZipFile(buffer, "w") as archive: ++ archive.writestr("pppub26.csv", person.to_csv(index=False)) ++ archive.writestr("ffpub26.csv", family.to_csv(index=False)) ++ archive.writestr("hhpub26.csv", household.to_csv(index=False)) ++ return buffer.getvalue() ++ ++ ++class _FakeResponse: ++ status_code = 200 ++ ++ def __init__(self, payload: bytes): ++ self._payload = payload ++ self.headers = {"content-length": str(len(payload))} ++ ++ def iter_content(self, chunk_size): ++ for start in range(0, len(self._payload), chunk_size): ++ yield self._payload[start : start + chunk_size] ++ ++ ++def test_generate_writes_an_income_year_2025_store_without_spm_bbsubval( ++ tmp_path, monkeypatch ++): ++ payload = _synthetic_asec_2026_archive() ++ requested = [] ++ ++ def fake_get(url, stream): ++ requested.append(url) ++ return _FakeResponse(payload) ++ ++ monkeypatch.setattr(census_cps.requests, "get", fake_get) ++ dataset = object.__new__(CensusCPS_2025) ++ dataset.file_path = tmp_path / "census_cps_2025.h5" ++ ++ dataset.generate() ++ ++ assert requested == [CPS_URL_BY_YEAR[2025]] ++ with pd.HDFStore(dataset.file_path, mode="r") as store: ++ assert sorted(store.keys()) == [ ++ "/family", ++ "/household", ++ "/person", ++ "/spm_unit", ++ "/tax_unit", ++ ] ++ person = store["person"] ++ spm_unit = store["spm_unit"] ++ household = store["household"] ++ family = store["family"] ++ assert "SPM_BBSUBVAL" not in person.columns ++ assert list(spm_unit.columns) == spm_unit_columns_for_year(2025) ++ assert spm_unit.index.tolist() == [9] ++ assert person["CENSUS_TAX_ID"].tolist() == [7, 7, 7] ++ # Only households and families that have persons are kept. ++ assert household["H_SEQ"].tolist() == [1] ++ assert family["FH_SEQ"].tolist() == [1] ++ ++ ++def test_generate_for_2024_still_requires_spm_bbsubval(tmp_path, monkeypatch): ++ payload = _synthetic_asec_2026_archive() ++ with zipfile.ZipFile(io.BytesIO(payload)) as source: ++ members = {name: source.read(name) for name in source.namelist()} ++ buffer = io.BytesIO() ++ with zipfile.ZipFile(buffer, "w") as archive: ++ for name, data in members.items(): ++ archive.writestr(name.replace("26", "25"), data) ++ monkeypatch.setattr( ++ census_cps.requests, ++ "get", ++ lambda url, stream: _FakeResponse(buffer.getvalue()), ++ ) ++ dataset = object.__new__(CensusCPS_2024) ++ dataset.file_path = tmp_path / "census_cps_2024.h5" ++ ++ with pytest.raises(KeyError, match="SPM_BBSUBVAL"): ++ dataset.generate() +-- +2.55.0 + diff --git a/experiments/us-asec-2025-source/receipts/build_2025.json b/experiments/us-asec-2025-source/receipts/build_2025.json new file mode 100644 index 000000000..ac0986c72 --- /dev/null +++ b/experiments/us-asec-2025-source/receipts/build_2025.json @@ -0,0 +1,20 @@ +{ + "income_year": 2025, + "output": "/build/y2025/census_cps_2025.h5", + "output_sha256": "4c5a32188b6acfbcfbeb9f5d719d3847873887b72ebdbb19bc402c16e0a64e58", + "output_bytes": 304753967, + "download": { + "url": "https://www2.census.gov/programs-surveys/cps/datasets/2026/march/asecpub26csv.zip", + "sha256": "fe819d0d2fc4470c282e76c247b8aa8b6054811619730307e9f2fe6186707d93", + "bytes": 139103894, + "last_modified": "Tue, 15 Sep 2026 10:55:35 GMT", + "etag": null + }, + "loader_commit": "9f177ddd6197dd0e296ad05d080865a3e5c2cdbf", + "loader_dirty": "", + "python": "3.14.7", + "pandas": "2.3.3", + "tables": "3.11.1", + "numpy": "2.4.3", + "policyengine_core": null +} \ No newline at end of file diff --git a/experiments/us-asec-2025-source/receipts/loader_controls_2024.json b/experiments/us-asec-2025-source/receipts/loader_controls_2024.json new file mode 100644 index 000000000..565100f7c --- /dev/null +++ b/experiments/us-asec-2025-source/receipts/loader_controls_2024.json @@ -0,0 +1,316 @@ +{ + "purpose": "Which archived CensusCPS loader revision reproduces the pinned census_cps_2024.h5? Each rebuilt asecpub25csv.zip (downloaded live, sha256 318845a2...) and was compared table by table (pandas assert_frame_equal, check_exact=True) with the pinned file.", + "controls": [ + { + "loader": "42ed5d45 (archived head)", + "run": { + "income_year": 2024, + "output": "/build/control_2024/census_cps_2024.h5", + "output_sha256": "f748f03f0e76df622fb27df7ac4df1133e294087f5e22a0e09bd34eb032df903", + "output_bytes": 331953991, + "download": { + "url": "https://www2.census.gov/programs-surveys/cps/datasets/2025/march/asecpub25csv.zip", + "sha256": "318845a2b5e0034eb2973898de1738f4df0025727de38499e7669cb9c0deef0b", + "bytes": 147271429, + "last_modified": "Tue, 09 Sep 2025 13:26:40 GMT", + "etag": null + }, + "loader_commit": "42ed5d45c56df80d754fbe24cce21cfeb8d05cbe", + "loader_dirty": "", + "python": "3.14.7", + "pandas": "2.3.3", + "tables": "3.11.1", + "numpy": "2.4.3", + "policyengine_core": null + }, + "pinned_file": "census_cps_2024.h5 (sha256 ec36604c...)", + "tables_vs_pinned": { + "/family": { + "frame_equal_exact": true, + "shape": [ + [ + 62479, + 85 + ], + [ + 62479, + 85 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/household": { + "frame_equal_exact": true, + "shape": [ + [ + 55762, + 157 + ], + [ + 55762, + 157 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/person": { + "frame_equal_exact": false, + "shape": [ + [ + 142125, + 162 + ], + [ + 142125, + 169 + ] + ], + "only_rebuilt": [ + "ED_VAL", + "FIN_VAL", + "NOW_GRPFTYP", + "NOW_HIPAID", + "NOW_OWNGRP", + "PERRP", + "SRVS_VAL" + ], + "only_pinned": [] + }, + "/spm_unit": { + "frame_equal_exact": true, + "shape": [ + [ + 58147, + 39 + ], + [ + 58147, + 39 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/tax_unit": { + "frame_equal_exact": true, + "shape": [ + [ + 74697, + 1 + ], + [ + 74697, + 1 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + } + } + }, + { + "loader": "40314a75 (#824)", + "run": { + "income_year": 2024, + "output": "/build/loader_40314a75/census_cps_2024.h5", + "output_sha256": "7c96fc77c06222522a83e1ba178c4a4bc62b775ddadc870ad8e4ed4dafb0ad3d", + "output_bytes": 323994739, + "download": { + "url": "https://www2.census.gov/programs-surveys/cps/datasets/2025/march/asecpub25csv.zip", + "sha256": "318845a2b5e0034eb2973898de1738f4df0025727de38499e7669cb9c0deef0b", + "bytes": 147271429, + "last_modified": "Tue, 09 Sep 2025 13:26:40 GMT", + "etag": null + }, + "loader_commit": "40314a759aafa9cf71b2073eab532e1059b5034f", + "loader_dirty": "", + "python": "3.14.7", + "pandas": "2.3.3", + "tables": "3.11.1", + "numpy": "2.4.3", + "policyengine_core": null + }, + "pinned_file": "census_cps_2024.h5 (sha256 ec36604c...)", + "tables_vs_pinned": { + "/family": { + "frame_equal_exact": true, + "shape": [ + [ + 62479, + 85 + ], + [ + 62479, + 85 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/household": { + "frame_equal_exact": true, + "shape": [ + [ + 55762, + 157 + ], + [ + 55762, + 157 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/person": { + "frame_equal_exact": true, + "shape": [ + [ + 142125, + 162 + ], + [ + 142125, + 162 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/spm_unit": { + "frame_equal_exact": true, + "shape": [ + [ + 58147, + 39 + ], + [ + 58147, + 39 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/tax_unit": { + "frame_equal_exact": true, + "shape": [ + [ + 74697, + 1 + ], + [ + 74697, + 1 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + } + } + }, + { + "loader": "cabe62d2", + "run": { + "income_year": 2024, + "output": "/build/loader_cabe62d2/census_cps_2024.h5", + "output_sha256": "32b8c958468bd27054428377403d191c90d6c01f3eacdfc682b2f245ac491ced", + "output_bytes": 323994739, + "download": { + "url": "https://www2.census.gov/programs-surveys/cps/datasets/2025/march/asecpub25csv.zip", + "sha256": "318845a2b5e0034eb2973898de1738f4df0025727de38499e7669cb9c0deef0b", + "bytes": 147271429, + "last_modified": "Tue, 09 Sep 2025 13:26:40 GMT", + "etag": null + }, + "loader_commit": "cabe62d255a02091cfb8f2dd43cac9a975065f1d", + "loader_dirty": "", + "python": "3.14.7", + "pandas": "2.3.3", + "tables": "3.11.1", + "numpy": "2.4.3", + "policyengine_core": null + }, + "pinned_file": "census_cps_2024.h5 (sha256 ec36604c...)", + "tables_vs_pinned": { + "/family": { + "frame_equal_exact": true, + "shape": [ + [ + 62479, + 85 + ], + [ + 62479, + 85 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/household": { + "frame_equal_exact": true, + "shape": [ + [ + 55762, + 157 + ], + [ + 55762, + 157 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/person": { + "frame_equal_exact": true, + "shape": [ + [ + 142125, + 162 + ], + [ + 142125, + 162 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/spm_unit": { + "frame_equal_exact": true, + "shape": [ + [ + 58147, + 39 + ], + [ + 58147, + 39 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + }, + "/tax_unit": { + "frame_equal_exact": true, + "shape": [ + [ + 74697, + 1 + ], + [ + 74697, + 1 + ] + ], + "only_rebuilt": [], + "only_pinned": [] + } + } + } + ] +} \ No newline at end of file diff --git a/experiments/us-asec-2025-source/receipts/measurements_2025.json b/experiments/us-asec-2025-source/receipts/measurements_2025.json new file mode 100644 index 000000000..2a754f099 --- /dev/null +++ b/experiments/us-asec-2025-source/receipts/measurements_2025.json @@ -0,0 +1,75 @@ +{ + "income_2024": { + "survey_year": 2025, + "households": 55762, + "persons": 142125, + "weighted_households_millions": 134.964, + "county_identified_households": 23021, + "county_identified_share": 0.41284387217101254, + "county_identified_share_weighted": 0.4527315190926383, + "distinct_coded_counties": 280, + "census_list4_counties": 277, + "coded_and_listed": 271, + "coded_not_listed": 9, + "listed_not_coded": 6, + "coded_not_listed_fips": [ + 6017, + 6061, + 6077, + 6113, + 25009, + 25021, + 48439, + 53033, + 53061 + ] + }, + "income_2025": { + "survey_year": 2026, + "households": 53344, + "persons": 134729, + "weighted_households_millions": 137.124, + "county_identified_households": 22267, + "county_identified_share": 0.4174227654469106, + "county_identified_share_weighted": 0.45391665873253445, + "distinct_coded_counties": 307, + "census_list4_counties": 361, + "coded_and_listed": 287, + "coded_not_listed": 20, + "listed_not_coded": 74, + "coded_not_listed_fips": [ + 1079, + 1103, + 1107, + 1125, + 6101, + 6115, + 9001, + 9005, + 9009, + 9015, + 12119, + 22105, + 26027, + 37019, + 37129, + 37141, + 42075, + 45085, + 54019, + 54081 + ] + }, + "rotation_income_2025_vs_2024": { + "shared_H_IDNUM": 16601, + "share_of_2025_households": 0.31120650869826033, + "share_of_2025_households_weighted": 0.31470292867802857, + "share_of_2024_households": 0.2977117033104982, + "linked_reference_person_sex_agrees": 0.9831335461719174, + "linked_reference_person_age_plus_0_to_2": 0.9637371242696223, + "random_pairs_sex_agrees": 0.49025, + "random_pairs_age_plus_0_to_2": 0.0463, + "shared_PERIDNUM": 39927, + "share_of_2025_persons": 0.29635045164738105 + } +} \ No newline at end of file diff --git a/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json b/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json new file mode 100644 index 000000000..f2a4d3c53 --- /dev/null +++ b/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json @@ -0,0 +1,800 @@ +{ + "purpose": "source_construction and pre_clone_enrichment of tools/build_us_puf_support_base.py on the default pool, fully pinned, 2026-09-27", + "run_config": { + "asec_h5": [ + "2023=~/PolicyEngine/policyengine-us-data/policyengine_us_data/storage/census_cps_2023.h5", + "2024=~/PolicyEngine/policyengine-us-data/policyengine_us_data/storage/census_cps_2024.h5", + "2025=~/.cache/microcosm/asec/census_cps_2025.h5" + ], + "asec_h5_sha256": { + "2023": "cb57817327799f42b741caed5f9be94d04021c2e6809c1ad7bd0686da5428d88", + "2024": "ec36604cb735a660b51b0b2f90be27d803b5878f3464fb30d0eacead59c1260d", + "2025": "4c5a32188b6acfbcfbeb9f5d719d3847873887b72ebdbb19bc402c16e0a64e58" + }, + "asec_education_source": { + "2023": "~/PolicyEngine/_buildm-runtime/inputs/asec_education/asecpub24csv.zip", + "2024": "~/PolicyEngine/_buildm-runtime/inputs/asec_education/asecpub25csv.zip", + "2025": "~/PolicyEngine/_buildm-runtime/inputs/asec_education/asecpub26csv.zip" + }, + "target_year": 2024, + "acs_h5": "~/PolicyEngine/policyengine-us-data/policyengine_us_data/storage/acs_2022.h5", + "seed": 0 + }, + "stage_profile": { + "pre_clone_enrichment": { + "entry_rss_bytes": 1970290688, + "error": null, + "exit_rss_bytes": 10873683968, + "peak_rss_bytes": 13574078464, + "rss_sample_count": 129, + "sampling_error": null, + "stage": "pre_clone_enrichment", + "status": "succeeded", + "wall_seconds": 38.22470487502869 + }, + "source_construction": { + "entry_rss_bytes": 1971683328, + "error": null, + "exit_rss_bytes": 9930113024, + "peak_rss_bytes": 10964992000, + "rss_sample_count": 202, + "sampling_error": null, + "stage": "source_construction", + "status": "succeeded", + "wall_seconds": 63.66323954198742 + } + }, + "row_counts": [ + { + "rows": 421119, + "table": "person" + }, + { + "rows": 165357, + "table": "household" + }, + { + "rows": 221390, + "table": "tax_unit" + }, + { + "rows": 172259, + "table": "spm_unit" + }, + { + "rows": 184978, + "table": "family" + }, + { + "rows": 336355, + "table": "marital_unit" + } + ], + "pooled_sources": [ + { + "census_person_columns": { + "archive_sha256": "cdb39cdac34bef99dd0940ab28e306f692404c2eea44d85dfd634214872a0a09", + "columns_added": [ + "NOW_MCAID", + "NOW_NONM", + "NOW_CHAMPVA", + "NOW_MIL", + "NOW_VACARE", + "NOW_OTHMT", + "NOW_IHSFLG", + "A_EXPRRP", + "PTOTVAL", + "A_ENRLW", + "A_FTPT" + ], + "columns_verified_equal": [], + "h5_person_rows": 144265, + "identity_columns_verified_equal": [ + "PH_SEQ", + "P_SEQ", + "A_LINENO", + "A_AGE" + ], + "income_year": 2023, + "join_key": "PERIDNUM (exact 22-digit string), one-to-one and total", + "joined_person_rows": 144265, + "member": "pppub24.csv", + "member_person_rows": 144265, + "member_sha256": "21a2b9e0e4b08534563578a45acad77868af4ae9a7d46f23776b707d4a559aa7", + "member_size_bytes": 277250415, + "member_values": { + "A_ENRLW": { + "0": 72991, + "1": 11620, + "2": 59654 + }, + "A_EXPRRP": { + "1": 37875, + "10": 2763, + "11": 147, + "12": 1752, + "13": 4387, + "14": 968, + "2": 18376, + "3": 11652, + "4": 16147, + "5": 42756, + "7": 2661, + "8": 2854, + "9": 1927 + }, + "A_FTPT": { + "0": 132645, + "1": 10174, + "2": 1446 + }, + "NOW_CHAMPVA": { + "1": 335, + "2": 143930 + }, + "NOW_IHSFLG": { + "1": 651, + "2": 143614 + }, + "NOW_MCAID": { + "1": 26367, + "2": 117898 + }, + "NOW_MIL": { + "1": 4318, + "2": 139947 + }, + "NOW_NONM": { + "1": 8435, + "2": 135830 + }, + "NOW_OTHMT": { + "1": 331, + "2": 143934 + }, + "NOW_VACARE": { + "1": 1178, + "2": 143087 + }, + "PTOTVAL": { + "negative_rows": 59, + "nonzero_rows": 102389 + } + }, + "official_archive_url": "https://www2.census.gov/programs-surveys/cps/datasets/2024/march/asecpub24csv.zip", + "operation": "exact_source_join", + "reviewed_columns": [ + "NOW_MCAID", + "NOW_NONM", + "NOW_CHAMPVA", + "NOW_MIL", + "NOW_VACARE", + "NOW_OTHMT", + "NOW_IHSFLG", + "A_EXPRRP", + "PTOTVAL", + "A_ENRLW", + "A_FTPT" + ], + "source_form": "archive", + "source_path": "~/PolicyEngine/_buildm-runtime/inputs/asec_education/asecpub24csv.zip", + "survey_year": 2024 + }, + "raw_household_rows": 56251, + "raw_person_population": 320890854.43000007, + "raw_person_rows": 144265, + "relationship_recode_source": "source:A_EXPRRP", + "scale": 0.3385911071839787, + "share": 0.3333333333333333, + "source_file": "census_cps_2023.h5", + "weighted_person_population": 108650789.68666664, + "year": 2023 + }, + { + "census_person_columns": { + "archive_sha256": "318845a2b5e0034eb2973898de1738f4df0025727de38499e7669cb9c0deef0b", + "columns_added": [], + "columns_verified_equal": [ + "NOW_MCAID", + "NOW_NONM", + "NOW_CHAMPVA", + "NOW_MIL", + "NOW_VACARE", + "NOW_OTHMT", + "NOW_IHSFLG", + "A_EXPRRP", + "PTOTVAL", + "A_ENRLW", + "A_FTPT" + ], + "h5_person_rows": 142125, + "identity_columns_verified_equal": [ + "PH_SEQ", + "P_SEQ", + "A_LINENO", + "A_AGE" + ], + "income_year": 2024, + "join_key": "PERIDNUM (exact 22-digit string), one-to-one and total", + "joined_person_rows": 142125, + "member": "pppub25.csv", + "member_person_rows": 142125, + "member_sha256": "06921fe83fc66c907e6c7b86b82255dc70458ee7d76258fc48297cb34f0c06b5", + "member_size_bytes": 277882549, + "member_values": { + "A_ENRLW": { + "0": 72079, + "1": 11445, + "2": 58601 + }, + "A_EXPRRP": { + "1": 37382, + "10": 2659, + "11": 134, + "12": 1734, + "13": 4257, + "14": 844, + "2": 18380, + "3": 11334, + "4": 15918, + "5": 42053, + "7": 2570, + "8": 2876, + "9": 1984 + }, + "A_FTPT": { + "0": 130680, + "1": 10014, + "2": 1431 + }, + "NOW_CHAMPVA": { + "1": 368, + "2": 141757 + }, + "NOW_IHSFLG": { + "1": 685, + "2": 141440 + }, + "NOW_MCAID": { + "1": 24844, + "2": 117281 + }, + "NOW_MIL": { + "1": 4413, + "2": 137712 + }, + "NOW_NONM": { + "1": 8486, + "2": 133639 + }, + "NOW_OTHMT": { + "1": 312, + "2": 141813 + }, + "NOW_VACARE": { + "1": 1270, + "2": 140855 + }, + "PTOTVAL": { + "negative_rows": 75, + "nonzero_rows": 101163 + } + }, + "official_archive_url": "https://www2.census.gov/programs-surveys/cps/datasets/2025/march/asecpub25csv.zip", + "operation": "exact_source_join", + "reviewed_columns": [ + "NOW_MCAID", + "NOW_NONM", + "NOW_CHAMPVA", + "NOW_MIL", + "NOW_VACARE", + "NOW_OTHMT", + "NOW_IHSFLG", + "A_EXPRRP", + "PTOTVAL", + "A_ENRLW", + "A_FTPT" + ], + "source_form": "archive", + "source_path": "~/PolicyEngine/_buildm-runtime/inputs/asec_education/asecpub25csv.zip", + "survey_year": 2025 + }, + "raw_household_rows": 55762, + "raw_person_population": 325952369.06, + "raw_person_rows": 142125, + "relationship_recode_source": "source:A_EXPRRP", + "scale": 0.3333333333333333, + "share": 0.3333333333333333, + "source_file": "census_cps_2024.h5", + "weighted_person_population": 108650789.68666667, + "year": 2024 + }, + { + "census_person_columns": { + "archive_sha256": "fe819d0d2fc4470c282e76c247b8aa8b6054811619730307e9f2fe6186707d93", + "columns_added": [], + "columns_verified_equal": [ + "NOW_MCAID", + "NOW_NONM", + "NOW_CHAMPVA", + "NOW_MIL", + "NOW_VACARE", + "NOW_OTHMT", + "NOW_IHSFLG", + "A_EXPRRP", + "PTOTVAL", + "A_ENRLW", + "A_FTPT" + ], + "h5_person_rows": 134729, + "identity_columns_verified_equal": [ + "PH_SEQ", + "P_SEQ", + "A_LINENO", + "A_AGE" + ], + "income_year": 2025, + "join_key": "PERIDNUM (exact 22-digit string), one-to-one and total", + "joined_person_rows": 134729, + "member": "pppub26.csv", + "member_person_rows": 134729, + "member_sha256": "f78920699bf671cc211cb720ff8c6274392524a22138aedaf9e3954f91ba4334", + "member_size_bytes": 259887437, + "member_values": { + "A_ENRLW": { + "0": 69572, + "1": 10761, + "2": 54396 + }, + "A_EXPRRP": { + "1": 35882, + "10": 2404, + "11": 97, + "12": 1408, + "13": 3816, + "14": 800, + "2": 17462, + "3": 10808, + "4": 15562, + "5": 39531, + "7": 2452, + "8": 2717, + "9": 1790 + }, + "A_FTPT": { + "0": 123968, + "1": 9470, + "2": 1291 + }, + "NOW_CHAMPVA": { + "1": 414, + "2": 134315 + }, + "NOW_IHSFLG": { + "1": 661, + "2": 134068 + }, + "NOW_MCAID": { + "1": 22688, + "2": 112041 + }, + "NOW_MIL": { + "1": 3994, + "2": 130735 + }, + "NOW_NONM": { + "1": 8102, + "2": 126627 + }, + "NOW_OTHMT": { + "1": 235, + "2": 134494 + }, + "NOW_VACARE": { + "1": 1377, + "2": 133352 + }, + "PTOTVAL": { + "negative_rows": 64, + "nonzero_rows": 96720 + } + }, + "official_archive_url": "https://www2.census.gov/programs-surveys/cps/datasets/2026/march/asecpub26csv.zip", + "operation": "exact_source_join", + "reviewed_columns": [ + "NOW_MCAID", + "NOW_NONM", + "NOW_CHAMPVA", + "NOW_MIL", + "NOW_VACARE", + "NOW_OTHMT", + "NOW_IHSFLG", + "A_EXPRRP", + "PTOTVAL", + "A_ENRLW", + "A_FTPT" + ], + "source_form": "archive", + "source_path": "~/PolicyEngine/_buildm-runtime/inputs/asec_education/asecpub26csv.zip", + "survey_year": 2026 + }, + "raw_household_rows": 53344, + "raw_person_population": 329518151.84999996, + "raw_person_rows": 134729, + "relationship_recode_source": "source:A_EXPRRP", + "scale": 0.3297262656902908, + "share": 0.3333333333333333, + "source_file": "census_cps_2025.h5", + "weighted_person_population": 108650789.68666668, + "year": 2025 + } + ], + "anchor_year": 2024, + "signals": { + "eligibility_inputs_signal": { + "details": { + "any_parent_id_resolved_share": 0.3089212009129508, + "disabled_share": 0.11166887475103247, + "disabled_share_band": [ + 0.05, + 0.2 + ], + "full_time_college_student_share": 0.036407549887310355, + "full_time_student_share_band": [ + 0.01, + 0.08 + ], + "parent_1_id_resolved_share": 0.2889970942615618, + "parent_2_id_resolved_share": 0.22796556014051636, + "parent_share": 0.28569636476386817, + "parent_share_band": [ + 0.12, + 0.4 + ], + "pointers_unnameable_at_person_id_zero": 0, + "unique_counts": { + "is_blind": 2, + "is_disabled": 2, + "is_full_time_college_student": 2, + "own_children_in_household": 14, + "parent_1_id": 72202, + "parent_2_id": 56390, + "veterans_benefits": 1127 + }, + "veterans_benefits_share": 0.016930529740286363, + "veterans_benefits_share_band": [ + 0.002, + 0.04 + ] + }, + "failures": [], + "passed": true + }, + "housing_inputs_signal": { + "details": { + "has_spm_support_channel_metadata": false, + "household_tenure_values": [ + "NONE", + "OWNED_WITH_MORTGAGE", + "RENTED" + ], + "housing_assistance_positive_by_support_channel": {}, + "housing_assistance_receipt_invalid_count": 0, + "housing_assistance_receipt_missing_count": 0, + "housing_assistance_share": 0.035399810204329224, + "housing_assistance_share_band": [ + 0.005, + 0.08 + ], + "housing_assistance_share_by_support_channel": {}, + "housing_assistance_take_up_invalid_count": 0, + "housing_assistance_take_up_missing_count": 0, + "housing_assistance_take_up_source_mismatch_count": 0, + "negative_rent": 0, + "nonfinite_rent": 0, + "positive_rent_nonhead": 0, + "positive_rent_nonrenter": 30838, + "pre_subsidy_rent_share": 0.1337012964806105, + "pre_subsidy_rent_share_band": [ + 0.05, + 0.25 + ], + "pre_subsidy_rent_total": 660061913741.3123, + "pre_subsidy_rent_unweighted_total": 772002575.613394, + "spm_tenure_values": [ + "OWNER_WITHOUT_MORTGAGE", + "OWNER_WITH_MORTGAGE", + "RENTER" + ], + "unknown_household_tenure_values": [], + "unknown_spm_tenure_values": [] + }, + "failures": [], + "passed": true + }, + "medicare_take_up_input_signal": { + "details": { + "missing_count": 0, + "positive_count": 80041, + "source_mismatch_count": 0, + "source_rows": 421119, + "unique_count": 2, + "weighted_enrolled_share": 0.2001841056589198, + "weighted_enrolled_share_band": [ + 0.15, + 0.24 + ] + }, + "failures": [], + "passed": true + }, + "pregnancy_signal": { + "details": { + "clone_disagreement_source_persons": 0, + "malformed_source_id_rows": 0, + "missing_rows": 0, + "non_boolean_rows": 0, + "pregnant_female_outside_age_range_rows": 0, + "pregnant_ineligible_rows": 0, + "pregnant_nonfemale_rows": 0, + "pregnant_share": 0.008103994246385747, + "pregnant_share_band": [ + 0.002, + 0.02 + ], + "source_persons_checked": 0, + "unique_count": 2 + }, + "failures": [], + "passed": true + }, + "relationship_inputs_signal": { + "details": { + "household_head_share": 0.414256876345531, + "household_head_share_band": [ + 0.3, + 0.55 + ], + "households_without_exactly_one_head": 0, + "separated_and_surviving": 0, + "separated_share": 0.013732140150035821, + "separated_share_band": [ + 0.003, + 0.04 + ], + "surviving_spouse_share": 0.04759139559033301, + "surviving_spouse_share_band": [ + 0.02, + 0.08 + ], + "unique_counts": { + "is_household_head": 2, + "is_separated": 2, + "is_surviving_spouse": 2 + } + }, + "failures": [], + "passed": true + }, + "spm_independence_role_signal": { + "details": { + "derivation": { + "adult_child_person_count_mismatch_units": 0, + "adult_rule": "age >= 18 OR (age >= 15 AND is_spm_independent_minor_role)", + "ages_changed": false, + "all_rows_equal_source_columns": [ + "A_AGE", + "A_LINENO", + "P_SEQ", + "SPM_HAGE", + "SPM_NUMADULTS", + "SPM_NUMKIDS", + "SPM_NUMPER" + ], + "classification_changed_units_vs_age_only": 304, + "complete_source_membership_units": 172259, + "dataset_sha256": "d0e362ece441a25bf62ee9c21fb2fed22fbf29a100acb875f2c8ca967aab6d14", + "evidence_column": "is_spm_independence_role", + "evidence_status": "Inference from documented primitive relationships, reconciled against every SPM count in pinned complete Census ASEC sources; Census production program not retrieved.", + "frame_projection_columns": [ + "person_id", + "person_spm_unit_id", + "source_year", + "PERIDNUM", + "age", + "source_household_id", + "source_person_id", + "source_row_id", + "A_AGE", + "A_LINENO", + "P_SEQ", + "SPM_HAGE", + "SPM_NUMADULTS", + "SPM_NUMKIDS", + "SPM_NUMPER", + "A_FAMTYP", + "A_FAMREL", + "A_SPOUSE", + "PECOHAB" + ], + "frame_projection_sha256": "d0e362ece441a25bf62ee9c21fb2fed22fbf29a100acb875f2c8ca967aab6d14", + "independent_minor_persons": 306, + "minor_only_units_resolved": 108, + "native_spm_units": 172259, + "nonmissing_optional_raw_fields_checked": { + "A_FAMREL": 276854, + "A_FAMTYP": 276854, + "A_SPOUSE": 421119, + "PECOHAB": 276854 + }, + "period_handling": "Store the role before the age gate; model age supplies the requested period.", + "person_role_binding_sha256": "f115f114c5a6fb0cd2471061aa55c7849665ad912e46b77e2aa66d45d19fb729", + "persons_joined": 421119, + "primitive_column": "is_spm_independent_minor_role", + "primitive_rule": "SPM_HEAD == 1 OR (A_FAMTYP in {1,4} AND A_FAMREL in {1,2})", + "source_checks": [ + { + "adult_child_person_count_mismatch_units": 0, + "archive_sha256": "cdb39cdac34bef99dd0940ab28e306f692404c2eea44d85dfd634214872a0a09", + "csv_sha256": "21a2b9e0e4b08534563578a45acad77868af4ae9a7d46f23776b707d4a559aa7", + "csv_size_bytes": 277250415, + "extra_independent_minor_people_vs_head_only": 1, + "extra_independent_minor_units_vs_head_only": 1, + "income_year": 2023, + "member": "pppub24.csv", + "official_archive_url": "https://www2.census.gov/programs-surveys/cps/datasets/2024/march/asecpub24csv.zip", + "persons": 144265, + "survey_year": 2024, + "units": 58711 + }, + { + "adult_child_person_count_mismatch_units": 0, + "archive_sha256": "318845a2b5e0034eb2973898de1738f4df0025727de38499e7669cb9c0deef0b", + "csv_sha256": "06921fe83fc66c907e6c7b86b82255dc70458ee7d76258fc48297cb34f0c06b5", + "csv_size_bytes": 277882549, + "extra_independent_minor_people_vs_head_only": 6, + "extra_independent_minor_units_vs_head_only": 5, + "income_year": 2024, + "member": "pppub25.csv", + "official_archive_url": "https://www2.census.gov/programs-surveys/cps/datasets/2025/march/asecpub25csv.zip", + "persons": 142125, + "survey_year": 2025, + "units": 58147 + }, + { + "adult_child_person_count_mismatch_units": 0, + "archive_sha256": "fe819d0d2fc4470c282e76c247b8aa8b6054811619730307e9f2fe6186707d93", + "csv_sha256": "f78920699bf671cc211cb720ff8c6274392524a22138aedaf9e3954f91ba4334", + "csv_size_bytes": 259887437, + "extra_independent_minor_people_vs_head_only": 0, + "extra_independent_minor_units_vs_head_only": 0, + "income_year": 2025, + "member": "pppub26.csv", + "official_archive_url": "https://www2.census.gov/programs-surveys/cps/datasets/2026/march/asecpub26csv.zip", + "persons": 134729, + "survey_year": 2026, + "units": 55401 + } + ], + "source_join": [ + "source_year (income year)", + "PERIDNUM (exact 22-digit string)" + ], + "table_key": [ + "person_id", + "person_spm_unit_id" + ], + "table_order": "exact original HDF person order; verified", + "table_rows": 421119, + "total_source_people": 421119, + "total_source_units": 172259, + "true_role_persons": 254095, + "unmatched_persons": 0, + "weights_used": false + }, + "independent_minor_persons": 306, + "minor_role_share": 0.01811345214849585, + "minor_role_share_band": [ + 0.003, + 0.06 + ], + "persons_aged_15_to_17": 18756, + "role_missing_values": 0, + "role_share": 0.6280602193192539, + "role_share_band": [ + 0.4, + 0.75 + ], + "spm_composition": { + "n_units": 172259, + "n_units_without_classified_adult": 0, + "n_units_without_member_aged_18_or_over": 108, + "role_source": "source_column", + "status": "PASS" + }, + "unique_counts": { + "is_spm_independent_minor_role": 2 + } + }, + "failures": [], + "passed": true + }, + "wic_claim_signal": { + "details": { + "breastfeeding_rate_validated_but_unassigned": 0.663, + "breastfeeding_source_available": false, + "category_assignment_order": [ + "pregnant", + "postpartum", + "infant", + "child", + "none" + ], + "category_counts": { + "child": 19618, + "infant": 3526, + "none": 391255, + "postpartum": 3345, + "pregnant": 3375 + }, + "category_rates": { + "breastfeeding": 0.663, + "child": 0.46, + "infant": 0.784, + "none": 0.0, + "postpartum": 0.689, + "pregnant": 0.456 + }, + "category_weighted_claim_share_bands": { + "child": [ + 0.3, + 0.63 + ], + "infant": [ + 0.55, + 0.96 + ], + "none": [ + 0.0, + 0.0 + ], + "postpartum": [ + 0.45, + 0.9 + ], + "pregnant": [ + 0.25, + 0.67 + ] + }, + "category_weighted_claim_shares": { + "child": 0.4601764063841181, + "infant": 0.7848720896732944, + "none": 0.0, + "postpartum": 0.7007073175751649, + "pregnant": 0.44336213531053303 + }, + "category_weights": { + "child": 13961478.734755214, + "infant": 2774160.0335391797, + "none": 303951373.98828775, + "postpartum": 2623840.1799597978, + "pregnant": 2641516.1234580437 + }, + "clone_category_mismatch_count": 0, + "clone_claim_mismatch_count": 0, + "clone_group_count": 0, + "missing_count": 0, + "positive_count": 15613, + "unique_count": 2, + "weighted_claim_share": 0.03562421151061383, + "weighted_claim_share_band": [ + 0.015, + 0.06 + ] + }, + "failures": [], + "passed": true + } + } +} \ No newline at end of file diff --git a/experiments/us-asec-2025-source/receipts/reproduction_2025.json b/experiments/us-asec-2025-source/receipts/reproduction_2025.json new file mode 100644 index 000000000..3dd52d543 --- /dev/null +++ b/experiments/us-asec-2025-source/receipts/reproduction_2025.json @@ -0,0 +1,20 @@ +{ + "income_year": 2025, + "output": "/build/repro2025/census_cps_2025.h5", + "output_sha256": "51b516852b0d4a968b453fa17dc32d10ea0d169a668eda3b8eb2f4739c268e53", + "output_bytes": 304753967, + "download": { + "url": "https://www2.census.gov/programs-surveys/cps/datasets/2026/march/asecpub26csv.zip", + "sha256": "fe819d0d2fc4470c282e76c247b8aa8b6054811619730307e9f2fe6186707d93", + "bytes": 139103894, + "last_modified": "Tue, 15 Sep 2026 10:55:35 GMT", + "etag": null + }, + "loader_commit": "9f177ddd6197dd0e296ad05d080865a3e5c2cdbf", + "loader_dirty": "", + "python": "3.14.7", + "pandas": "2.3.3", + "tables": "3.11.1", + "numpy": "2.4.3", + "compared_with": "census_cps_2025.h5 as uploaded (sha256 4c5a3218...): keys, and all five tables equal under pandas assert_frame_equal(check_exact=True); same byte length; different sha256 (HDF5 metadata is not byte-deterministic)." +} \ No newline at end of file diff --git a/experiments/us-asec-2025-source/receipts/schema_diff_2024_2025.json b/experiments/us-asec-2025-source/receipts/schema_diff_2024_2025.json new file mode 100644 index 000000000..b6bdfef14 --- /dev/null +++ b/experiments/us-asec-2025-source/receipts/schema_diff_2024_2025.json @@ -0,0 +1,2826 @@ +{ + "keys": { + "old": [ + "/family", + "/household", + "/person", + "/spm_unit", + "/tax_unit" + ], + "new": [ + "/family", + "/household", + "/person", + "/spm_unit", + "/tax_unit" + ] + }, + "family": { + "rows": [ + 62479, + 59262 + ], + "n_columns": [ + 85, + 85 + ], + "dropped": [], + "added": [], + "index_name": [ + null, + null + ], + "dtype_changes": {}, + "column_order_same_for_shared": true, + "range_widened": { + "FPOVCUT": { + "old": [ + -1.0, + 69810.0 + ], + "new": [ + -1.0, + 71647.0 + ] + }, + "FHEADIDX": { + "old": [ + 1.0, + 13.0 + ], + "new": [ + 1.0, + 15.0 + ] + }, + "FANNVAL": { + "old": [ + 0.0, + 290000.0 + ], + "new": [ + 0.0, + 400000.0 + ] + }, + "FDISVAL": { + "old": [ + 0.0, + 100000.0 + ], + "new": [ + 0.0, + 186000.0 + ] + }, + "FDIVVAL": { + "old": [ + 0.0, + 700000.0 + ], + "new": [ + 0.0, + 1007499.0 + ] + }, + "FEARNVAL": { + "old": [ + -12998.0, + 3200598.0 + ], + "new": [ + -19998.0, + 3252900.0 + ] + }, + "FEDVAL": { + "old": [ + 0.0, + 123999.0 + ], + "new": [ + 0.0, + 145000.0 + ] + }, + "FFPOS": { + "old": [ + 1.0, + 10.0 + ], + "new": [ + 1.0, + 11.0 + ] + }, + "FFRVAL": { + "old": [ + -19998.0, + 400000.0 + ], + "new": [ + -29997.0, + 1072500.0 + ] + }, + "FINTVAL": { + "old": [ + 0.0, + 400050.0 + ], + "new": [ + 0.0, + 401200.0 + ] + }, + "FMED_VAL": { + "old": [ + 0.0, + 221000.0 + ], + "new": [ + 0.0, + 1004999.0 + ] + }, + "FMOOP": { + "old": [ + 0.0, + 324016.0 + ], + "new": [ + 0.0, + 1024839.0 + ] + }, + "FMOOP2": { + "old": [ + 0.0, + 276716.0 + ], + "new": [ + 0.0, + 1024839.0 + ] + }, + "FOTHVAL": { + "old": [ + -15396.0, + 1469659.0 + ], + "new": [ + -12500.0, + 1514506.0 + ] + }, + "FRNTVAL": { + "old": [ + -19998.0, + 1199999.0 + ], + "new": [ + -29997.0, + 1269999.0 + ] + }, + "FSEVAL": { + "old": [ + -19998.0, + 1124999.0 + ], + "new": [ + -19998.0, + 2199998.0 + ] + }, + "FSSVAL": { + "old": [ + 0.0, + 143308.0 + ], + "new": [ + 0.0, + 186000.0 + ] + }, + "FSUP_WGT": { + "old": [ + 11456.0, + 1324797.0 + ], + "new": [ + 12341.0, + 2046402.0 + ] + }, + "FSURVAL": { + "old": [ + 0.0, + 130000.0 + ], + "new": [ + 0.0, + 200000.0 + ] + }, + "FTOTVAL": { + "old": [ + -12997.0, + 3217166.0 + ], + "new": [ + -19897.0, + 3417539.0 + ] + }, + "FWCVAL": { + "old": [ + 0.0, + 72000.0 + ], + "new": [ + 0.0, + 99999.0 + ] + }, + "FWSVAL": { + "old": [ + 0.0, + 3200298.0 + ], + "new": [ + 0.0, + 3252900.0 + ] + }, + "F_MV_FS": { + "old": [ + 0.0, + 35000.0 + ], + "new": [ + 0.0, + 42250.0 + ] + }, + "F_MV_SL": { + "old": [ + 0.0, + 6629.0 + ], + "new": [ + 0.0, + 7746.0 + ] + }, + "FILEDATE": { + "old": [ + 72825.0, + 72825.0 + ], + "new": [ + 72726.0, + 72726.0 + ] + }, + "YYYYMM": { + "old": [ + 202503.0, + 202503.0 + ], + "new": [ + 202603.0, + 202603.0 + ] + } + }, + "code_set_changed": { + "FPOVCUT": { + "new_codes": [ + 15440, + 16749, + 19460, + 21558, + 22106, + 22190, + 25183, + 25913, + 25938, + 32649, + 32762, + 33207, + 33750, + 37833, + 38421, + 39384, + 40045, + 40628, + 42213, + 43018, + 44376, + 45289, + 46060, + 46242, + 46287, + 48183, + 49911, + 51392, + 52187, + 52523, + 52972, + 52997, + 53328, + 54740, + 56439, + 57777, + 58720, + 59273, + 59796, + 62241, + 64734, + 65139, + 66774, + 68581, + 69894, + 70694, + 71301, + 71647 + ], + "gone_codes": [ + 15045, + 16320, + 18961, + 21006, + 21540, + 21621, + 24537, + 25249, + 25273, + 31812, + 31922, + 32355, + 32884, + 36863, + 37436, + 38374, + 39019, + 39586, + 41131, + 41915, + 43238, + 44128, + 44879, + 45057, + 45100, + 46948, + 48631, + 50075, + 50849, + 51177, + 51614, + 51638, + 51961, + 53337, + 54992, + 56296, + 57215, + 57753, + 58263, + 60645, + 63075, + 63469, + 65062, + 66822, + 68102, + 68882, + 69473, + 69810 + ] + }, + "FPERSONS": { + "new_codes": [], + "gone_codes": [ + 14 + ] + }, + "FHEADIDX": { + "new_codes": [ + 14, + 15 + ], + "gone_codes": [] + }, + "FSPOUIDX": { + "new_codes": [ + 6, + 7 + ], + "gone_codes": [ + 8, + 9 + ] + }, + "FFPOS": { + "new_codes": [ + 11 + ], + "gone_codes": [] + }, + "F_MV_SL": { + "new_codes": [ + 67, + 133, + 265, + 398, + 431, + 526, + 531, + 574, + 618, + 663, + 789, + 796, + 861, + 922, + 993, + 1052, + 1054, + 1126, + 1147, + 1184, + 1187, + 1236, + 1259, + 1291, + 1578, + 1711, + 1721, + 1854, + 1987, + 2367, + 2582, + 2715, + 2765, + 3156, + 3443, + 3945, + 4303, + 4734, + 5164, + 6024, + 6312, + 6885, + 7746 + ], + "gone_codes": [ + 43, + 65, + 86, + 129, + 171, + 207, + 257, + 379, + 386, + 415, + 514, + 568, + 600, + 622, + 643, + 757, + 771, + 829, + 886, + 900, + 957, + 1014, + 1036, + 1086, + 1105, + 1136, + 1214, + 1243, + 1514, + 1643, + 1657, + 1771, + 1786, + 1799, + 1900, + 1914, + 2271, + 2486, + 3028, + 3107, + 3314, + 3785, + 4143, + 4542, + 4972, + 5299, + 6629 + ] + }, + "FILEDATE": { + "new_codes": [ + 72726 + ], + "gone_codes": [ + 72825 + ] + }, + "YYYYMM": { + "new_codes": [ + 202603 + ], + "gone_codes": [ + 202503 + ] + } + }, + "added_summaries": {}, + "dropped_summaries_old": {} + }, + "household": { + "rows": [ + 55762, + 53344 + ], + "n_columns": [ + 157, + 153 + ], + "dropped": [ + "HBBSUB_MNTH", + "HBBSUB_YN", + "HECC_PROBTIME", + "HXCC_PROBTIME", + "I_HBBSUBMNTH", + "I_HBBSUBYN" + ], + "added": [ + "HTCC_PROBLOSS", + "HXCC_PROBLOSS" + ], + "index_name": [ + null, + null + ], + "dtype_changes": {}, + "column_order_same_for_shared": false, + "range_widened": { + "H_YEAR": { + "old": [ + 2025.0, + 2025.0 + ], + "new": [ + 2026.0, + 2026.0 + ] + }, + "HANNVAL": { + "old": [ + 0.0, + 290000.0 + ], + "new": [ + 0.0, + 400000.0 + ] + }, + "HDISVAL": { + "old": [ + 0.0, + 127200.0 + ], + "new": [ + 0.0, + 186000.0 + ] + }, + "HDIVVAL": { + "old": [ + 0.0, + 700000.0 + ], + "new": [ + 0.0, + 1007499.0 + ] + }, + "HEARNVAL": { + "old": [ + -12998.0, + 3350997.0 + ], + "new": [ + -19998.0, + 3252900.0 + ] + }, + "HEDVAL": { + "old": [ + 0.0, + 126721.0 + ], + "new": [ + 0.0, + 228999.0 + ] + }, + "HENGVAL": { + "old": [ + 0.0, + 7000.0 + ], + "new": [ + 0.0, + 10000.0 + ] + }, + "HFDVAL": { + "old": [ + 0.0, + 35000.0 + ], + "new": [ + 0.0, + 42250.0 + ] + }, + "HFLUNNO": { + "old": [ + 0.0, + 8.0 + ], + "new": [ + 0.0, + 9.0 + ] + }, + "HFRVAL": { + "old": [ + -19998.0, + 400000.0 + ], + "new": [ + -29997.0, + 1072500.0 + ] + }, + "HHOTNO": { + "old": [ + 0.0, + 8.0 + ], + "new": [ + 0.0, + 9.0 + ] + }, + "HINTVAL": { + "old": [ + 0.0, + 400050.0 + ], + "new": [ + 0.0, + 401200.0 + ] + }, + "HNUMFAM": { + "old": [ + 1.0, + 10.0 + ], + "new": [ + 1.0, + 11.0 + ] + }, + "HOTHVAL": { + "old": [ + -15396.0, + 1469659.0 + ], + "new": [ + -12500.0, + 1514506.0 + ] + }, + "HRNTVAL": { + "old": [ + -19998.0, + 1199999.0 + ], + "new": [ + -29997.0, + 1269999.0 + ] + }, + "HSEVAL": { + "old": [ + -19998.0, + 1124999.0 + ], + "new": [ + -19998.0, + 2199998.0 + ] + }, + "HSSVAL": { + "old": [ + 0.0, + 143308.0 + ], + "new": [ + 0.0, + 186000.0 + ] + }, + "HSUP_WGT": { + "old": [ + 11456.0, + 1312189.0 + ], + "new": [ + 12341.0, + 2046402.0 + ] + }, + "HSURVAL": { + "old": [ + 0.0, + 148000.0 + ], + "new": [ + 0.0, + 200000.0 + ] + }, + "HTOTVAL": { + "old": [ + -12997.0, + 3350997.0 + ], + "new": [ + -19897.0, + 3417539.0 + ] + }, + "HWCVAL": { + "old": [ + 0.0, + 72000.0 + ], + "new": [ + 0.0, + 99999.0 + ] + }, + "HUNDER14": { + "old": [ + 0.0, + 9.0 + ], + "new": [ + 0.0, + 10.0 + ] + }, + "FILEDATE": { + "old": [ + 72825.0, + 72825.0 + ], + "new": [ + 72726.0, + 72726.0 + ] + }, + "YYYYMM": { + "old": [ + 202503.0, + 202503.0 + ], + "new": [ + 202603.0, + 202603.0 + ] + } + }, + "code_set_changed": { + "H_YEAR": { + "new_codes": [ + 2026 + ], + "gone_codes": [ + 2025 + ] + }, + "H_HHNUM": { + "new_codes": [], + "gone_codes": [ + 5 + ] + }, + "H_LIVQRT": { + "new_codes": [], + "gone_codes": [ + 9 + ] + }, + "H_RESPNM": { + "new_codes": [], + "gone_codes": [ + 13 + ] + }, + "H_NUMPER": { + "new_codes": [], + "gone_codes": [ + 14 + ] + }, + "HFLUNNO": { + "new_codes": [ + 9 + ], + "gone_codes": [] + }, + "HHOTNO": { + "new_codes": [ + 9 + ], + "gone_codes": [] + }, + "HNUMFAM": { + "new_codes": [ + 9, + 11 + ], + "gone_codes": [ + 10 + ] + }, + "HRNUMWIC": { + "new_codes": [], + "gone_codes": [ + 3 + ] + }, + "HUNDER14": { + "new_codes": [ + 10 + ], + "gone_codes": [] + }, + "GTCBSASZ": { + "new_codes": [ + 1 + ], + "gone_codes": [] + }, + "GTCSA": { + "new_codes": [ + 105, + 163, + 260, + 405 + ], + "gone_codes": [] + }, + "FILEDATE": { + "new_codes": [ + 72726 + ], + "gone_codes": [ + 72825 + ] + }, + "YYYYMM": { + "new_codes": [ + 202603 + ], + "gone_codes": [ + 202503 + ] + } + }, + "added_summaries": { + "HTCC_PROBLOSS": { + "kind": "num", + "min": 0.0, + "max": 260.0, + "n_unique": 64, + "zero_share": 0.989877024595081, + "neg_share": 0.0 + }, + "HXCC_PROBLOSS": { + "kind": "num", + "min": 0.0, + "max": 1.0, + "n_unique": 2, + "zero_share": 0.9964944511097781, + "neg_share": 0.0 + } + }, + "dropped_summaries_old": { + "HBBSUB_MNTH": { + "kind": "num", + "min": 0.0, + "max": 12.0, + "n_unique": 13, + "zero_share": 0.975036763387253, + "neg_share": 0.0 + }, + "HBBSUB_YN": { + "kind": "num", + "min": 1.0, + "max": 2.0, + "n_unique": 2, + "zero_share": 0.0, + "neg_share": 0.0 + }, + "HECC_PROBTIME": { + "kind": "num", + "min": 0.0, + "max": 800.0, + "n_unique": 63, + "zero_share": 0.9904415193142283, + "neg_share": 0.0 + }, + "HXCC_PROBTIME": { + "kind": "num", + "min": 0.0, + "max": 1.0, + "n_unique": 2, + "zero_share": 0.9970768623793982, + "neg_share": 0.0 + }, + "I_HBBSUBMNTH": { + "kind": "num", + "min": 0.0, + "max": 1.0, + "n_unique": 2, + "zero_share": 0.9939205910835336, + "neg_share": 0.0 + }, + "I_HBBSUBYN": { + "kind": "num", + "min": 0.0, + "max": 1.0, + "n_unique": 2, + "zero_share": 0.763656253362505, + "neg_share": 0.0 + } + } + }, + "person": { + "rows": [ + 142125, + 134729 + ], + "n_columns": [ + 162, + 161 + ], + "dropped": [ + "SPM_BBSUBVAL" + ], + "added": [], + "index_name": [ + null, + null + ], + "dtype_changes": {}, + "column_order_same_for_shared": true, + "range_widened": { + "PF_SEQ": { + "old": [ + 1.0, + 10.0 + ], + "new": [ + 1.0, + 11.0 + ] + }, + "PEINUSYR": { + "old": [ + 0.0, + 28.0 + ], + "new": [ + 0.0, + 29.0 + ] + }, + "AGI": { + "old": [ + -71592.0, + 3124826.0 + ], + "new": [ + -19998.0, + 3266881.0 + ] + }, + "ANN_VAL": { + "old": [ + 0.0, + 250000.0 + ], + "new": [ + 0.0, + 400000.0 + ] + }, + "CTC_CRD": { + "old": [ + 0.0, + 14000.0 + ], + "new": [ + 0.0, + 17600.0 + ] + }, + "DIS_VAL2": { + "old": [ + 0.0, + 46800.0 + ], + "new": [ + 0.0, + 88000.0 + ] + }, + "DIV_VAL": { + "old": [ + 0.0, + 500000.0 + ], + "new": [ + 0.0, + 999999.0 + ] + }, + "DST_VAL2": { + "old": [ + 0.0, + 300000.0 + ], + "new": [ + 0.0, + 500000.0 + ] + }, + "DST_VAL2_YNG": { + "old": [ + 0.0, + 150000.0 + ], + "new": [ + 0.0, + 180000.0 + ] + }, + "EIT_CRED": { + "old": [ + 0.0, + 7830.0 + ], + "new": [ + 0.0, + 8046.0 + ] + }, + "FEDTAX_AC": { + "old": [ + -12097.0, + 982908.0 + ], + "new": [ + -12579.0, + 1013848.0 + ] + }, + "FEDTAX_BC": { + "old": [ + 0.0, + 982908.0 + ], + "new": [ + 0.0, + 1013848.0 + ] + }, + "FRSE_VAL": { + "old": [ + -19998.0, + 400000.0 + ], + "new": [ + -19998.0, + 1072500.0 + ] + }, + "INT_VAL": { + "old": [ + 0.0, + 230999.0 + ], + "new": [ + 0.0, + 249999.0 + ] + }, + "MOOP": { + "old": [ + 0.0, + 278616.0 + ], + "new": [ + 0.0, + 1006439.0 + ] + }, + "PMED_VAL": { + "old": [ + 0.0, + 220000.0 + ], + "new": [ + 0.0, + 999999.0 + ] + }, + "PTOTVAL": { + "old": [ + -9999.0, + 3149999.0 + ], + "new": [ + -9999.0, + 3277551.0 + ] + }, + "RETCB_VAL": { + "old": [ + 0.0, + 38500.0 + ], + "new": [ + 0.0, + 39000.0 + ] + }, + "SEMP_VAL": { + "old": [ + -19998.0, + 1099999.0 + ], + "new": [ + -9999.0, + 1102999.0 + ] + }, + "SPM_CAPWKCCXPNS": { + "old": [ + 0.0, + 151820.0 + ], + "new": [ + 0.0, + 163712.0 + ] + }, + "SPM_EITC": { + "old": [ + 0.0, + 13046.0 + ], + "new": [ + 0.0, + 13380.0 + ] + }, + "SPM_ENGVAL": { + "old": [ + 0.0, + 7000.0 + ], + "new": [ + 0.0, + 10000.0 + ] + }, + "SPM_FEDTAX": { + "old": [ + -17266.0, + 1029034.0 + ], + "new": [ + -22615.0, + 1048282.0 + ] + }, + "SPM_FEDTAXBC": { + "old": [ + 0.0, + 1029034.0 + ], + "new": [ + 0.0, + 1048282.0 + ] + }, + "SPM_FICA": { + "old": [ + 0.0, + 101166.0 + ], + "new": [ + 0.0, + 106601.0 + ] + }, + "SPM_MEDXPNS": { + "old": [ + 0.0, + 324016.0 + ], + "new": [ + 0.0, + 1024839.0 + ] + }, + "SPM_POVTHRESHOLD": { + "old": [ + 12936.0, + 125019.0 + ], + "new": [ + 13681.0, + 135415.0 + ] + }, + "SPM_RESOURCES": { + "old": [ + -269381.0, + 1832837.0 + ], + "new": [ + -876582.0, + 2182293.0 + ] + }, + "SPM_SCHLUNCH": { + "old": [ + 0.0, + 7592.0 + ], + "new": [ + 0.0, + 7749.0 + ] + }, + "SPM_SNAPSUB": { + "old": [ + 0.0, + 35000.0 + ], + "new": [ + 0.0, + 42250.0 + ] + }, + "SPM_TOTVAL": { + "old": [ + -12997.0, + 3217166.0 + ], + "new": [ + -19897.0, + 3417539.0 + ] + }, + "SPM_WEIGHT": { + "old": [ + 11456.0, + 1324797.0 + ], + "new": [ + 12341.0, + 2046402.0 + ] + }, + "SPM_WICVAL": { + "old": [ + 0.0, + 3925.0 + ], + "new": [ + 0.0, + 4091.0 + ] + }, + "SPM_WKXPNS": { + "old": [ + 0.0, + 14560.0 + ], + "new": [ + 0.0, + 16704.0 + ] + }, + "SSI_VAL": { + "old": [ + 0.0, + 30600.0 + ], + "new": [ + 0.0, + 60000.0 + ] + }, + "SS_VAL": { + "old": [ + 0.0, + 72000.0 + ], + "new": [ + 0.0, + 74220.0 + ] + }, + "TAX_INC": { + "old": [ + 0.0, + 2860763.0 + ], + "new": [ + 0.0, + 2946515.0 + ] + }, + "WC_VAL": { + "old": [ + 0.0, + 72000.0 + ], + "new": [ + 0.0, + 99999.0 + ] + }, + "WSAL_VAL": { + "old": [ + 0.0, + 2099999.0 + ], + "new": [ + 0.0, + 2199998.0 + ] + } + }, + "code_set_changed": { + "A_LINENO": { + "new_codes": [], + "gone_codes": [ + 16 + ] + }, + "PF_SEQ": { + "new_codes": [ + 11 + ], + "gone_codes": [] + }, + "PEINUSYR": { + "new_codes": [ + 29 + ], + "gone_codes": [] + }, + "PEPAR1": { + "new_codes": [], + "gone_codes": [ + 11, + 13, + 14 + ] + }, + "PEPAR2": { + "new_codes": [], + "gone_codes": [ + 13 + ] + }, + "PECOHAB": { + "new_codes": [], + "gone_codes": [ + 12 + ] + }, + "A_SPOUSE": { + "new_codes": [], + "gone_codes": [ + 13, + 14 + ] + }, + "DIS_SC2": { + "new_codes": [], + "gone_codes": [ + 9 + ] + }, + "DIS_VAL2": { + "new_codes": [ + 3200, + 3600, + 5100, + 6000, + 6300, + 9000, + 9120, + 14400, + 20000, + 23348, + 30000, + 80000, + 88000 + ], + "gone_codes": [ + 12, + 120, + 636, + 1700, + 3900, + 4200, + 4800, + 5700, + 6088, + 9002, + 10400, + 12000, + 16000, + 21600, + 46800 + ] + }, + "DST_SC1_YNG": { + "new_codes": [ + 5 + ], + "gone_codes": [] + }, + "DST_SC2_YNG": { + "new_codes": [ + 2 + ], + "gone_codes": [ + 6 + ] + }, + "DST_VAL2_YNG": { + "new_codes": [ + 1200, + 1468, + 2049, + 2200, + 5000, + 7500, + 8000, + 14400, + 20000, + 22000, + 30000, + 36000, + 48996, + 52000, + 100000, + 180000 + ], + "gone_codes": [ + 100, + 200, + 500, + 996, + 1171, + 2000, + 2750, + 3600, + 3924, + 4615, + 6000, + 9000, + 10000, + 12000, + 18000, + 60000, + 65000, + 150000 + ] + }, + "OI_OFF": { + "new_codes": [ + 5, + 6, + 8 + ], + "gone_codes": [] + }, + "PEN_SC2": { + "new_codes": [ + 7 + ], + "gone_codes": [] + }, + "RESNSSI2": { + "new_codes": [ + 4 + ], + "gone_codes": [] + }, + "SPM_EQUIVSCALE": { + "new_codes": [ + 1.5864, + 1.6809, + 1.7733 + ], + "gone_codes": [ + 2.4831 + ] + }, + "SPM_NUMADULTS": { + "new_codes": [ + 10 + ], + "gone_codes": [ + 11 + ] + }, + "SPM_NUMPER": { + "new_codes": [], + "gone_codes": [ + 14 + ] + } + }, + "added_summaries": {}, + "dropped_summaries_old": { + "SPM_BBSUBVAL": { + "kind": "num", + "min": 0.0, + "max": 375.0, + "n_unique": 22, + "zero_share": 0.9732348284960423, + "neg_share": 0.0 + } + } + }, + "spm_unit": { + "rows": [ + 58147, + 55401 + ], + "n_columns": [ + 39, + 38 + ], + "dropped": [ + "SPM_BBSUBVAL" + ], + "added": [], + "index_name": [ + "SPM_ID", + "SPM_ID" + ], + "dtype_changes": {}, + "column_order_same_for_shared": true, + "range_widened": { + "SPM_CAPWKCCXPNS": { + "old": [ + 0.0, + 151820.0 + ], + "new": [ + 0.0, + 163712.0 + ] + }, + "SPM_EITC": { + "old": [ + 0.0, + 13046.0 + ], + "new": [ + 0.0, + 13380.0 + ] + }, + "SPM_ENGVAL": { + "old": [ + 0.0, + 7000.0 + ], + "new": [ + 0.0, + 10000.0 + ] + }, + "SPM_FEDTAX": { + "old": [ + -17266.0, + 1029034.0 + ], + "new": [ + -22615.0, + 1048282.0 + ] + }, + "SPM_FEDTAXBC": { + "old": [ + 0.0, + 1029034.0 + ], + "new": [ + 0.0, + 1048282.0 + ] + }, + "SPM_FICA": { + "old": [ + 0.0, + 101166.0 + ], + "new": [ + 0.0, + 106601.0 + ] + }, + "SPM_MEDXPNS": { + "old": [ + 0.0, + 324016.0 + ], + "new": [ + 0.0, + 1024839.0 + ] + }, + "SPM_POVTHRESHOLD": { + "old": [ + 12936.0, + 125019.0 + ], + "new": [ + 13681.0, + 135415.0 + ] + }, + "SPM_RESOURCES": { + "old": [ + -269381.0, + 1832837.0 + ], + "new": [ + -876582.0, + 2182293.0 + ] + }, + "SPM_SCHLUNCH": { + "old": [ + 0.0, + 7592.0 + ], + "new": [ + 0.0, + 7749.0 + ] + }, + "SPM_SNAPSUB": { + "old": [ + 0.0, + 35000.0 + ], + "new": [ + 0.0, + 42250.0 + ] + }, + "SPM_TOTVAL": { + "old": [ + -12997.0, + 3217166.0 + ], + "new": [ + -19897.0, + 3417539.0 + ] + }, + "SPM_WEIGHT": { + "old": [ + 11456.0, + 1324797.0 + ], + "new": [ + 12341.0, + 2046402.0 + ] + }, + "SPM_WICVAL": { + "old": [ + 0.0, + 3925.0 + ], + "new": [ + 0.0, + 4091.0 + ] + }, + "SPM_WKXPNS": { + "old": [ + 0.0, + 14560.0 + ], + "new": [ + 0.0, + 16704.0 + ] + } + }, + "code_set_changed": { + "SPM_EQUIVSCALE": { + "new_codes": [ + 1.5864, + 1.6809, + 1.7733 + ], + "gone_codes": [ + 2.4831 + ] + }, + "SPM_NUMADULTS": { + "new_codes": [ + 10 + ], + "gone_codes": [ + 11 + ] + }, + "SPM_NUMPER": { + "new_codes": [], + "gone_codes": [ + 14 + ] + }, + "SPM_WNEWHEAD": { + "new_codes": [], + "gone_codes": [ + 0 + ] + } + }, + "added_summaries": {}, + "dropped_summaries_old": { + "SPM_BBSUBVAL": { + "kind": "num", + "min": 0.0, + "max": 375.0, + "n_unique": 22, + "zero_share": 0.9751491908438956, + "neg_share": 0.0 + } + } + }, + "tax_unit": { + "rows": [ + 74697, + 71085 + ], + "n_columns": [ + 1, + 1 + ], + "dropped": [], + "added": [], + "index_name": [ + null, + null + ], + "dtype_changes": {}, + "column_order_same_for_shared": true, + "range_widened": {}, + "code_set_changed": {}, + "added_summaries": {}, + "dropped_summaries_old": {} + }, + "coverage": { + "old": { + "NOW_columns": [ + "NOW_CAID", + "NOW_CHAMPVA", + "NOW_COV", + "NOW_DIR", + "NOW_GRP", + "NOW_IHSFLG", + "NOW_MCAID", + "NOW_MCARE", + "NOW_MIL", + "NOW_MRK", + "NOW_MRKS", + "NOW_MRKUN", + "NOW_NONM", + "NOW_OTHMT", + "NOW_PCHIP", + "NOW_PRIV", + "NOW_PUB", + "NOW_VACARE" + ], + "NOW_CAID": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 51.52 + }, + "NOW_CHAMPVA": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 0.68 + }, + "NOW_COV": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 248.33 + }, + "NOW_DIR": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 23.66 + }, + "NOW_GRP": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 166.31 + }, + "NOW_IHSFLG": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 0.79 + }, + "NOW_MCAID": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 53.8 + }, + "NOW_MCARE": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 6.85 + }, + "NOW_MIL": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 6.5 + }, + "NOW_MRK": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 14.61 + }, + "NOW_MRKS": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 10.13 + }, + "NOW_MRKUN": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 4.49 + }, + "NOW_NONM": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 9.05 + }, + "NOW_OTHMT": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 0.67 + }, + "NOW_PCHIP": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 1.61 + }, + "NOW_PRIV": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 194.26 + }, + "NOW_PUB": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 60.67 + }, + "NOW_VACARE": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 1.98 + }, + "geo": { + "GTCO": { + "dtype": "int64", + "n_unique": 100, + "nonzero_share_hh": 0.41284387217101254, + "nonzero_share_weighted": 0.4527315190926383 + }, + "GTCBSA": { + "dtype": "int64", + "n_unique": 261, + "nonzero_share_hh": 0.7544385065098096, + "nonzero_share_weighted": 0.825117676362696 + }, + "GTCSA": { + "dtype": "int64", + "n_unique": 45, + "nonzero_share_hh": 0.42338868763674187, + "nonzero_share_weighted": 0.47331041588035755 + }, + "GESTFIPS": { + "dtype": "int64", + "n_unique": 51, + "nonzero_share_hh": 1.0, + "nonzero_share_weighted": 1.0 + }, + "GTMETSTA": { + "dtype": "int64", + "n_unique": 3, + "nonzero_share_hh": 1.0, + "nonzero_share_weighted": 1.0 + }, + "GTINDVPC": { + "dtype": "int64", + "n_unique": 8, + "nonzero_share_hh": 0.13588106595889674, + "nonzero_share_weighted": 0.1528812739610012 + }, + "distinct_identified_counties": 280, + "identified_county_fips": [ + 1003, + 1081, + 1097, + 4013, + 4019, + 4021, + 4025, + 4027, + 6001, + 6007, + 6017, + 6019, + 6029, + 6031, + 6037, + 6053, + 6059, + 6061, + 6067, + 6073, + 6075, + 6077, + 6079, + 6081, + 6083, + 6087, + 6089, + 6095, + 6097, + 6099, + 6107, + 6111, + 6113, + 8013, + 8031, + 8059, + 8069, + 8123, + 9001, + 9005, + 9009, + 9015, + 10001, + 10003, + 10005, + 11001, + 12005, + 12009, + 12011, + 12019, + 12021, + 12033, + 12053, + 12057, + 12069, + 12071, + 12083, + 12085, + 12086, + 12095, + 12099, + 12101, + 12103, + 12105, + 12109, + 12111, + 12113, + 13015, + 13045, + 13057, + 13063, + 13077, + 13097, + 13113, + 13117, + 13135, + 13139, + 13151, + 13223, + 15003, + 17097, + 17111, + 17119, + 17163, + 17179, + 18019, + 18039, + 18063, + 18081, + 18089, + 18105, + 18141, + 19103, + 19113, + 19163, + 20091, + 20173, + 21015, + 21067, + 21111, + 21117, + 22005, + 22033, + 22051, + 22063, + 22071, + 22073, + 22103, + 23001, + 23005, + 23011, + 23019, + 24003, + 24013, + 24015, + 24017, + 24025, + 24031, + 24033, + 24510, + 25001, + 25005, + 25009, + 25013, + 25015, + 25017, + 25021, + 25023, + 25025, + 25027, + 26005, + 26021, + 26025, + 26049, + 26075, + 26081, + 26093, + 26099, + 26115, + 26121, + 26125, + 26145, + 26161, + 26163, + 27003, + 27123, + 27139, + 27163, + 27171, + 29071, + 29099, + 29189, + 30111, + 31055, + 32003, + 33011, + 33013, + 33015, + 33017, + 34003, + 34005, + 34007, + 34011, + 34013, + 34017, + 34019, + 34021, + 34023, + 34027, + 34031, + 34035, + 34037, + 34039, + 35001, + 35013, + 35045, + 35049, + 36005, + 36047, + 36055, + 36059, + 36061, + 36067, + 36069, + 36071, + 36081, + 36085, + 36087, + 36091, + 36103, + 36119, + 37001, + 37021, + 37057, + 37067, + 37119, + 37133, + 37147, + 37155, + 37159, + 37179, + 37191, + 39025, + 39057, + 39085, + 39089, + 39095, + 39103, + 39109, + 39113, + 39133, + 39153, + 41017, + 41029, + 41039, + 42003, + 42007, + 42011, + 42017, + 42019, + 42021, + 42029, + 42043, + 42045, + 42049, + 42055, + 42071, + 42081, + 42085, + 42089, + 42091, + 42101, + 42107, + 42125, + 42129, + 42133, + 45041, + 45051, + 45083, + 45091, + 47009, + 47093, + 47125, + 47165, + 47189, + 48041, + 48061, + 48135, + 48139, + 48181, + 48215, + 48251, + 48309, + 48423, + 48439, + 48441, + 48479, + 48485, + 49053, + 51013, + 51041, + 51087, + 51107, + 51153, + 51177, + 51179, + 51550, + 51700, + 51710, + 51760, + 51810, + 53033, + 53057, + 53061, + 54039, + 55059, + 55073, + 55101, + 55105, + 55139 + ] + }, + "households": 55762, + "persons": 142125, + "H_IDNUM": { + "present": true, + "dtype": "str", + "unique": 55762 + } + }, + "new": { + "NOW_columns": [ + "NOW_CAID", + "NOW_CHAMPVA", + "NOW_COV", + "NOW_DIR", + "NOW_GRP", + "NOW_IHSFLG", + "NOW_MCAID", + "NOW_MCARE", + "NOW_MIL", + "NOW_MRK", + "NOW_MRKS", + "NOW_MRKUN", + "NOW_NONM", + "NOW_OTHMT", + "NOW_PCHIP", + "NOW_PRIV", + "NOW_PUB", + "NOW_VACARE" + ], + "NOW_CAID": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 48.48 + }, + "NOW_CHAMPVA": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 0.76 + }, + "NOW_COV": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 244.39 + }, + "NOW_DIR": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 23.21 + }, + "NOW_GRP": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 165.5 + }, + "NOW_IHSFLG": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 0.89 + }, + "NOW_MCAID": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 50.93 + }, + "NOW_MCARE": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 6.79 + }, + "NOW_MIL": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 6.11 + }, + "NOW_MRK": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 14.31 + }, + "NOW_MRKS": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 9.32 + }, + "NOW_MRKUN": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 4.99 + }, + "NOW_NONM": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 8.9 + }, + "NOW_OTHMT": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 0.74 + }, + "NOW_PCHIP": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 1.72 + }, + "NOW_PRIV": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 193.11 + }, + "NOW_PUB": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 57.95 + }, + "NOW_VACARE": { + "codes": [ + 1, + 2 + ], + "w_yes_u65_m": 2.0 + }, + "geo": { + "GTCO": { + "dtype": "int64", + "n_unique": 101, + "nonzero_share_hh": 0.4174227654469106, + "nonzero_share_weighted": 0.45391665873253445 + }, + "GTCBSA": { + "dtype": "int64", + "n_unique": 280, + "nonzero_share_hh": 0.7422952909418117, + "nonzero_share_weighted": 0.8196354199542601 + }, + "GTCSA": { + "dtype": "int64", + "n_unique": 49, + "nonzero_share_hh": 0.4147420515896821, + "nonzero_share_weighted": 0.46409062674982166 + }, + "GESTFIPS": { + "dtype": "int64", + "n_unique": 51, + "nonzero_share_hh": 1.0, + "nonzero_share_weighted": 1.0 + }, + "GTMETSTA": { + "dtype": "int64", + "n_unique": 3, + "nonzero_share_hh": 1.0, + "nonzero_share_weighted": 1.0 + }, + "GTINDVPC": { + "dtype": "int64", + "n_unique": 8, + "nonzero_share_hh": 0.13585407918416317, + "nonzero_share_weighted": 0.15405975264111071 + }, + "distinct_identified_counties": 307, + "identified_county_fips": [ + 1003, + 1079, + 1081, + 1097, + 1103, + 1107, + 1125, + 4003, + 4013, + 4019, + 4021, + 4025, + 4027, + 6001, + 6007, + 6017, + 6019, + 6029, + 6031, + 6037, + 6039, + 6053, + 6059, + 6061, + 6067, + 6073, + 6075, + 6077, + 6079, + 6081, + 6083, + 6087, + 6089, + 6095, + 6097, + 6099, + 6101, + 6107, + 6111, + 6113, + 6115, + 8013, + 8031, + 8059, + 8069, + 8101, + 8123, + 9001, + 9005, + 9009, + 9015, + 10001, + 10003, + 10005, + 11001, + 12005, + 12009, + 12011, + 12015, + 12019, + 12021, + 12033, + 12053, + 12057, + 12061, + 12069, + 12071, + 12083, + 12085, + 12086, + 12095, + 12099, + 12101, + 12103, + 12105, + 12109, + 12111, + 12113, + 12119, + 13015, + 13045, + 13057, + 13063, + 13077, + 13097, + 13113, + 13117, + 13135, + 13139, + 13151, + 13223, + 15003, + 17097, + 17111, + 17119, + 17163, + 17179, + 18019, + 18039, + 18063, + 18081, + 18089, + 18105, + 18141, + 19103, + 19113, + 19163, + 20091, + 20173, + 21015, + 21067, + 21111, + 21117, + 22005, + 22033, + 22051, + 22063, + 22071, + 22073, + 22103, + 22105, + 23001, + 23005, + 23011, + 23019, + 24003, + 24013, + 24015, + 24017, + 24025, + 24031, + 24033, + 24043, + 24510, + 25001, + 25005, + 25009, + 25013, + 25015, + 25017, + 25021, + 25023, + 25025, + 25027, + 26005, + 26021, + 26025, + 26027, + 26049, + 26075, + 26081, + 26093, + 26099, + 26115, + 26121, + 26125, + 26145, + 26161, + 26163, + 27003, + 27123, + 27137, + 27139, + 27163, + 27171, + 29019, + 29071, + 29099, + 29189, + 30111, + 31055, + 32003, + 33011, + 33013, + 33015, + 33017, + 34003, + 34005, + 34007, + 34011, + 34013, + 34017, + 34019, + 34021, + 34023, + 34027, + 34031, + 34035, + 34037, + 34039, + 35001, + 35013, + 35045, + 35049, + 36005, + 36047, + 36055, + 36059, + 36061, + 36067, + 36069, + 36071, + 36081, + 36085, + 36087, + 36091, + 36103, + 36111, + 36119, + 37001, + 37019, + 37021, + 37057, + 37067, + 37119, + 37129, + 37133, + 37141, + 37147, + 37155, + 37159, + 37179, + 37191, + 39023, + 39025, + 39057, + 39085, + 39089, + 39095, + 39103, + 39109, + 39113, + 39133, + 39153, + 41017, + 41029, + 41039, + 42003, + 42007, + 42011, + 42017, + 42019, + 42021, + 42029, + 42043, + 42045, + 42049, + 42055, + 42071, + 42075, + 42081, + 42085, + 42089, + 42091, + 42101, + 42107, + 42125, + 42129, + 42133, + 45041, + 45051, + 45083, + 45085, + 45091, + 47009, + 47093, + 47125, + 47165, + 47189, + 48041, + 48061, + 48135, + 48139, + 48181, + 48215, + 48251, + 48309, + 48423, + 48439, + 48441, + 48479, + 48485, + 49053, + 51013, + 51041, + 51087, + 51107, + 51153, + 51177, + 51179, + 51550, + 51700, + 51710, + 51760, + 51810, + 53033, + 53035, + 53057, + 53061, + 54019, + 54039, + 54081, + 55059, + 55073, + 55101, + 55105, + 55139 + ] + }, + "households": 53344, + "persons": 134729, + "H_IDNUM": { + "present": true, + "dtype": "str", + "unique": 53344 + } + } + }, + "rotation": { + "new_households": 53344, + "old_households": 55762, + "shared_H_IDNUM": 16601, + "share_of_new_in_old": 0.31120650869826033, + "share_of_new_in_old_weighted": 0.31470292867802857, + "share_of_old_in_new": 0.2977117033104982, + "by_new_H_MIS": { + "1": 0.5097978706989662, + "2": 0.545229627338797, + "3": 0.5380194518125553, + "4": 0.5515855292541313, + "5": 0.09496533011757612, + "6": 0.0843768634466309, + "7": 0.09169221784202074, + "8": 0.08818263205013428 + }, + "matched_state_agreement": 1.0 + } +} \ No newline at end of file diff --git a/experiments/us-asec-2025-source/scripts/compare_stores.py b/experiments/us-asec-2025-source/scripts/compare_stores.py new file mode 100644 index 000000000..7bae1fb98 --- /dev/null +++ b/experiments/us-asec-2025-source/scripts/compare_stores.py @@ -0,0 +1,40 @@ +"""Table-by-table exact comparison of two processed CPS ASEC HDF stores.""" + +import json +import sys + +import pandas as pd + +a, b = sys.argv[1], sys.argv[2] +out = {} +with pd.HDFStore(a, "r") as sa, pd.HDFStore(b, "r") as sb: + ka, kb = sorted(sa.keys()), sorted(sb.keys()) + out["keys_equal"] = ka == kb + out["keys"] = [ka, kb] + for k in sorted(set(ka) & set(kb)): + da, db = sa[k], sb[k] + r = dict( + shape=[list(da.shape), list(db.shape)], + columns_equal_ordered=list(da.columns) == list(db.columns), + only_a=sorted(set(da.columns) - set(db.columns)), + only_b=sorted(set(db.columns) - set(da.columns)), + index_name=[da.index.name, db.index.name], + index_equal=bool(da.index.equals(db.index)), + dtype_diffs={ + c: [str(da[c].dtype), str(db[c].dtype)] + for c in set(da.columns) & set(db.columns) + if str(da[c].dtype) != str(db[c].dtype) + }, + ) + try: + pd.testing.assert_frame_equal(da, db, check_exact=True) + r["frame_equal_exact"] = True + except AssertionError as e: + r["frame_equal_exact"] = False + r["first_diff"] = str(e)[:600] + if da.shape == db.shape and list(da.columns) == list(db.columns): + r["value_diff_columns"] = [ + c for c in da.columns if not da[c].equals(db[c]) + ][:50] + out[k] = r +print(json.dumps(out, indent=1, default=str)) diff --git a/experiments/us-asec-2025-source/scripts/measure_pool.py b/experiments/us-asec-2025-source/scripts/measure_pool.py new file mode 100644 index 000000000..4e75ef338 --- /dev/null +++ b/experiments/us-asec-2025-source/scripts/measure_pool.py @@ -0,0 +1,93 @@ +"""Measure the income-year 2025 processed ASEC file against income year 2024. + +usage: measure_pool.py OLD_2024.h5 NEW_2025.h5 COUNTY_LIST_CSV OUT.json + +Households and persons, H_IDNUM rotation overlap (validated by reference-person +sex and age), county identification (GTCO != 0), and the coded counties against +the Census identified-county list for the matching survey year. +""" + +import csv +import json +import sys + +import numpy as np +import pandas as pd + +old_p, new_p, county_csv, out_p = sys.argv[1:5] + + +def load(p): + h = pd.read_hdf(p, "household") + per = pd.read_hdf(p, "person") + return h, per + + +res = {} +for label, path, survey in (("income_2024", old_p, 2025), ("income_2025", new_p, 2026)): + h, per = load(path) + w = h["HSUP_WGT"] / 100.0 + coded = h["GTCO"] != 0 + fips = (h.loc[coded, "GESTFIPS"] * 1000 + h.loc[coded, "GTCO"]).astype(int) + listed = { + int(r["state_fips"]) * 1000 + int(r["county_code"]) + for r in csv.DictReader(open(county_csv)) + if int(r["asec_year"]) == survey + } + coded_set = set(fips.unique().tolist()) + res[label] = dict( + survey_year=survey, + households=int(len(h)), + persons=int(len(per)), + weighted_households_millions=round(float(w.sum()) / 1e6, 3), + county_identified_households=int(coded.sum()), + county_identified_share=float(coded.mean()), + county_identified_share_weighted=float(w[coded].sum() / w.sum()), + distinct_coded_counties=len(coded_set), + census_list4_counties=len(listed), + coded_and_listed=len(coded_set & listed), + coded_not_listed=len(coded_set - listed), + listed_not_coded=len(listed - coded_set), + coded_not_listed_fips=sorted(coded_set - listed), + ) + +ho, po = load(old_p) +hn, pn = load(new_p) + + +def ref(h, p): + """Households with their reference person's age and sex.""" + persons = p[p.A_EXPRRP.isin([1, 2])].drop_duplicates("PH_SEQ") + return h[["H_SEQ", "H_IDNUM", "HSUP_WGT"]].merge( + persons[["PH_SEQ", "A_AGE", "A_SEX"]], + left_on="H_SEQ", + right_on="PH_SEQ", + how="left", + ) + + +o, n = ref(ho, po), ref(hn, pn) +in_old = n["H_IDNUM"].isin(set(o["H_IDNUM"])) +m = n.merge(o, on="H_IDNUM", suffixes=("_new", "_old")) +rng = np.random.default_rng(20260927) +a = n.iloc[rng.integers(0, len(n), 20000)].reset_index(drop=True) +b = o.iloc[rng.integers(0, len(o), 20000)].reset_index(drop=True) +d_rand = a.A_AGE.values - b.A_AGE.values +res["rotation_income_2025_vs_2024"] = dict( + shared_H_IDNUM=int(in_old.sum()), + share_of_2025_households=float(in_old.mean()), + share_of_2025_households_weighted=float( + n.HSUP_WGT[in_old].sum() / n.HSUP_WGT.sum() + ), + share_of_2024_households=float(o["H_IDNUM"].isin(set(n["H_IDNUM"])).mean()), + linked_reference_person_sex_agrees=float((m.A_SEX_new == m.A_SEX_old).mean()), + linked_reference_person_age_plus_0_to_2=float( + (m.A_AGE_new - m.A_AGE_old).between(0, 2).mean() + ), + random_pairs_sex_agrees=float((a.A_SEX.values == b.A_SEX.values).mean()), + random_pairs_age_plus_0_to_2=float(((d_rand >= 0) & (d_rand <= 2)).mean()), + shared_PERIDNUM=int(len(set(pn.PERIDNUM) & set(po.PERIDNUM))), + share_of_2025_persons=float(pn.PERIDNUM.isin(set(po.PERIDNUM)).mean()), +) +json.dump(res, open(out_p, "w"), indent=1) +print(json.dumps(res, indent=1)) diff --git a/experiments/us-asec-2025-source/scripts/run_loader.py b/experiments/us-asec-2025-source/scripts/run_loader.py new file mode 100644 index 000000000..cebe267e2 --- /dev/null +++ b/experiments/us-asec-2025-source/scripts/run_loader.py @@ -0,0 +1,94 @@ +"""Run the archived US data package's CensusCPS loader for one income year. + + cd + python run_loader.py INCOME_YEAR OUT.h5 + +Writes to an explicit output path (never the package's storage folder) and +records, beside the output, the SHA-256 of the archive bytes the loader +downloaded, the output's SHA-256, the loader commit and the library versions. +The loader is not modified; ``requests.get`` is wrapped only to hash the +streamed bytes. +""" + +import hashlib +import json +import platform +import subprocess +import sys +from pathlib import Path + + +def _hash_downloads(requests, record: dict) -> None: + original_get = requests.get + + def hashed_get(url, *args, **kwargs): + response = original_get(url, *args, **kwargs) + original_iter = response.iter_content + digest = hashlib.sha256() + size = [0] + + def iter_content(chunk_size=1, *iter_args, **iter_kwargs): + for chunk in original_iter(chunk_size, *iter_args, **iter_kwargs): + digest.update(chunk) + size[0] += len(chunk) + yield chunk + record.update( + url=url, + sha256=digest.hexdigest(), + bytes=size[0], + last_modified=response.headers.get("last-modified"), + etag=response.headers.get("etag"), + ) + + response.iter_content = iter_content + return response + + requests.get = hashed_get + + +def main(year: int, out: Path) -> dict: + import numpy + import pandas + import requests + import tables + + download: dict = {} + _hash_downloads(requests, download) + + from policyengine_core.data import Dataset + from policyengine_us_data.datasets.cps import census_cps + + class Run(census_cps.CensusCPS): + time_period = year + label = f"Census CPS ({year})" + name = f"census_cps_{year}" + file_path = out + data_format = Dataset.TABLES + + out.parent.mkdir(parents=True, exist_ok=True) + if out.exists(): + out.unlink() + Run().generate() + + def git(*args: str) -> str: + return subprocess.check_output(["git", *args], text=True).strip() + + record = dict( + income_year=year, + output=str(out), + output_sha256=hashlib.sha256(out.read_bytes()).hexdigest(), + output_bytes=out.stat().st_size, + download=download, + loader_commit=git("rev-parse", "HEAD"), + loader_dirty=git("status", "--porcelain", "--", "policyengine_us_data"), + python=platform.python_version(), + pandas=pandas.__version__, + tables=tables.__version__, + numpy=numpy.__version__, + ) + (out.parent / f"run_{year}.json").write_text(json.dumps(record, indent=1)) + return record + + +if __name__ == "__main__": + print(json.dumps(main(int(sys.argv[1]), Path(sys.argv[2])), indent=1)) diff --git a/experiments/us-asec-2025-source/scripts/schema_diff.py b/experiments/us-asec-2025-source/scripts/schema_diff.py new file mode 100644 index 000000000..78eb44f35 --- /dev/null +++ b/experiments/us-asec-2025-source/scripts/schema_diff.py @@ -0,0 +1,156 @@ +"""Schema, value-range and coverage diff between two processed CPS ASEC stores. + +usage: schema_diff.py OLD.h5 NEW.h5 OUT.json +Reports, per table: row counts, added/dropped columns, dtype changes, and for every +shared column a value summary (min, max, distinct count, zero share) with flags for +range changes and code-set changes on low-cardinality columns. Also measures the +NOW_* coverage recodes, county identification (GTCO) and H_IDNUM rotation overlap. +""" + +import json +import sys + +import numpy as np +import pandas as pd + +old_p, new_p, out_p = sys.argv[1:4] +LOW_CARD = 60 + + +def summ(s): + if s.dtype.kind in "iufb": + v = s.to_numpy() + return dict( + kind="num", + min=float(np.nanmin(v)), + max=float(np.nanmax(v)), + n_unique=int(s.nunique()), + zero_share=float((v == 0).mean()), + neg_share=float((v < 0).mean()), + ) + return dict( + kind="str", n_unique=int(s.nunique()), sample=[str(x) for x in s.head(3)] + ) + + +res = {} +with pd.HDFStore(old_p, "r") as so, pd.HDFStore(new_p, "r") as sn: + tables = { + k.strip("/"): (so[k], sn[k]) + for k in sorted(set(so.keys()) | set(sn.keys())) + if k in so and k in sn + } + res["keys"] = dict(old=sorted(so.keys()), new=sorted(sn.keys())) +for name, (a, b) in tables.items(): + shared = [c for c in a.columns if c in b.columns] + t = dict( + rows=[len(a), len(b)], + n_columns=[a.shape[1], b.shape[1]], + dropped=sorted(set(a.columns) - set(b.columns)), + added=sorted(set(b.columns) - set(a.columns)), + index_name=[a.index.name, b.index.name], + dtype_changes={ + c: [str(a[c].dtype), str(b[c].dtype)] + for c in shared + if str(a[c].dtype) != str(b[c].dtype) + }, + column_order_same_for_shared=[c for c in b.columns if c in shared] == shared, + ) + range_flags, code_flags = {}, {} + for c in shared: + sa, sb = summ(a[c]), summ(b[c]) + if sa["kind"] == "num" and sb["kind"] == "num": + if sb["min"] < sa["min"] or sb["max"] > sa["max"]: + range_flags[c] = dict( + old=[sa["min"], sa["max"]], new=[sb["min"], sb["max"]] + ) + if max(sa["n_unique"], sb["n_unique"]) <= LOW_CARD: + ca, cb = set(a[c].unique().tolist()), set(b[c].unique().tolist()) + if ca != cb: + code_flags[c] = dict( + new_codes=sorted(cb - ca), gone_codes=sorted(ca - cb) + ) + t["range_widened"] = range_flags + t["code_set_changed"] = code_flags + t["added_summaries"] = {c: summ(b[c]) for c in t["added"]} + t["dropped_summaries_old"] = {c: summ(a[c]) for c in t["dropped"]} + res[name] = t + + +def cov(path): + with pd.HDFStore(path, "r") as s: + p, h = s["person"], s["household"] + w = p["A_FNLWGT"] / 100.0 + u65 = p["A_AGE"] < 65 + now = sorted(c for c in p.columns if c.startswith("NOW_")) + d = {"NOW_columns": now} + for c in now: + v = p[c] + d[c] = dict( + codes=sorted(map(int, v.unique())), + w_yes_u65_m=round(float(w[(v == 1) & u65].sum()) / 1e6, 2), + ) + hw = h["HSUP_WGT"] / 100.0 if "HSUP_WGT" in h else None + geo = {} + for col in [ + "GTCO", + "GTCBSA", + "GTCSA", + "GESTFIPS", + "GTMETSTA", + "GTINDVPC", + "GTCOUNTY", + ]: + if col in h: + geo[col] = dict( + dtype=str(h[col].dtype), + n_unique=int(h[col].nunique()), + nonzero_share_hh=float((h[col] != 0).mean()), + nonzero_share_weighted=float(hw[h[col] != 0].sum() / hw.sum()) + if hw is not None + else None, + ) + if "GTCO" in h: + cty = h.loc[h["GTCO"] != 0, "GESTFIPS"] * 1000 + h.loc[h["GTCO"] != 0, "GTCO"] + geo["distinct_identified_counties"] = int(cty.nunique()) + geo["identified_county_fips"] = sorted(map(int, cty.unique())) + d["geo"] = geo + d["households"] = len(h) + d["persons"] = len(p) + d["H_IDNUM"] = dict( + present="H_IDNUM" in h, + dtype=str(h["H_IDNUM"].dtype) if "H_IDNUM" in h else None, + unique=int(h["H_IDNUM"].nunique()) if "H_IDNUM" in h else None, + ) + return d, h + + +co, ho = cov(old_p) +cn, hn = cov(new_p) +res["coverage"] = dict(old=co, new=cn) +if "H_IDNUM" in ho and "H_IDNUM" in hn: + ids_o, ids_n = set(ho["H_IDNUM"].astype(str)), set(hn["H_IDNUM"].astype(str)) + shared_ids = ids_o & ids_n + hw_n = hn["HSUP_WGT"] / 100.0 + in_old = hn["H_IDNUM"].astype(str).isin(ids_o) + res["rotation"] = dict( + new_households=len(hn), + old_households=len(ho), + shared_H_IDNUM=len(shared_ids), + share_of_new_in_old=float(in_old.mean()), + share_of_new_in_old_weighted=float(hw_n[in_old].sum() / hw_n.sum()), + share_of_old_in_new=float(ho["H_IDNUM"].astype(str).isin(ids_n).mean()), + ) + if "H_MIS" in hn: + res["rotation"]["by_new_H_MIS"] = { + int(k): float(v) for k, v in in_old.groupby(hn["H_MIS"]).mean().items() + } + # same-address check on matched households: state must agree + m = hn[in_old][["H_IDNUM", "GESTFIPS"]].merge( + ho[["H_IDNUM", "GESTFIPS"]], on="H_IDNUM", suffixes=("_new", "_old") + ) + res["rotation"]["matched_state_agreement"] = ( + float((m["GESTFIPS_new"] == m["GESTFIPS_old"]).mean()) if len(m) else None + ) +json.dump(res, open(out_p, "w"), indent=1, default=str) +print("wrote", out_p) diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/asec_census_person_columns.py b/packages/microcosm-build/src/microcosm/build/us_runtime/asec_census_person_columns.py index b0c565040..6517ac631 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/asec_census_person_columns.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/asec_census_person_columns.py @@ -115,8 +115,9 @@ class AsecCensusPersonColumn: ) #: The reviewed restoration set, in the order columns are appended. Domains are -#: the Census codes observed in all three pinned person members (pppub23/24/25, -#: 2026-09-23); ``A_EXPRRP`` has no code 6 in the Census codebook or the data. +#: the Census codes observed in the pinned person members pppub23/24/25 +#: (2026-09-23); pppub26 (income year 2025) falls inside every one of them +#: (2026-09-27). ``A_EXPRRP`` has no code 6 in the Census codebook or the data. ASEC_CENSUS_PERSON_COLUMNS: tuple[AsecCensusPersonColumn, ...] = ( AsecCensusPersonColumn( "NOW_MCAID", diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/asec_sources.py b/packages/microcosm-build/src/microcosm/build/us_runtime/asec_sources.py index 34d9e5ea0..557bf9668 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/asec_sources.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/asec_sources.py @@ -1,12 +1,22 @@ """Pinned processed CPS ASEC inputs for the US support-base build. -The base stage reads three processed ASEC HDF5 files (income years 2022, 2023 -and 2024) through ``--asec-h5 YEAR=PATH``. Until 2026-09-18 they existed only -in one untracked directory on one build machine, and the build recorded their -digests without comparing them to anything. They are now mirrored, unchanged, -to the public Hugging Face dataset repository -``policyengine/microcosm-us-sources``, and this module owns the immutable -coordinates: repository, revision, per-year filename, SHA-256 and byte size. +The base stage reads processed ASEC HDF5 files, one per income year, through +``--asec-h5 YEAR=PATH``. Four are pinned: income years 2022, 2023 and 2024 +(mirrored unchanged on 2026-09-18; until then they existed only in one +untracked directory on one build machine) and income year 2025 (built from the +ASEC 2026 public-use file on 2026-09-27; see +``docs/us-asec-source-pins.md``). They live in the public Hugging Face dataset +repository ``policyengine/microcosm-us-sources``, and this module owns the +immutable coordinates: repository, per-file upload revision, per-year +filename, SHA-256 and byte size. + +Each artifact is addressed at the revision of the upload that added it, so +adding a year never moves the URL an earlier year resolves to. + +:data:`ASEC_DEFAULT_POOL_INCOME_YEARS` is the pool a new build uses: the newest +three pinned income years, derived from :data:`ASEC_SOURCE_ARTIFACTS` rather +than written out. Every pinned year stays selectable, so a historical +income-2022..2024 build remains byte-reproducible. ``fetch_asec_source`` resolves one year's file the way :func:`microcosm.build.us_runtime.sipp_financial_assets.fetch_sipp_2023_financial_asset_donor` @@ -15,33 +25,43 @@ download included — verified against the pinned byte length and SHA-256 before it is returned. -The same digests are recorded, independently, in the hermetic-input evidence -of ``microcosm/build/us/ecps_parity_known_gaps.json``; a test asserts the two -rosters agree so there is one source of truth. +The income-2022..2024 digests are also recorded, independently, in the +hermetic-input evidence of ``microcosm/build/us/ecps_parity_known_gaps.json`` +(the historical Build J inputs); a test asserts the two agree wherever both +name a year. """ from __future__ import annotations import hashlib +from collections.abc import Mapping from dataclasses import dataclass from pathlib import Path from types import MappingProxyType __all__ = [ + "ASEC_DEFAULT_POOL_INCOME_YEARS", + "ASEC_DEFAULT_POOL_SIZE", "ASEC_SOURCE_REPOSITORY_ID", "ASEC_SOURCE_REPOSITORY_TYPE", "ASEC_SOURCE_REVISION", + "ASEC_SOURCE_REVISION_2025", "ASEC_SOURCE_ARTIFACTS", "AsecSourceArtifact", "asec_source_artifact", "fetch_asec_source", + "newest_pinned_income_years", ] ASEC_SOURCE_REPOSITORY_ID = "policyengine/microcosm-us-sources" ASEC_SOURCE_REPOSITORY_TYPE = "dataset" -#: The upload commit of 2026-09-18. Files in this repository are addressed by -#: revision, never by branch, so a later upload cannot move what a build reads. +#: The upload commit of 2026-09-18, which added the income-year 2022, 2023 and +#: 2024 files. Files in this repository are addressed by revision, never by +#: branch, so a later upload cannot move what a build reads. ASEC_SOURCE_REVISION = "78acc83ea8b099a97cb0d658bbed91ea75aae8b0" +#: The upload commit of 2026-09-27, which added the income-year 2025 file and +#: left the three earlier files' bytes unchanged. +ASEC_SOURCE_REVISION_2025 = "efe5e2107b252506ea9b868071967cf73cc410bc" @dataclass(frozen=True) @@ -52,12 +72,14 @@ class AsecSourceArtifact: filename: str sha256: str size_bytes: int + #: The upload commit that added this file. + revision: str = ASEC_SOURCE_REVISION @property def url(self) -> str: return ( f"https://huggingface.co/datasets/{ASEC_SOURCE_REPOSITORY_ID}" - f"/resolve/{ASEC_SOURCE_REVISION}/{self.filename}" + f"/resolve/{self.revision}/{self.filename}" ) @@ -81,9 +103,47 @@ def url(self) -> str: sha256=("ec36604cb735a660b51b0b2f90be27d803b5878f3464fb30d0eacead59c1260d"), size_bytes=323_994_739, ), + 2025: AsecSourceArtifact( + income_year=2025, + filename="census_cps_2025.h5", + sha256=("4c5a32188b6acfbcfbeb9f5d719d3847873887b72ebdbb19bc402c16e0a64e58"), + size_bytes=304_753_967, + revision=ASEC_SOURCE_REVISION_2025, + ), } ) +#: How many income years a new build pools. +ASEC_DEFAULT_POOL_SIZE = 3 + + +def newest_pinned_income_years( + size: int = ASEC_DEFAULT_POOL_SIZE, + artifacts: Mapping[int, AsecSourceArtifact] = ASEC_SOURCE_ARTIFACTS, +) -> tuple[int, ...]: + """Return the ``size`` newest pinned income years, oldest first.""" + + if size < 1: + raise ValueError(f"A pool needs at least one income year, got {size}.") + if size > len(artifacts): + raise ValueError( + f"Cannot pool the {size} newest income years; only " + f"{sorted(artifacts)} are pinned." + ) + return tuple(sorted(artifacts)[-size:]) + + +#: The default pool: income years 2023, 2024 and 2025 (ASEC 2024-2026). +#: Income year 2022 left the default on 2026-09-27 for two reasons. It is the +#: oldest vintage, and its processed file carries only 2 of the 18 ``NOW_*`` +#: at-interview coverage recodes (``NOW_GRP``, ``NOW_MRK``; microcosm #720), +#: so its reported-coverage inputs depend on +#: :mod:`.asec_census_person_columns` restoring reviewed recodes from the +#: Census archive. The 2025 file carries all 18. Income year 2023 has the same +#: gap and stays in the default until a fourth newer year displaces it. 2022 +#: remains pinned and selectable. +ASEC_DEFAULT_POOL_INCOME_YEARS: tuple[int, ...] = newest_pinned_income_years() + def asec_source_artifact(income_year: int) -> AsecSourceArtifact: """Return the pinned coordinates for one ASEC income year, or refuse.""" @@ -166,7 +226,7 @@ def fetch_asec_source( repo_id=ASEC_SOURCE_REPOSITORY_ID, filename=artifact.filename, repo_type=ASEC_SOURCE_REPOSITORY_TYPE, - revision=ASEC_SOURCE_REVISION, + revision=artifact.revision, cache_dir=str(Path(cache_dir).expanduser()) if cache_dir is not None else None, diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py b/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py index c24a819c7..2f3ace509 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py @@ -32,6 +32,8 @@ import numpy as np import pandas as pd +from microcosm.build.us_runtime.asec_sources import ASEC_DEFAULT_POOL_INCOME_YEARS + __all__ = [ "ASEC_EDUCATION_ASSISTANCE_ARCHIVES", "ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS", @@ -88,7 +90,8 @@ class AsecEducationArchive: weighted_total: float -#: One pinned archive per pooled income year. The survey-year file published +#: One pinned archive per pinned income year (a build pools a subset: +#: :data:`.asec_sources.ASEC_DEFAULT_POOL_INCOME_YEARS` by default). The survey-year file published #: the March after each income year carries that income year's person #: universe: row counts equal the pooled cohorts exactly and PERIDNUM #: coverage is 100.0% per year (and only ~33% against any adjacent survey @@ -163,6 +166,28 @@ class AsecEducationArchive: weighted_positive_share=0.023247141039884213, weighted_total=83_164_893_924.73, ), + AsecEducationArchive( + survey_year=2026, + income_year=2025, + zip_url=( + "https://www2.census.gov/programs-surveys/cps/datasets/2026/" + "march/asecpub26csv.zip" + ), + zip_size_bytes=139_103_894, + zip_sha256=( + "fe819d0d2fc4470c282e76c247b8aa8b6054811619730307e9f2fe6186707d93" + ), + member="pppub26.csv", + member_size_bytes=259_887_437, + member_crc32="477a9e19", + member_sha256=( + "f78920699bf671cc211cb720ff8c6274392524a22138aedaf9e3954f91ba4334" + ), + rows=134_729, + positive_rows=2_574, + weighted_positive_share=0.022465627092607597, + weighted_total=84_013_616_601.91, + ), ) } @@ -390,7 +415,7 @@ def _load_one_source( def load_asec_education_assistance_sources( paths: Mapping[int, str | Path] | None = None, *, - income_years: tuple[int, ...] = ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS, + income_years: tuple[int, ...] = ASEC_DEFAULT_POOL_INCOME_YEARS, chunk_size: int = 8 * 1024 * 1024, ) -> pd.DataFrame: """Load and pin-verify the pooled education-assistance sidecar. diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/public_assistance_type_source.py b/packages/microcosm-build/src/microcosm/build/us_runtime/public_assistance_type_source.py index 9fcc81f93..5b2a51c22 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/public_assistance_type_source.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/public_assistance_type_source.py @@ -30,6 +30,7 @@ import numpy as np import pandas as pd +from microcosm.build.us_runtime.asec_sources import ASEC_DEFAULT_POOL_INCOME_YEARS from microcosm.build.us_runtime.education_assistance_source import ( ASEC_EDUCATION_ASSISTANCE_ARCHIVES, AsecEducationArchive, @@ -81,7 +82,7 @@ class AsecPublicAssistanceTypeAudit: paw_positive_tanf_rows: int -#: One exact audit pin per pooled income year, measured from the archives +#: One exact audit pin per pinned income year, measured from the archives #: pinned in :data:`ASEC_EDUCATION_ASSISTANCE_ARCHIVES`. All values are #: integers, so drift detection is exact. ASEC_PUBLIC_ASSISTANCE_TYPE_AUDIT_PINS: dict[int, AsecPublicAssistanceTypeAudit] = { @@ -108,6 +109,13 @@ class AsecPublicAssistanceTypeAudit: paw_positive_rows=710, paw_positive_tanf_rows=421, ), + AsecPublicAssistanceTypeAudit( + income_year=2025, + rows=134_729, + paw_type_counts=(134_163, 317, 223, 26), + paw_positive_rows=566, + paw_positive_tanf_rows=343, + ), ) } @@ -232,7 +240,7 @@ def _load_one_source( def load_asec_public_assistance_type_sources( paths: Mapping[int, str | Path] | None = None, *, - income_years: tuple[int, ...] = ASEC_PUBLIC_ASSISTANCE_TYPE_INCOME_YEARS, + income_years: tuple[int, ...] = ASEC_DEFAULT_POOL_INCOME_YEARS, ) -> pd.DataFrame: """Load and audit-verify the pooled public-assistance-type sidecar. diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/spm_independence_role.py b/packages/microcosm-build/src/microcosm/build/us_runtime/spm_independence_role.py index 0bcd6b39f..d84e10f30 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/spm_independence_role.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/spm_independence_role.py @@ -51,8 +51,8 @@ SourceRuntimeError, run_source_stage, ) +from microcosm.build.us_runtime.asec_sources import ASEC_DEFAULT_POOL_INCOME_YEARS from microcosm.build.us_runtime.education_assistance_source import ( - ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS, fetch_asec_education_assistance_source, ) from microcosm.build.us_runtime.spm_composition import check_spm_composition @@ -163,7 +163,7 @@ def us_spm_independence_role_stage_spec() -> SourceStageSpec: def resolve_asec_spm_role_source_paths( paths: Mapping[int, str | Path] | None, *, - income_years: tuple[int, ...] = ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS, + income_years: tuple[int, ...] = ASEC_DEFAULT_POOL_INCOME_YEARS, ) -> dict[int, Path]: """Return one pinned complete person source path per pooled income year. diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py b/packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py index af384ff96..a7aaf7d2e 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/spm_role_source.py @@ -58,7 +58,7 @@ class AsecSpmRoleSource: csv_sha256=pin.member_sha256, csv_size_bytes=pin.member_size_bytes, persons=pin.rows, - units={2022: 59_181, 2023: 58_711, 2024: 58_147}[year], + units={2022: 59_181, 2023: 58_711, 2024: 58_147, 2025: 55_401}[year], official_archive_url=pin.zip_url, archive_sha256=pin.zip_sha256, member=pin.member, @@ -66,6 +66,11 @@ class AsecSpmRoleSource: for year, pin in ASEC_EDUCATION_ASSISTANCE_ARCHIVES.items() } +#: The income years the certified BuildP parent pooled. They are the default +#: pins of :func:`derive_spm_role_source`, whose lineage is that parent; the +#: registry above also pins later years for new builds. +BUILDP_SPM_ROLE_INCOME_YEARS: tuple[int, ...] = (2022, 2023, 2024) + @dataclass(frozen=True) class SpmRoleSourceResult: @@ -319,7 +324,11 @@ def derive_spm_role_source( """ path = Path(parent_h5) _require(_sha256(path) == expected_parent_sha256, "Parent H5 SHA-256 mismatch.") - pins = ASEC_SPM_ROLE_SOURCES if source_pins is None else source_pins + pins = ( + {year: ASEC_SPM_ROLE_SOURCES[year] for year in BUILDP_SPM_ROLE_INCOME_YEARS} + if source_pins is None + else source_pins + ) parent = pd.read_hdf(path, "person") spm = pd.read_hdf(path, "spm_unit") required = ( diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_asec_census_person_columns.py b/packages/microcosm-build/tests/engine_free/us/test_us_asec_census_person_columns.py index ad12d6bdc..ecb036298 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_asec_census_person_columns.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_asec_census_person_columns.py @@ -355,7 +355,7 @@ def test_default_pins_are_the_shared_census_person_pins(sources): """Without an explicit pin the production table applies (and refuses here).""" member, _, archive, _ = sources - assert set(ASEC_SPM_ROLE_SOURCES) == {2022, 2023, 2024} + assert set(ASEC_SPM_ROLE_SOURCES) == {2022, 2023, 2024, 2025} with pytest.raises(AsecCensusPersonColumnsError, match="archive SHA-256"): restore_asec_census_person_columns( _h5_person(member), income_year=2022, source_path=archive diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py b/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py index 80e404386..cc45f9cd9 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py @@ -1,12 +1,14 @@ """Pinned CPS ASEC source coordinates and the base builder's digest check. ``microcosm.build.us_runtime.asec_sources`` owns the immutable coordinates of -the three processed ASEC inputs. ``tools/build_us_puf_support_base.py ---asec-h5-sha256`` refuses an input whose bytes differ from its declared -digest before any stage runs, and ``tools/fetch_us_asec_sources.py`` resolves -the pinned files and prints those arguments. These tests pin all three, and -pin the module's digests to the hermetic-input evidence already recorded in -``ecps_parity_known_gaps.json`` so there is one roster. +the four processed ASEC inputs and the default pool (the newest three). +``tools/build_us_puf_support_base.py --asec-h5-sha256`` refuses an input whose +bytes differ from its declared digest before any stage runs, and +``tools/fetch_us_asec_sources.py`` resolves the pinned files and prints those +arguments. These tests pin all four, hold the income-2022..2024 coordinates to +their 2026-09-18 values so historical builds stay byte-reproducible, and bind +the module's digests to the hermetic-input evidence already recorded in +``ecps_parity_known_gaps.json`` wherever both name a year. """ from __future__ import annotations @@ -20,20 +22,33 @@ from types import MappingProxyType, SimpleNamespace import pytest +from hypothesis import given +from hypothesis import strategies as st from microcosm.build.us_runtime import asec_sources from microcosm.build.us_runtime.asec_sources import ( + ASEC_DEFAULT_POOL_INCOME_YEARS, + ASEC_DEFAULT_POOL_SIZE, ASEC_SOURCE_ARTIFACTS, ASEC_SOURCE_REPOSITORY_ID, ASEC_SOURCE_REPOSITORY_TYPE, ASEC_SOURCE_REVISION, + ASEC_SOURCE_REVISION_2025, AsecSourceArtifact, asec_source_artifact, fetch_asec_source, + newest_pinned_income_years, +) +from microcosm.build.us_runtime.education_assistance_source import ( + ASEC_EDUCATION_ASSISTANCE_ARCHIVES, ) from microcosm.build.us_runtime.parity_reference import ( ECPS_PARITY_KNOWN_GAPS_RESOURCE, ) +from microcosm.build.us_runtime.public_assistance_type_source import ( + ASEC_PUBLIC_ASSISTANCE_TYPE_AUDIT_PINS, +) +from microcosm.build.us_runtime.spm_role_source import ASEC_SPM_ROLE_SOURCES from test_support.paths import paths_for _TEST_PATHS = paths_for("microcosm-build") @@ -141,8 +156,11 @@ def walk(node: object) -> None: def test_pins_agree_with_the_hermetic_input_evidence() -> None: + # The evidence records the historical Build J inputs (income 2022-2024); + # every year it names must carry exactly the pinned digest. evidence = _hermetic_asec_digests() - assert set(evidence) == set(ASEC_SOURCE_ARTIFACTS) == {2022, 2023, 2024} + assert set(evidence) == {2022, 2023, 2024} + assert set(evidence) <= set(ASEC_SOURCE_ARTIFACTS) for year, digests in evidence.items(): assert digests == {ASEC_SOURCE_ARTIFACTS[year].sha256}, year @@ -150,30 +168,130 @@ def test_pins_agree_with_the_hermetic_input_evidence() -> None: def test_coordinates_are_well_formed_and_immutable() -> None: assert ASEC_SOURCE_REPOSITORY_ID == "policyengine/microcosm-us-sources" assert ASEC_SOURCE_REPOSITORY_TYPE == "dataset" - assert re.fullmatch(r"[0-9a-f]{40}", ASEC_SOURCE_REVISION) + assert set(ASEC_SOURCE_ARTIFACTS) == {2022, 2023, 2024, 2025} for year, artifact in ASEC_SOURCE_ARTIFACTS.items(): assert artifact.income_year == year assert artifact.filename == f"census_cps_{year}.h5" assert re.fullmatch(r"[0-9a-f]{64}", artifact.sha256) + assert re.fullmatch(r"[0-9a-f]{40}", artifact.revision) assert artifact.size_bytes > 0 assert artifact.url == ( "https://huggingface.co/datasets/policyengine/microcosm-us-sources" - f"/resolve/{ASEC_SOURCE_REVISION}/census_cps_{year}.h5" + f"/resolve/{artifact.revision}/census_cps_{year}.h5" ) - assert len({artifact.sha256 for artifact in ASEC_SOURCE_ARTIFACTS.values()}) == 3 + assert len({artifact.sha256 for artifact in ASEC_SOURCE_ARTIFACTS.values()}) == ( + len(ASEC_SOURCE_ARTIFACTS) + ) with pytest.raises(TypeError): - ASEC_SOURCE_ARTIFACTS[2025] = _fixture_artifact(2025) # type: ignore[index] + ASEC_SOURCE_ARTIFACTS[2026] = _fixture_artifact(2026) # type: ignore[index] with pytest.raises(dataclasses.FrozenInstanceError): ASEC_SOURCE_ARTIFACTS[2022].sha256 = _WRONG_SHA256 # type: ignore[misc] +def test_historical_coordinates_are_unchanged() -> None: + # A historical income-2022..2024 build resolves exactly what it resolved + # when the files were first mirrored: same revision, URL, digest and size. + # Adding a year uploads a new revision and never moves these. + assert ASEC_SOURCE_REVISION == "78acc83ea8b099a97cb0d658bbed91ea75aae8b0" + assert { + year: (artifact.revision, artifact.sha256, artifact.size_bytes) + for year, artifact in ASEC_SOURCE_ARTIFACTS.items() + if year <= 2024 + } == { + 2022: ( + ASEC_SOURCE_REVISION, + "7ccca976284bb47815d84460cc4f75a0a65d26d7754ab0a0f417de351b3d474e", + 301_129_278, + ), + 2023: ( + ASEC_SOURCE_REVISION, + "cb57817327799f42b741caed5f9be94d04021c2e6809c1ad7bd0686da5428d88", + 299_036_610, + ), + 2024: ( + ASEC_SOURCE_REVISION, + "ec36604cb735a660b51b0b2f90be27d803b5878f3464fb30d0eacead59c1260d", + 323_994_739, + ), + } + + +def test_income_year_2025_is_pinned_at_its_own_upload() -> None: + artifact = asec_source_artifact(2025) + assert artifact.revision == ASEC_SOURCE_REVISION_2025 + assert ASEC_SOURCE_REVISION_2025 != ASEC_SOURCE_REVISION + assert artifact.sha256 == ( + "4c5a32188b6acfbcfbeb9f5d719d3847873887b72ebdbb19bc402c16e0a64e58" + ) + assert artifact.size_bytes == 304_753_967 + + +def test_default_pool_is_the_newest_three_pinned_years() -> None: + assert ASEC_DEFAULT_POOL_SIZE == 3 + assert ASEC_DEFAULT_POOL_INCOME_YEARS == (2023, 2024, 2025) + assert ASEC_DEFAULT_POOL_INCOME_YEARS == tuple( + sorted(ASEC_SOURCE_ARTIFACTS)[-ASEC_DEFAULT_POOL_SIZE:] + ) + # 2022 leaves the default but stays pinned and resolvable. + assert 2022 not in ASEC_DEFAULT_POOL_INCOME_YEARS + assert asec_source_artifact(2022).filename == "census_cps_2022.h5" + + +@given( + years=st.sets(st.integers(min_value=1990, max_value=2100), min_size=1, max_size=12), + size=st.integers(min_value=-2, max_value=14), +) +def test_newest_pinned_income_years_properties(years: set[int], size: int) -> None: + artifacts = {year: _fixture_artifact(year) for year in years} + if size < 1 or size > len(years): + with pytest.raises(ValueError): + newest_pinned_income_years(size, artifacts) + return + pool = newest_pinned_income_years(size, artifacts) + assert len(pool) == size + assert list(pool) == sorted(pool) + assert set(pool) <= years + # Every pinned year outside the pool is older than every pooled year. + assert all(other < min(pool) for other in years - set(pool)) + # Pinning a newer year shifts the pool by exactly one year. + newer = max(years) + 1 + shifted = newest_pinned_income_years( + size, {**artifacts, newer: _fixture_artifact(newer)} + ) + assert shifted == (*pool[1:], newer) + + +def test_every_pinned_year_is_pinned_in_every_year_keyed_registry() -> None: + # A pooled year needs its processed H5 and its Census survey archive + # (education and PAW_TYP sidecars, SPM role, #720 person columns). If any + # registry lacked a pinned year, a build of that year would refuse + # mid-construction instead of here. + pinned = set(ASEC_SOURCE_ARTIFACTS) + assert set(ASEC_EDUCATION_ASSISTANCE_ARCHIVES) == pinned + assert set(ASEC_SPM_ROLE_SOURCES) == pinned + assert set(ASEC_PUBLIC_ASSISTANCE_TYPE_AUDIT_PINS) == pinned + assert set(ASEC_DEFAULT_POOL_INCOME_YEARS) <= pinned + for year in pinned: + archive = ASEC_EDUCATION_ASSISTANCE_ARCHIVES[year] + assert archive.survey_year == year + 1 + assert archive.member == f"pppub{(year + 1) % 100:02d}.csv" + assert archive.zip_url == ( + "https://www2.census.gov/programs-surveys/cps/datasets/" + f"{year + 1}/march/asecpub{(year + 1) % 100:02d}csv.zip" + ) + assert ASEC_PUBLIC_ASSISTANCE_TYPE_AUDIT_PINS[year].rows == archive.rows + assert ASEC_SPM_ROLE_SOURCES[year].persons == archive.rows + + def test_unknown_year_refuses_before_any_transfer( monkeypatch: pytest.MonkeyPatch, ) -> None: _forbid_download(monkeypatch) with pytest.raises(ValueError, match=r"No pinned ASEC source for income year 2019"): asec_source_artifact(2019) - with pytest.raises(ValueError, match=r"pinned years are \[2022, 2023, 2024\]"): + with pytest.raises( + ValueError, match=r"pinned years are \[2022, 2023, 2024, 2025\]" + ): fetch_asec_source(2019) @@ -233,6 +351,29 @@ def test_fetch_skips_a_cached_file_of_the_wrong_size( assert calls["revision"] == ASEC_SOURCE_REVISION +def test_fetch_downloads_each_year_at_its_own_revision( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + newer = AsecSourceArtifact( + income_year=2025, + filename="census_cps_2025.h5", + sha256=_FIXTURE_SHA256, + size_bytes=len(_FIXTURE_BYTES), + revision=ASEC_SOURCE_REVISION_2025, + ) + monkeypatch.setattr( + asec_sources, + "ASEC_SOURCE_ARTIFACTS", + MappingProxyType({2022: _fixture_artifact(), 2025: newer}), + ) + calls = _fake_download(monkeypatch, _FIXTURE_BYTES, tmp_path) + fetch_asec_source(2025, tmp_path / "hf-cache") + assert calls["revision"] == ASEC_SOURCE_REVISION_2025 + fetch_asec_source(2022, tmp_path / "hf-cache") + assert calls["revision"] == ASEC_SOURCE_REVISION + + def test_fetch_with_a_cache_dir_goes_to_the_pinned_revision( monkeypatch: pytest.MonkeyPatch, home: Path, @@ -533,7 +674,7 @@ def never(*args, **kwargs): ) -def test_fetch_tool_prints_builder_arguments_for_every_pinned_year( +def test_fetch_tool_prints_builder_arguments_for_the_default_pool( monkeypatch: pytest.MonkeyPatch, tmp_path: Path, capsys: pytest.CaptureFixture[str], @@ -547,14 +688,35 @@ def fake_fetch(year: int, cache_dir=None) -> Path: monkeypatch.setattr(tool, "fetch_asec_source", fake_fetch) assert tool.main([]) == 0 - assert seen == [(2022, None), (2023, None), (2024, None)] + assert seen == [(2023, None), (2024, None), (2025, None)] assert capsys.readouterr().out.splitlines() == [ f"--asec-h5 {year}={tmp_path / f'census_cps_{year}.h5'} " f"--asec-h5-sha256 {year}={ASEC_SOURCE_ARTIFACTS[year].sha256}" - for year in (2022, 2023, 2024) + for year in ASEC_DEFAULT_POOL_INCOME_YEARS ] +def test_fetch_tool_resolves_every_pinned_year_on_request( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + capsys: pytest.CaptureFixture[str], +) -> None: + tool = _load_tool_module("fetch_us_asec_sources") + seen: list[int] = [] + + def fake_fetch(year: int, cache_dir=None) -> Path: + seen.append(year) + return tmp_path / f"census_cps_{year}.h5" + + monkeypatch.setattr(tool, "fetch_asec_source", fake_fetch) + assert tool.main(["--all-pinned"]) == 0 + assert seen == [2022, 2023, 2024, 2025] + assert tool.main(["2022", "2023", "2024"]) == 0 + assert seen[4:] == [2022, 2023, 2024] + with pytest.raises(SystemExit): + tool.main(["--all-pinned", "2022"]) + + def test_fetch_tool_takes_years_and_a_cache_dir( monkeypatch: pytest.MonkeyPatch, tmp_path: Path, diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_education_assistance_source.py b/packages/microcosm-build/tests/engine_free/us/test_us_education_assistance_source.py index f821ea2cc..e1e4cd082 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_education_assistance_source.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_education_assistance_source.py @@ -102,7 +102,7 @@ def test_income_years_map_to_next_survey_year() -> None: assert pins.survey_year == income_year + 1 assert str(pins.survey_year) in pins.zip_url assert pins.member == f"pppub{str(pins.survey_year)[2:]}.csv" - assert ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS == (2022, 2023, 2024) + assert ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS == (2022, 2023, 2024, 2025) def test_loader_reads_pinned_zip_and_audits(pinned_archive) -> None: diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_spine_blindness.py b/packages/microcosm-build/tests/engine_free/us/test_us_spine_blindness.py index 2c5a35548..d537a3f64 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_spine_blindness.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_spine_blindness.py @@ -3472,13 +3472,15 @@ def test_pool_build_tool_import_graph_is_source_spine_blind() -> None: for tool in _SPINE_BLIND_BUILD_TOOLS: runtime_graph, missing_modules = _us_runtime_import_graph(tool) - # 73 = main's 70 plus spm_independence_role.py, spm_role_source.py and + # 74 = main's 70 plus spm_independence_role.py, spm_role_source.py and # spm_composition.py, reached because the pool's engine-input - # projection names the SPM role as a required source input (#893). - # All three are classified in _OTHER_US_RUNTIME_MODULES and scanned - # below like every other reached module. - assert len(runtime_graph) == 73, ( - f"{tool.name} must reach the pinned 73-module runtime graph; " + # projection names the SPM role as a required source input (#893), + # plus asec_sources.py, whose default ASEC pool the education sidecar + # loader now defaults to. All four are classified in + # _OTHER_US_RUNTIME_MODULES and scanned below like every other + # reached module. + assert len(runtime_graph) == 74, ( + f"{tool.name} must reach the pinned 74-module runtime graph; " f"reached {len(runtime_graph)}" ) assert not missing_modules, ( diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_spm_independence_role.py b/packages/microcosm-build/tests/engine_free/us/test_us_spm_independence_role.py index cb1738a44..62dcd3350 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_spm_independence_role.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_spm_independence_role.py @@ -97,7 +97,7 @@ def test_resolver_refuses_unpinned_years_and_keeps_explicit_paths( resolve_asec_spm_role_source_paths({1999: explicit}, income_years=(1999,)) with pytest.raises(ValueError, match="have no pinned ASEC SPM role source"): resolve_asec_spm_role_source_paths(None, income_years=(1999,)) - assert set(ASEC_SPM_ROLE_SOURCES) == {2022, 2023, 2024} + assert set(ASEC_SPM_ROLE_SOURCES) == {2022, 2023, 2024, 2025} class TestDerivation: diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_spm_role_source.py b/packages/microcosm-build/tests/engine_free/us/test_us_spm_role_source.py index d1979449c..65cc7f1cd 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_spm_role_source.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_spm_role_source.py @@ -11,6 +11,7 @@ from microcosm.build.us_runtime.spm_role_source import ( _SOURCE_COLUMNS, ASEC_SPM_ROLE_SOURCES, + BUILDP_SPM_ROLE_INCOME_YEARS, EVIDENCE_SPM_ROLE, AsecSpmRoleSource, derive_spm_role_source, @@ -27,12 +28,19 @@ def test_release_contract_pins_match_actual_source_acquisition(): ) from microcosm.data.source_enrichment import CENSUS_ARCHIVE_PINS, CENSUS_PERSON_PINS - assert ( - set(CENSUS_ARCHIVE_PINS) - == set(CENSUS_PERSON_PINS) - == {pin.survey_year for pin in ASEC_EDUCATION_ASSISTANCE_ARCHIVES.values()} - ) - for income_year, pin in ASEC_EDUCATION_ASSISTANCE_ARCHIVES.items(): + # The release contract is bound to its Build P parent, which pooled income + # years 2022-2024 (survey 2023-2025); the registry also pins later years + # for new builds. Every year the contract names must carry the registry's + # exact pins, and every registered year has a role pin. + assert set(CENSUS_ARCHIVE_PINS) == set(CENSUS_PERSON_PINS) == {2023, 2024, 2025} + registered = { + pin.survey_year: (income_year, pin) + for income_year, pin in ASEC_EDUCATION_ASSISTANCE_ARCHIVES.items() + } + assert set(CENSUS_PERSON_PINS) <= set(registered) + assert set(ASEC_SPM_ROLE_SOURCES) == set(ASEC_EDUCATION_ASSISTANCE_ARCHIVES) + for survey_year in CENSUS_PERSON_PINS: + income_year, pin = registered[survey_year] assert CENSUS_PERSON_PINS[pin.survey_year] == pin.member_sha256 assert CENSUS_ARCHIVE_PINS[pin.survey_year] == { "income_year": income_year, @@ -40,6 +48,11 @@ def test_release_contract_pins_match_actual_source_acquisition(): "archive_sha256": pin.zip_sha256, "member": pin.member, } + # derive_spm_role_source's default pins are that parent's years. + assert set(BUILDP_SPM_ROLE_INCOME_YEARS) == { + pin["income_year"] for pin in CENSUS_ARCHIVE_PINS.values() + } + for income_year, pin in ASEC_EDUCATION_ASSISTANCE_ARCHIVES.items(): role = ASEC_SPM_ROLE_SOURCES[income_year] assert role.survey_year == pin.survey_year assert role.csv_sha256 == pin.member_sha256 diff --git a/tools/build_us_spm_role_enrichment.py b/tools/build_us_spm_role_enrichment.py index f7167c64c..626e4add4 100644 --- a/tools/build_us_spm_role_enrichment.py +++ b/tools/build_us_spm_role_enrichment.py @@ -18,6 +18,7 @@ from microcosm.build.us_runtime import education_assistance_source from microcosm.build.us_runtime.spm_role_source import ( ASEC_SPM_ROLE_SOURCES, + BUILDP_SPM_ROLE_INCOME_YEARS, derive_spm_role_source, ) from microcosm.data.contract import ReleaseContractError @@ -295,8 +296,8 @@ def main(argv: Sequence[str] | None = None) -> int: parent_h5=args.parent_h5, parent_release_dir=args.parent_release_dir, source_paths={ - year: args.source_cache / pin.member - for year, pin in ASEC_SPM_ROLE_SOURCES.items() + year: args.source_cache / ASEC_SPM_ROLE_SOURCES[year].member + for year in BUILDP_SPM_ROLE_INCOME_YEARS }, output_dir=args.output_dir, release_id=args.release_id, diff --git a/tools/fetch_us_asec_sources.py b/tools/fetch_us_asec_sources.py index 82cc7ba75..d8e72d5c2 100644 --- a/tools/fetch_us_asec_sources.py +++ b/tools/fetch_us_asec_sources.py @@ -1,14 +1,18 @@ """Resolve the pinned CPS ASEC inputs and print the base builder's arguments. - uv run python tools/fetch_us_asec_sources.py [--cache-dir DIR] [YEAR ...] + uv run python tools/fetch_us_asec_sources.py [--cache-dir DIR] [--all-pinned | YEAR ...] -Each requested income year (default: every pinned year) resolves through -``microcosm.build.us_runtime.asec_sources.fetch_asec_source``: a known local -copy whose byte length and SHA-256 match the pin, else a download of the -pinned Hugging Face revision, verified the same way. One line per year is -printed in the form ``tools/build_us_puf_support_base.py`` takes:: +With no years, the default pool is resolved: +``microcosm.build.us_runtime.asec_sources.ASEC_DEFAULT_POOL_INCOME_YEARS``, the +newest three pinned income years. ``--all-pinned`` resolves every pinned year, +and explicit years resolve exactly those, so income year 2022 stays available +for byte-reproducible historical builds. Each year resolves through +``fetch_asec_source``: a known local copy whose byte length and SHA-256 match +the pin, else a download of that file's pinned Hugging Face revision, verified +the same way. One line per year is printed in the form +``tools/build_us_puf_support_base.py`` takes:: - --asec-h5 2022=/path/census_cps_2022.h5 --asec-h5-sha256 2022=7ccca976... + --asec-h5 2025=/path/census_cps_2025.h5 --asec-h5-sha256 2025=4c5a3218... so the output pastes into a base-build invocation and the builder re-verifies the bytes before any stage runs. @@ -21,6 +25,7 @@ from pathlib import Path from microcosm.build.us_runtime.asec_sources import ( + ASEC_DEFAULT_POOL_INCOME_YEARS, ASEC_SOURCE_ARTIFACTS, asec_source_artifact, fetch_asec_source, @@ -36,7 +41,15 @@ def _parse_args(argv: list[str] | None) -> argparse.Namespace: "years", nargs="*", type=int, - help="Income years to resolve; the default is every pinned year.", + help=( + "Income years to resolve; the default is the default pool " + f"{list(ASEC_DEFAULT_POOL_INCOME_YEARS)}." + ), + ) + parser.add_argument( + "--all-pinned", + action="store_true", + help=f"Resolve every pinned year {sorted(ASEC_SOURCE_ARTIFACTS)}.", ) parser.add_argument( "--cache-dir", @@ -47,12 +60,18 @@ def _parse_args(argv: list[str] | None) -> argparse.Namespace: "local convenience copies are not consulted." ), ) - return parser.parse_args(argv) + args = parser.parse_args(argv) + if args.all_pinned and args.years: + parser.error("--all-pinned takes no explicit years.") + return args def main(argv: list[str] | None = None) -> int: args = _parse_args(argv) - years = args.years or sorted(ASEC_SOURCE_ARTIFACTS) + if args.all_pinned: + years = sorted(ASEC_SOURCE_ARTIFACTS) + else: + years = args.years or list(ASEC_DEFAULT_POOL_INCOME_YEARS) try: for year in years: artifact = asec_source_artifact(year) From e96165359b690cce7745bbec3854ca003cd91ae7 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Mon, 28 Sep 2026 00:43:43 -0400 Subject: [PATCH 2/5] Address review: loader-default test, arrival count source, wording Pin that the three year-keyed loaders default to the default pool; add the PEINUSYR top-code counts (1,121 income-2025 persons with code 29) to the measurement receipt; say "Build P vintages" where the SPM-role bands and the 2026-09-23 review describe 2022-2024; rewrap the registry comment; add the source-enrichment producer-identity operator note. Co-Authored-By: Claude Opus 5.5 --- docs/us-asec-source-pins.md | 13 ++++++++++ docs/us-spm-role-stage.md | 4 ++-- .../receipts/measurements_2025.json | 14 +++++++++-- .../scripts/measure_pool.py | 4 ++++ .../us_runtime/education_assistance_source.py | 14 +++++------ .../build/us_runtime/spm_independence_role.py | 3 ++- .../engine_free/us/test_us_asec_sources.py | 24 +++++++++++++++++++ 7 files changed, 64 insertions(+), 12 deletions(-) diff --git a/docs/us-asec-source-pins.md b/docs/us-asec-source-pins.md index 3a661bc95..5d4bd07af 100644 --- a/docs/us-asec-source-pins.md +++ b/docs/us-asec-source-pins.md @@ -227,3 +227,16 @@ README). Things that still name income year 2022 or the old pool: 28 → 2023 stays inside both intervals. 29 → 2024 clamps the 1,121 most recent arrivals of 2025 to the 2024 target year (zero years in the US). + +## Operator note: source-enrichment producer identity + +`education_assistance_source.py`, `public_assistance_type_source.py`, +`spm_role_source.py` and `tools/build_us_spm_role_enrichment.py` are in the +source-enrichment producer inventories (`source_enrichment.PRODUCER_SOURCE_FILES` +and `RECEIPT_QUALIFICATION_SOURCE_FILES`). Publication checks each file's bytes +in the current checkout against the build's recorded hashes. The income-2025 +registry entries changed those bytes on 27 September 2026, so any SPM-role or +reported-receipt candidate built earlier must be certified and published from +a checkout of its own recorded producer commit, not from `main`. The published +`populace-us-2024-spm-20260915` and `populace-us-2024-spm-receipts-20260923` +are unaffected. diff --git a/docs/us-spm-role-stage.md b/docs/us-spm-role-stage.md index a552cbbed..2c6e443dc 100644 --- a/docs/us-spm-role-stage.md +++ b/docs/us-spm-role-stage.md @@ -140,7 +140,7 @@ Everything after the read (person count, key, missing-value, integer, role, reconciliation and unit-count checks) is shared and unchanged. In an independent review on 2026-09-23 (a local run, not a committed receipt), the archive and the extracted CSV gave equal frames and equal source checks for -each of the three pinned vintages: 146,133 / 144,265 / 142,125 persons and +each of the three Build P vintages (income years 2022-2024): 146,133 / 144,265 / 142,125 persons and 59,181 / 58,711 / 58,147 units. `test_us_spm_role_source.py` pins, on a synthetic fixture, that the two forms derive identical roles, evidence and provenance, and that an unpinned archive, an archive without the pinned @@ -195,7 +195,7 @@ person/age/unit/role fingerprint agree with the current frame; the column is present, Boolean-valued, and not degenerate; `check_spm_composition` on the frame reports zero units without a classified adult with `role_source == "source_column"`; and two plausibility bands taken -from the three pinned vintages — the role share among all persons (measured +from the three Build P vintages (income years 2022-2024) — the role share among all persons (measured 0.601–0.603 per vintage on the phase-2 base, 0.567 on Build P; band 0.40–0.75) and among 15-to-17-year-olds (measured 1.49 %–1.73 % per vintage; band 0.3 %–6 %). diff --git a/experiments/us-asec-2025-source/receipts/measurements_2025.json b/experiments/us-asec-2025-source/receipts/measurements_2025.json index 2a754f099..3e901ac20 100644 --- a/experiments/us-asec-2025-source/receipts/measurements_2025.json +++ b/experiments/us-asec-2025-source/receipts/measurements_2025.json @@ -22,7 +22,12 @@ 48439, 53033, 53061 - ] + ], + "peinusyr_top_code_persons": { + "27": 1277, + "28": 2749, + "29": 0 + } }, "income_2025": { "survey_year": 2026, @@ -58,7 +63,12 @@ 45085, 54019, 54081 - ] + ], + "peinusyr_top_code_persons": { + "27": 1047, + "28": 1741, + "29": 1121 + } }, "rotation_income_2025_vs_2024": { "shared_H_IDNUM": 16601, diff --git a/experiments/us-asec-2025-source/scripts/measure_pool.py b/experiments/us-asec-2025-source/scripts/measure_pool.py index 4e75ef338..4c982d3d5 100644 --- a/experiments/us-asec-2025-source/scripts/measure_pool.py +++ b/experiments/us-asec-2025-source/scripts/measure_pool.py @@ -49,6 +49,10 @@ def load(p): coded_not_listed=len(coded_set - listed), listed_not_coded=len(listed - coded_set), coded_not_listed_fips=sorted(coded_set - listed), + # PEINUSYR top codes, whose intervals Census re-bins every year. + peinusyr_top_code_persons={ + int(code): int((per["PEINUSYR"] == code).sum()) for code in (27, 28, 29) + }, ) ho, po = load(old_p) diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py b/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py index 2f3ace509..eea8a97de 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py @@ -90,13 +90,13 @@ class AsecEducationArchive: weighted_total: float -#: One pinned archive per pinned income year (a build pools a subset: -#: :data:`.asec_sources.ASEC_DEFAULT_POOL_INCOME_YEARS` by default). The survey-year file published -#: the March after each income year carries that income year's person -#: universe: row counts equal the pooled cohorts exactly and PERIDNUM -#: coverage is 100.0% per year (and only ~33% against any adjacent survey -#: year, the CPS rotation-group overlap — pinning the wrong vintage fails the -#: full-coverage join loudly). +#: One pinned archive per pinned income year; a build pools a subset +#: (:data:`.asec_sources.ASEC_DEFAULT_POOL_INCOME_YEARS` by default). The +#: survey-year file published the March after each income year carries that +#: income year's person universe: row counts equal the pooled cohorts exactly +#: and PERIDNUM coverage is 100.0% per year (and only ~33% against any adjacent +#: survey year, the CPS rotation-group overlap — pinning the wrong vintage +#: fails the full-coverage join loudly). ASEC_EDUCATION_ASSISTANCE_ARCHIVES: dict[int, AsecEducationArchive] = { archive.income_year: archive for archive in ( diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/spm_independence_role.py b/packages/microcosm-build/src/microcosm/build/us_runtime/spm_independence_role.py index d84e10f30..f0b8497b9 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/spm_independence_role.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/spm_independence_role.py @@ -130,7 +130,8 @@ _SPM_UNIT_ID = "spm_unit_id" _DERIVE_PARAMETER_KEYS = frozenset() -#: Plausibility bands, measured on the three pinned vintages: the role share +#: Plausibility bands, measured on the three Build P vintages (income years +#: 2022-2024; the 2023-2025 default pool measured 0.628 and 1.81 %): the role share #: among all persons (the SPM head plus spouses; 0.601-0.603 per vintage on the #: phase-2 base, 0.567 on Build P) and among 15-to-17-year-olds (1.49 %-1.73 % #: per vintage). See ``docs/us-spm-role-stage.md`` §2 and the committed diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py b/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py index cc45f9cd9..ba7824585 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py @@ -743,3 +743,27 @@ def test_fetch_tool_refuses_an_unpinned_year(monkeypatch: pytest.MonkeyPatch) -> tool = _load_tool_module("fetch_us_asec_sources") with pytest.raises(SystemExit, match=r"No pinned ASEC source for income year 2019"): tool.main(["2019"]) + + +def test_year_keyed_loaders_default_to_the_default_pool() -> None: + # A bare call loads the default pool, never every registered year: with + # 2022 still pinned, "every year" would silently mean a four-year pool. + import inspect + + from microcosm.build.us_runtime.education_assistance_source import ( + load_asec_education_assistance_sources, + ) + from microcosm.build.us_runtime.public_assistance_type_source import ( + load_asec_public_assistance_type_sources, + ) + from microcosm.build.us_runtime.spm_independence_role import ( + resolve_asec_spm_role_source_paths, + ) + + for loader in ( + load_asec_education_assistance_sources, + load_asec_public_assistance_type_sources, + resolve_asec_spm_role_source_paths, + ): + default = inspect.signature(loader).parameters["income_years"].default + assert default == ASEC_DEFAULT_POOL_INCOME_YEARS, loader.__name__ From 6ef38093905d15cbdabdbd4a2e8e71c8a1172949 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Mon, 28 Sep 2026 08:36:34 -0400 Subject: [PATCH 3/5] Tie the pipeline smoke receipt to e96165359 Rerun source_construction and pre_clone_enrichment on the default pool at e96165359 with a clean tree and record the commit. Both stage checkpoints are byte-identical to the earlier working-tree run. Co-Authored-By: Claude Opus 5.5 --- experiments/us-asec-2025-source/README.md | 14 ++++--- .../receipts/pipeline_smoke_2023_2025.json | 37 ++++++++++++------- 2 files changed, 32 insertions(+), 19 deletions(-) diff --git a/experiments/us-asec-2025-source/README.md b/experiments/us-asec-2025-source/README.md index 92c707580..bb996d0f6 100644 --- a/experiments/us-asec-2025-source/README.md +++ b/experiments/us-asec-2025-source/README.md @@ -174,11 +174,15 @@ changed dtype. ## Pipeline smoke on the default pool [`receipts/pipeline_smoke_2023_2025.json`](receipts/pipeline_smoke_2023_2025.json). -The run used `tools/build_us_puf_support_base.py` with `--stage -source_construction`, then `--stage pre_clone_enrichment`. It passed -`--asec-h5` and `--asec-h5-sha256` for 2023, 2024 and 2025, the three local -Census archives, and `PYTHONHASHSEED=0`. Both stages succeeded: 64 s and 11.0 -GB peak, then 38 s and 13.6 GB. +The run used `tools/build_us_puf_support_base.py` at commit `e96165359` (clean +tree) with `--stage source_construction`, then `--stage pre_clone_enrichment`. +It passed `--asec-h5` and `--asec-h5-sha256` for 2023, 2024 and 2025, the +three local Census archives, and `PYTHONHASHSEED=0`. Both stages succeeded: 63 +s and 10.9 GB peak, then 60 s and 13.5 GB. + +An earlier run from the uncommitted working tree gave byte-identical checkpoints +for both stages and identical signal values. The one differing field was the +digest of the SPM-role stage's intermediate projection file. - **Pins.** All three `--asec-h5-sha256` pins were verified before either stage ran; the 2025 pin now has a canonical value to match. diff --git a/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json b/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json index f2a4d3c53..726819147 100644 --- a/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json +++ b/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json @@ -1,5 +1,9 @@ { - "purpose": "source_construction and pre_clone_enrichment of tools/build_us_puf_support_base.py on the default pool, fully pinned, 2026-09-27", + "purpose": "source_construction and pre_clone_enrichment of tools/build_us_puf_support_base.py on the default pool, fully pinned", + "microcosm_commit": "e96165359b690cce7745bbec3854ca003cd91ae7", + "microcosm_worktree_dirty": false, + "run_date": "2026-09-28", + "environment": "PYTHONHASHSEED=0; uv sync --all-packages --locked --extra us", "run_config": { "asec_h5": [ "2023=~/PolicyEngine/policyengine-us-data/policyengine_us_data/storage/census_cps_2023.h5", @@ -22,28 +26,32 @@ }, "stage_profile": { "pre_clone_enrichment": { - "entry_rss_bytes": 1970290688, + "entry_rss_bytes": 2030764032, "error": null, - "exit_rss_bytes": 10873683968, - "peak_rss_bytes": 13574078464, - "rss_sample_count": 129, + "exit_rss_bytes": 10943578112, + "peak_rss_bytes": 13504036864, + "rss_sample_count": 210, "sampling_error": null, "stage": "pre_clone_enrichment", "status": "succeeded", - "wall_seconds": 38.22470487502869 + "wall_seconds": 60.14432191697415 }, "source_construction": { - "entry_rss_bytes": 1971683328, + "entry_rss_bytes": 2030878720, "error": null, - "exit_rss_bytes": 9930113024, - "peak_rss_bytes": 10964992000, - "rss_sample_count": 202, + "exit_rss_bytes": 10335617024, + "peak_rss_bytes": 10880221184, + "rss_sample_count": 197, "sampling_error": null, "stage": "source_construction", "status": "succeeded", - "wall_seconds": 63.66323954198742 + "wall_seconds": 62.91089479095535 } }, + "checkpoint_sha256": { + "source_construction": "e0776a293b40cedb2f3dc865d4bf55a45cc9bfe31f0a6ad5e583049bda3f78d0", + "pre_clone_enrichment": "ae91435fce01006522c43ec3b327712f48005a314efbfd6881c17d1c9effa31d" + }, "row_counts": [ { "rows": 421119, @@ -593,7 +601,7 @@ ], "classification_changed_units_vs_age_only": 304, "complete_source_membership_units": 172259, - "dataset_sha256": "d0e362ece441a25bf62ee9c21fb2fed22fbf29a100acb875f2c8ca967aab6d14", + "dataset_sha256": "3f75dcec42791d923a8611b1c8c3a4a341d7198e05052b83900448b1ff1edeaa", "evidence_column": "is_spm_independence_role", "evidence_status": "Inference from documented primitive relationships, reconciled against every SPM count in pinned complete Census ASEC sources; Census production program not retrieved.", "frame_projection_columns": [ @@ -617,7 +625,7 @@ "A_SPOUSE", "PECOHAB" ], - "frame_projection_sha256": "d0e362ece441a25bf62ee9c21fb2fed22fbf29a100acb875f2c8ca967aab6d14", + "frame_projection_sha256": "3f75dcec42791d923a8611b1c8c3a4a341d7198e05052b83900448b1ff1edeaa", "independent_minor_persons": 306, "minor_only_units_resolved": 108, "native_spm_units": 172259, @@ -796,5 +804,6 @@ "failures": [], "passed": true } - } + }, + "reproducibility_note": "An earlier run of the same two stages on 2026-09-27, from the working tree before these commits, produced byte-identical stage checkpoints (source_construction and pre_clone_enrichment checkpoint_sha256 equal) and identical signal values. The only differing field is spm_independence_role_signal.details.derivation.dataset_sha256 / frame_projection_sha256, the digest of the stage's intermediate projection file." } \ No newline at end of file From ace0a71a01cd70dffc4a4d7f02e25278f74bfba0 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Mon, 28 Sep 2026 08:47:10 -0400 Subject: [PATCH 4/5] Record the earlier smoke run's checkpoint digests and fix doc dates The reproducibility claim now rests on the earlier run's checkpoint digests recorded in the receipt; the smoke date and the producer-identity note name the commit and PR rather than a day; rewrap two long doc lines. Co-Authored-By: Claude Opus 5.5 --- docs/us-asec-source-pins.md | 6 +++--- docs/us-spm-role-stage.md | 6 ++++-- experiments/us-asec-2025-source/README.md | 5 +++-- .../receipts/pipeline_smoke_2023_2025.json | 10 +++++++++- 4 files changed, 19 insertions(+), 8 deletions(-) diff --git a/docs/us-asec-source-pins.md b/docs/us-asec-source-pins.md index 5d4bd07af..f2c7eccf7 100644 --- a/docs/us-asec-source-pins.md +++ b/docs/us-asec-source-pins.md @@ -196,8 +196,8 @@ and the US release build rule is where it belongs. ## What a 2022-free pool still reads The `source_construction` and `pre_clone_enrichment` stages ran on the real -2023/2024/2025 files, fully pinned, on 27 September 2026 (see the experiment -README). Things that still name income year 2022 or the old pool: +2023/2024/2025 files, fully pinned, at commit `e96165359` on 28 September 2026 +(see the experiment README). Things that still name income year 2022 or the old pool: - **The LKWEEKS sidecar.** The builder loads the income-2022 sidecar `asecpub23csv.zip` in every mode. It fetches the file when @@ -235,7 +235,7 @@ README). Things that still name income year 2022 or the old pool: source-enrichment producer inventories (`source_enrichment.PRODUCER_SOURCE_FILES` and `RECEIPT_QUALIFICATION_SOURCE_FILES`). Publication checks each file's bytes in the current checkout against the build's recorded hashes. The income-2025 -registry entries changed those bytes on 27 September 2026, so any SPM-role or +registry entries in the PR that pinned it changed those bytes, so any SPM-role or reported-receipt candidate built earlier must be certified and published from a checkout of its own recorded producer commit, not from `main`. The published `populace-us-2024-spm-20260915` and `populace-us-2024-spm-receipts-20260923` diff --git a/docs/us-spm-role-stage.md b/docs/us-spm-role-stage.md index 2c6e443dc..a10ae91d2 100644 --- a/docs/us-spm-role-stage.md +++ b/docs/us-spm-role-stage.md @@ -140,7 +140,8 @@ Everything after the read (person count, key, missing-value, integer, role, reconciliation and unit-count checks) is shared and unchanged. In an independent review on 2026-09-23 (a local run, not a committed receipt), the archive and the extracted CSV gave equal frames and equal source checks for -each of the three Build P vintages (income years 2022-2024): 146,133 / 144,265 / 142,125 persons and +each of the three Build P vintages (income years 2022-2024): +146,133 / 144,265 / 142,125 persons and 59,181 / 58,711 / 58,147 units. `test_us_spm_role_source.py` pins, on a synthetic fixture, that the two forms derive identical roles, evidence and provenance, and that an unpinned archive, an archive without the pinned @@ -195,7 +196,8 @@ person/age/unit/role fingerprint agree with the current frame; the column is present, Boolean-valued, and not degenerate; `check_spm_composition` on the frame reports zero units without a classified adult with `role_source == "source_column"`; and two plausibility bands taken -from the three Build P vintages (income years 2022-2024) — the role share among all persons (measured +from the three Build P vintages (income years 2022-2024) — the role +share among all persons (measured 0.601–0.603 per vintage on the phase-2 base, 0.567 on Build P; band 0.40–0.75) and among 15-to-17-year-olds (measured 1.49 %–1.73 % per vintage; band 0.3 %–6 %). diff --git a/experiments/us-asec-2025-source/README.md b/experiments/us-asec-2025-source/README.md index bb996d0f6..a2d926d23 100644 --- a/experiments/us-asec-2025-source/README.md +++ b/experiments/us-asec-2025-source/README.md @@ -180,8 +180,9 @@ It passed `--asec-h5` and `--asec-h5-sha256` for 2023, 2024 and 2025, the three local Census archives, and `PYTHONHASHSEED=0`. Both stages succeeded: 63 s and 10.9 GB peak, then 60 s and 13.5 GB. -An earlier run from the uncommitted working tree gave byte-identical checkpoints -for both stages and identical signal values. The one differing field was the +An earlier run from the uncommitted working tree, on 27 September, gave the same +checkpoint digests for both stages (recorded in the receipt's `earlier_run`) +and identical signal values. The one differing field was the digest of the SPM-role stage's intermediate projection file. - **Pins.** All three `--asec-h5-sha256` pins were verified before either stage diff --git a/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json b/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json index 726819147..d6987312b 100644 --- a/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json +++ b/experiments/us-asec-2025-source/receipts/pipeline_smoke_2023_2025.json @@ -805,5 +805,13 @@ "passed": true } }, - "reproducibility_note": "An earlier run of the same two stages on 2026-09-27, from the working tree before these commits, produced byte-identical stage checkpoints (source_construction and pre_clone_enrichment checkpoint_sha256 equal) and identical signal values. The only differing field is spm_independence_role_signal.details.derivation.dataset_sha256 / frame_projection_sha256, the digest of the stage's intermediate projection file." + "reproducibility_note": "The earlier run (earlier_run) produced the same checkpoint_sha256 for both stages and identical signal values; the only differing field is spm_independence_role_signal.details.derivation.dataset_sha256 / frame_projection_sha256, the digest of the stage's intermediate projection file.", + "earlier_run": { + "date": "2026-09-27", + "code": "working tree before e518fb86 was committed", + "checkpoint_sha256": { + "source_construction": "e0776a293b40cedb2f3dc865d4bf55a45cc9bfe31f0a6ad5e583049bda3f78d0", + "pre_clone_enrichment": "ae91435fce01006522c43ec3b327712f48005a314efbfd6881c17d1c9effa31d" + } + } } \ No newline at end of file From a677312dda77385daaf69701c79450cffcd440b7 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Mon, 28 Sep 2026 08:57:14 -0400 Subject: [PATCH 5/5] Keep the Build P proof scripts on their own 2022-2024 sources experiments/spm_role_stage_proof.py and spm_role_stage_wrapper_proof.py iterated every year in ASEC_SPM_ROLE_SOURCES, which now includes 2025: the wrapper raised KeyError('2025') on the base receipt and the proof passed a fourth CSV path that derive_spm_role_source refuses. Both now iterate BUILDP_SPM_ROLE_INCOME_YEARS, as tools/build_us_spm_role_enrichment.py does. Also pin the historical URLs, assert PAW_TYP internal consistency per registered year, and fix the stale three-inputs wording. Found by the independent review from the LKWEEKS-sidecar session. Co-Authored-By: Claude Opus 5.5 --- docs/us-spm-role-stage.md | 3 ++- experiments/spm_role_stage_proof.py | 8 +++++++- experiments/spm_role_stage_wrapper_proof.py | 4 +++- .../build/us_runtime/asec_census_person_columns.py | 8 ++++---- .../tests/engine_free/us/test_us_asec_sources.py | 10 +++++++++- 5 files changed, 25 insertions(+), 8 deletions(-) diff --git a/docs/us-spm-role-stage.md b/docs/us-spm-role-stage.md index a10ae91d2..52a22c93e 100644 --- a/docs/us-spm-role-stage.md +++ b/docs/us-spm-role-stage.md @@ -41,7 +41,8 @@ The pool-global `SPM_ID` pattern the brief points at is exactly what instead reconstructs source membership through `(source_year, PERIDNUM)`. **Where they do exist:** the pinned complete Census ASEC person CSVs -(`pppub23.csv`, `pppub24.csv`, `pppub25.csv`, income years 2022-2024), pinned in +(`pppub23.csv`, `pppub24.csv`, `pppub25.csv`, income years 2022-2024; since +2026-09-28 also `pppub26.csv`, income year 2025), pinned in `education_assistance_source.ASEC_EDUCATION_ASSISTANCE_ARCHIVES` and re-exposed as `spm_role_source.ASEC_SPM_ROLE_SOURCES`, cached by `fetch_asec_education_assistance_source` under diff --git a/experiments/spm_role_stage_proof.py b/experiments/spm_role_stage_proof.py index 96930f473..dda823004 100644 --- a/experiments/spm_role_stage_proof.py +++ b/experiments/spm_role_stage_proof.py @@ -48,6 +48,7 @@ from microcosm.build.us_runtime.release_gate_preflight import check_spm_composition from microcosm.build.us_runtime.spm_role_source import ( ASEC_SPM_ROLE_SOURCES, + BUILDP_SPM_ROLE_INCOME_YEARS, NATIVE_SPM_ROLE, derive_spm_role_source, ) @@ -120,7 +121,12 @@ def _check(frame: Frame) -> dict: def _source_paths(cache: Path) -> dict[int, Path]: - return {year: cache / pin.member for year, pin in ASEC_SPM_ROLE_SOURCES.items()} + # The phase-2 base and Build P pooled income years 2022-2024; the registry + # also pins later years for new builds. + return { + year: cache / ASEC_SPM_ROLE_SOURCES[year].member + for year in BUILDP_SPM_ROLE_INCOME_YEARS + } def _environment() -> dict: diff --git a/experiments/spm_role_stage_wrapper_proof.py b/experiments/spm_role_stage_wrapper_proof.py index 3f0f86a1a..e02c8319c 100644 --- a/experiments/spm_role_stage_wrapper_proof.py +++ b/experiments/spm_role_stage_wrapper_proof.py @@ -250,6 +250,7 @@ def prove(mode: str, state: dict[str, str], source: dict[str, Any]) -> dict[str, ) from microcosm.build.us_runtime.spm_role_source import ( ASEC_SPM_ROLE_SOURCES, + BUILDP_SPM_ROLE_INCOME_YEARS, EVIDENCE_SPM_ROLE, NATIVE_SPM_ROLE, derive_spm_role_source, @@ -268,7 +269,8 @@ def prove(mode: str, state: dict[str, str], source: dict[str, Any]) -> dict[str, _require(input_sha == PARENT_DATASET_SHA256, "buildp_parent_pin_drift") source_paths = {} inputs = {"population": (input_path, input_sha)} - for year, pin in ASEC_SPM_ROLE_SOURCES.items(): + for year in BUILDP_SPM_ROLE_INCOME_YEARS: + pin = ASEC_SPM_ROLE_SOURCES[year] recorded = base_receipt["source_csvs"][str(year)] _require(recorded["pinned_sha256"] == pin.csv_sha256, "source_pin_drift") _require(recorded["size_bytes"] == pin.csv_size_bytes, "source_size_pin_drift") diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/asec_census_person_columns.py b/packages/microcosm-build/src/microcosm/build/us_runtime/asec_census_person_columns.py index 6517ac631..665789e6b 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/asec_census_person_columns.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/asec_census_person_columns.py @@ -1,9 +1,9 @@ """Restore the reviewed Census person columns the pinned ASEC H5 inputs lack. -Microcosm #720. The base pools three processed ASEC HDF5 inputs (pins in -:mod:`.asec_sources`: ``census_cps_2022.h5`` 7ccca976…, ``census_cps_2023.h5`` -cb578173…, ``census_cps_2024.h5`` ec36604c…). The income-year 2022 and 2023 -files were extracted with an older column list: they carry 2 of the 18 +Microcosm #720. Of the processed ASEC HDF5 inputs pinned in +:mod:`.asec_sources`, the income-year 2022 and 2023 files +(``census_cps_2022.h5`` 7ccca976…, ``census_cps_2023.h5`` cb578173…) were +extracted with an older column list: they carry 2 of the 18 ``NOW_*`` at-interview coverage recodes (``NOW_GRP``, ``NOW_MRK``) and lack ``A_EXPRRP``, ``PTOTVAL``, ``A_ENRLW`` and ``A_FTPT``, which the 2024 file carries. :func:`.asec_pool.pool_asec_sources` concatenates per-year person diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py b/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py index ba7824585..e2169822d 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_asec_sources.py @@ -193,6 +193,11 @@ def test_historical_coordinates_are_unchanged() -> None: # when the files were first mirrored: same revision, URL, digest and size. # Adding a year uploads a new revision and never moves these. assert ASEC_SOURCE_REVISION == "78acc83ea8b099a97cb0d658bbed91ea75aae8b0" + for year in (2022, 2023, 2024): + assert ASEC_SOURCE_ARTIFACTS[year].url == ( + "https://huggingface.co/datasets/policyengine/microcosm-us-sources" + f"/resolve/78acc83ea8b099a97cb0d658bbed91ea75aae8b0/census_cps_{year}.h5" + ) assert { year: (artifact.revision, artifact.sha256, artifact.size_bytes) for year, artifact in ASEC_SOURCE_ARTIFACTS.items() @@ -279,7 +284,10 @@ def test_every_pinned_year_is_pinned_in_every_year_keyed_registry() -> None: "https://www2.census.gov/programs-surveys/cps/datasets/" f"{year + 1}/march/asecpub{(year + 1) % 100:02d}csv.zip" ) - assert ASEC_PUBLIC_ASSISTANCE_TYPE_AUDIT_PINS[year].rows == archive.rows + audit = ASEC_PUBLIC_ASSISTANCE_TYPE_AUDIT_PINS[year] + assert audit.rows == archive.rows + assert sum(audit.paw_type_counts) == audit.rows + assert audit.paw_positive_tanf_rows <= audit.paw_positive_rows assert ASEC_SPM_ROLE_SOURCES[year].persons == archive.rows