diff --git a/chronicle/source_package.py b/chronicle/source_package.py index d2c76ef9..7e7888ce 100644 --- a/chronicle/source_package.py +++ b/chronicle/source_package.py @@ -358,6 +358,7 @@ ), "soi-table-4-3": Path("irs_soi/table_4_3"), "soi-state-2022": Path("irs_soi/state_2022"), + "soi-state-2023": Path("irs_soi/state_2023"), "soi-county-2022": Path("irs_soi/county_2022"), "soi-congressional-district-2022": Path("irs_soi/congressional_district_2022"), "soi-historic-table-2": Path("irs_soi/historic_table_2"), @@ -374,7 +375,11 @@ "soi-ira-traditional-contributions-2022": Path( "irs_soi/ira_traditional_contributions_2022" ), + "soi-ira-traditional-contributions-2023": Path( + "irs_soi/ira_traditional_contributions_2023" + ), "soi-ira-roth-contributions-2022": Path("irs_soi/ira_roth_contributions_2022"), + "soi-ira-roth-contributions-2023": Path("irs_soi/ira_roth_contributions_2023"), "ssa-annual-statistical-supplement-2025": Path( "ssa/annual_statistical_supplement_2025" ), diff --git a/chronicle/sources/cells.py b/chronicle/sources/cells.py index f4fcfe99..a94119f3 100644 --- a/chronicle/sources/cells.py +++ b/chronicle/sources/cells.py @@ -18,6 +18,7 @@ from zipfile import ZipFile import openpyxl +from openpyxl.cell.cell import MergedCell import xlrd from chronicle.epoch import EMIT_EPOCH, HASH_DOMAINS, Epoch, hash_domain @@ -172,8 +173,9 @@ def source_cells_from_xlsx( if sheets and sheet.title not in sheets: continue formula_sheet = formula_workbook[sheet.title] - for row_index in range(1, sheet.max_row + 1): - for column_index in range(1, sheet.max_column + 1): + max_row, max_column = _xlsx_used_range_bounds(sheet) + for row_index in range(1, max_row + 1): + for column_index in range(1, max_column + 1): cell = sheet.cell(row=row_index, column=column_index) formula_cell = formula_sheet.cell( row=row_index, @@ -199,6 +201,48 @@ def source_cells_from_xlsx( return cells +# The last column (XFD) and row an .xlsx worksheet can address. +_XLSX_MAX_COLUMN = 16_384 +_XLSX_MAX_ROW = 1_048_576 + + +def _xlsx_used_range_bounds(sheet: Any) -> tuple[int, int]: + """Return the (max_row, max_column) of a worksheet's used range. + + This is openpyxl's ``(sheet.max_row, sheet.max_column)`` unless the sheet + has a merged range that runs to the worksheet edge. openpyxl fills every + position of a merged range with a ``MergedCell`` placeholder, so one + footnote row merged across the whole sheet width (``A179:XFD179`` in the + IRS's TY2023 ``23in54us.xlsx``) stretches ``max_column`` to 16,384 and the + used range to about 70 million empty cells. Such an edge-to-edge range is + sheet-width formatting, not table extent: its placeholders do not widen + the bounds, while its anchor cell and every other cell still do. Sheets + without one keep openpyxl's bounds exactly. + """ + edge_ranges = [ + cell_range + for cell_range in sheet.merged_cells.ranges + if cell_range.max_col >= _XLSX_MAX_COLUMN or cell_range.max_row >= _XLSX_MAX_ROW + ] + if not edge_ranges: + return sheet.max_row, sheet.max_column + + def edge_placeholder(row_index: int, column_index: int, cell: Any) -> bool: + return isinstance(cell, MergedCell) and any( + cell_range.min_row <= row_index <= cell_range.max_row + and cell_range.min_col <= column_index <= cell_range.max_col + for cell_range in edge_ranges + ) + + max_row = max_column = 1 + for (row_index, column_index), cell in sheet._cells.items(): + if edge_placeholder(row_index, column_index, cell): + continue + max_row = max(max_row, row_index) + max_column = max(max_column, column_index) + return max_row, max_column + + def source_cells_from_ods( content: bytes, artifact: SourceArtifactMetadata, diff --git a/chronicle/targets/us_poverty.py b/chronicle/targets/us_poverty.py index cfc5f3af..607134d2 100644 --- a/chronicle/targets/us_poverty.py +++ b/chronicle/targets/us_poverty.py @@ -85,6 +85,7 @@ def has_chronicle_package(self) -> bool: "soi-filing-season-week47-2024-eitc-total", "soi-table-4-3", "soi-state-2022", + "soi-state-2023", "soi-historic-table-2", "soi-historic-table-2-state-agi-2022", "soi-historic-table-2-state-broad-2022", diff --git a/db/data/irs_soi/ira_contributions/23in05ira.xlsx b/db/data/irs_soi/ira_contributions/23in05ira.xlsx new file mode 100644 index 00000000..982e0870 Binary files /dev/null and b/db/data/irs_soi/ira_contributions/23in05ira.xlsx differ diff --git a/db/data/irs_soi/ira_contributions/23in06ira.xlsx b/db/data/irs_soi/ira_contributions/23in06ira.xlsx new file mode 100644 index 00000000..da122a3a Binary files /dev/null and b/db/data/irs_soi/ira_contributions/23in06ira.xlsx differ diff --git a/db/data/irs_soi/ira_contributions/manifest_roth_2023_source_package.yaml b/db/data/irs_soi/ira_contributions/manifest_roth_2023_source_package.yaml new file mode 100644 index 00000000..ff9d3f98 --- /dev/null +++ b/db/data/irs_soi/ira_contributions/manifest_roth_2023_source_package.yaml @@ -0,0 +1,21 @@ +source_id: irs_soi +package_id: soi-ira-roth-contributions-2023 +dataset: irs_soi_roth_ira_contributions_2023 +source_page: https://www.irs.gov/statistics/soi-tax-stats-accumulation-and-distribution-of-individual-retirement-arrangements +table: Table 6. Taxpayers with Roth Individual Retirement Arrangement (IRA) Plan Contributions, + by Size of Contribution and Age of Taxpayer +files: + 2023: + filename: 23in06ira.xlsx + source_url: https://www.irs.gov/pub/irs-soi/23in06ira.xlsx + sha256: b526e4d4163172ed9992d16375cbf601b1c8b4d5bcf66f86e038c0e91c420418 + size_bytes: 11380 + fetched_at: '2026-09-27T05:55:43+00:00' + source_table: Table 6. Taxpayers with Roth Individual Retirement Arrangement (IRA) + Plan Contributions, by Size of Contribution and Age of Taxpayer + storage: + r2: + provider: r2 + bucket: ledger-raw + key: raw/irs_soi/soi-ira-roth-contributions-2023/2023/b526e4d4163172ed9992d16375cbf601b1c8b4d5bcf66f86e038c0e91c420418/23in06ira.xlsx + uri: r2://ledger-raw/raw/irs_soi/soi-ira-roth-contributions-2023/2023/b526e4d4163172ed9992d16375cbf601b1c8b4d5bcf66f86e038c0e91c420418/23in06ira.xlsx diff --git a/db/data/irs_soi/ira_contributions/manifest_traditional_2023_source_package.yaml b/db/data/irs_soi/ira_contributions/manifest_traditional_2023_source_package.yaml new file mode 100644 index 00000000..62d676e9 --- /dev/null +++ b/db/data/irs_soi/ira_contributions/manifest_traditional_2023_source_package.yaml @@ -0,0 +1,21 @@ +source_id: irs_soi +package_id: soi-ira-traditional-contributions-2023 +dataset: irs_soi_traditional_ira_contributions_2023 +source_page: https://www.irs.gov/statistics/soi-tax-stats-accumulation-and-distribution-of-individual-retirement-arrangements +table: Table 5. Taxpayers with Traditional Individual Retirement Arrangement (IRA) + Plan Contributions, by Size of Contribution and Age of Taxpayer +files: + 2023: + filename: 23in05ira.xlsx + source_url: https://www.irs.gov/pub/irs-soi/23in05ira.xlsx + sha256: 17b02670eeb35dab5d3f8be6a26333d527993f8b22933fa77bb3a06f3c8ad143 + size_bytes: 11345 + fetched_at: '2026-09-27T05:55:45+00:00' + source_table: Table 5. Taxpayers with Traditional Individual Retirement Arrangement + (IRA) Plan Contributions, by Size of Contribution and Age of Taxpayer + storage: + r2: + provider: r2 + bucket: ledger-raw + key: raw/irs_soi/soi-ira-traditional-contributions-2023/2023/17b02670eeb35dab5d3f8be6a26333d527993f8b22933fa77bb3a06f3c8ad143/23in05ira.xlsx + uri: r2://ledger-raw/raw/irs_soi/soi-ira-traditional-contributions-2023/2023/17b02670eeb35dab5d3f8be6a26333d527993f8b22933fa77bb3a06f3c8ad143/23in05ira.xlsx diff --git a/db/data/irs_soi/state_2023/23in54us.xlsx b/db/data/irs_soi/state_2023/23in54us.xlsx new file mode 100644 index 00000000..de69dddd Binary files /dev/null and b/db/data/irs_soi/state_2023/23in54us.xlsx differ diff --git a/db/data/irs_soi/state_2023/manifest.yaml b/db/data/irs_soi/state_2023/manifest.yaml new file mode 100644 index 00000000..de87c4cc --- /dev/null +++ b/db/data/irs_soi/state_2023/manifest.yaml @@ -0,0 +1,18 @@ +source_id: irs_soi +package_id: soi-state-2023 +dataset: irs_soi_state_2023 +source_page: https://www.irs.gov/statistics/soi-tax-stats-historic-table-2 +table: IRS SOI State Data 2023 +files: + 2023: + filename: 23in54us.xlsx + source_url: https://www.irs.gov/pub/irs-soi/23in54us.xlsx + sha256: 4bc45f2586fb924c1e0e7818773096bdcf71223b6407120ca058efaed728607a + size_bytes: 265549 + fetched_at: '2026-09-27T05:55:34+00:00' + storage: + r2: + provider: r2 + bucket: ledger-raw + key: raw/irs_soi/soi-state-2023/2023/4bc45f2586fb924c1e0e7818773096bdcf71223b6407120ca058efaed728607a/23in54us.xlsx + uri: r2://ledger-raw/raw/irs_soi/soi-state-2023/2023/4bc45f2586fb924c1e0e7818773096bdcf71223b6407120ca058efaed728607a/23in54us.xlsx diff --git a/docs/pe-calibration-targets.md b/docs/pe-calibration-targets.md index 899a8c39..82fbb12b 100644 --- a/docs/pe-calibration-targets.md +++ b/docs/pe-calibration-targets.md @@ -36,7 +36,7 @@ them to model variables. |---|---|---| | Population by age, sex, state, and congressional district | Census PEP and ACS demographics | `census-pep-2024-national-age-sex`, `census-pep-2024-state-age-sex`, `census-acs-s0101-national-age-2024`, `census-acs-s0101-state-age-2024`, `census-acs-s0101-congressional-district-age-2024` | | NIPA personal income, transfers, taxes, and pensions | BEA NIPA full-population aggregates | `bea-nipa-total-wages-salaries`, `bea-nipa-personal-income-components`, `bea-nipa-personal-income-disposition`, `bea-nipa-pension-contributions` | -| SOI filer income, taxes, deductions, and credits | IRS SOI administrative totals | `soi-table-1-1`, `soi-table-1-2`, `soi-table-1-4`, `soi-table-2-1`, `soi-table-2-5`, `soi-table-2-5-eitc-agi-children-2023`, `soi-table-4-3`, `soi-state-2022`, `soi-historic-table-2`, `soi-historic-table-2-state-agi-2022`, `soi-historic-table-2-state-broad-2022`, `soi-historic-table-2-state-eitc-2022`, `soi-w2-statistics-2020` | +| SOI filer income, taxes, deductions, and credits | IRS SOI administrative totals | `soi-table-1-1`, `soi-table-1-2`, `soi-table-1-4`, `soi-table-2-1`, `soi-table-2-5`, `soi-table-2-5-eitc-agi-children-2023`, `soi-table-4-3`, `soi-state-2022`, `soi-state-2023`, `soi-historic-table-2`, `soi-historic-table-2-state-agi-2022`, `soi-historic-table-2-state-broad-2022`, `soi-historic-table-2-state-eitc-2022`, `soi-w2-statistics-2020` | | Social Security and SSI | SSA administrative totals | `ssa-annual-statistical-supplement-2025`, `ssa-ssi-table-7b1-2024` | | SNAP | USDA FNS administrative totals | `usda-snap-fy69-to-current` | | TANF | HHS ACF caseload and financial totals | `hhs-acf-tanf-caseload-2024`, `hhs-acf-tanf-financial-2024` | diff --git a/packages/irs_soi/ira_roth_contributions_2023/source_package.yaml b/packages/irs_soi/ira_roth_contributions_2023/source_package.yaml new file mode 100644 index 00000000..8ce32df9 --- /dev/null +++ b/packages/irs_soi/ira_roth_contributions_2023/source_package.yaml @@ -0,0 +1,69 @@ +schema_version: ledger.source_package.v1 +package_id: soi-ira-roth-contributions-2023 +label: IRS SOI 2023 Roth IRA contribution totals +dimension_labels: + 'irs_soi.ira_contribution_size': 'IRA contribution size' +artifact: + source_name: irs_soi + source_table: Table 6. Taxpayers with Roth Individual Retirement Arrangement (IRA) Plan Contributions, by Size of Contribution and Age of Taxpayer + resource_package: db + resource_directory: data/irs_soi/ira_contributions + manifest: manifest_roth_2023_source_package.yaml + vintage: tax_year_2023 + extracted_at: '2026-09-27' + extraction_method: xlsx whole-workbook used-range cell parse + parser: xlsx_used_range + artifact_year: 2023 +record_sets: +- record_set_id: irs_soi.ty2023.roth_ira_contributions.all_taxpayers + provenance_class: administrative + record_set_spec_id: irs_soi.roth_ira_contributions.all_taxpayers.v1 + source_record_id_prefix: irs_soi.ty2023.roth_ira_contributions.all_taxpayers + sheet_name: Estimates + period_type: tax_year + period: '2023' + geography_id: 0100000US + geography_level: country + geography_name: United States + geography_vintage: current + entity: tax_unit + entity_role: filing_unit + domain: individual_retirement_arrangement_contributions + groupby_dimension: irs_soi.ira_contribution_size + rows: + - value_id: all_taxpayers + label: All taxpayers + ordinal: 0 + row_number: 7 + expected_row_header_column: A + expected_row_header: All taxpayers + table_record_kind: total + guard_cells: + - column: B + row: 4 + expected_value: Total + label: total contribution size + measures: + - measure_id: taxpayer_count + label: Number of taxpayers + ordinal: 0 + column: B + source_column_id: number_of_taxpayers + expected_column_header_row: 5 + expected_column_header: Number of taxpayers + concept: irs_soi.roth_ira_contributors + unit: count + aggregation: sum + expected_cell_type: number + - measure_id: amount + label: Contribution amount + ordinal: 1 + column: C + source_column_id: amount + expected_column_header_row: 5 + expected_column_header: Amount + concept: irs_soi.roth_ira_contributions + unit: usd + aggregation: sum + value_scale: 1000 + expected_cell_type: number diff --git a/packages/irs_soi/ira_traditional_contributions_2023/source_package.yaml b/packages/irs_soi/ira_traditional_contributions_2023/source_package.yaml new file mode 100644 index 00000000..e1b3ce50 --- /dev/null +++ b/packages/irs_soi/ira_traditional_contributions_2023/source_package.yaml @@ -0,0 +1,69 @@ +schema_version: ledger.source_package.v1 +package_id: soi-ira-traditional-contributions-2023 +label: IRS SOI 2023 Traditional IRA contribution totals +dimension_labels: + 'irs_soi.ira_contribution_size': 'IRA contribution size' +artifact: + source_name: irs_soi + source_table: Table 5. Taxpayers with Traditional Individual Retirement Arrangement (IRA) Plan Contributions, by Size of Contribution and Age of Taxpayer + resource_package: db + resource_directory: data/irs_soi/ira_contributions + manifest: manifest_traditional_2023_source_package.yaml + vintage: tax_year_2023 + extracted_at: '2026-09-27' + extraction_method: xlsx whole-workbook used-range cell parse + parser: xlsx_used_range + artifact_year: 2023 +record_sets: +- record_set_id: irs_soi.ty2023.traditional_ira_contributions.all_taxpayers + provenance_class: administrative + record_set_spec_id: irs_soi.traditional_ira_contributions.all_taxpayers.v1 + source_record_id_prefix: irs_soi.ty2023.traditional_ira_contributions.all_taxpayers + sheet_name: Estimates + period_type: tax_year + period: '2023' + geography_id: 0100000US + geography_level: country + geography_name: United States + geography_vintage: current + entity: tax_unit + entity_role: filing_unit + domain: individual_retirement_arrangement_contributions + groupby_dimension: irs_soi.ira_contribution_size + rows: + - value_id: all_taxpayers + label: All taxpayers + ordinal: 0 + row_number: 7 + expected_row_header_column: A + expected_row_header: All taxpayers + table_record_kind: total + guard_cells: + - column: B + row: 4 + expected_value: Total + label: total contribution size + measures: + - measure_id: taxpayer_count + label: Number of taxpayers + ordinal: 0 + column: B + source_column_id: number_of_taxpayers + expected_column_header_row: 5 + expected_column_header: Number of taxpayers + concept: irs_soi.traditional_ira_contributors + unit: count + aggregation: sum + expected_cell_type: number + - measure_id: amount + label: Contribution amount + ordinal: 1 + column: C + source_column_id: amount + expected_column_header_row: 5 + expected_column_header: Amount + concept: irs_soi.traditional_ira_contributions + unit: usd + aggregation: sum + value_scale: 1000 + expected_cell_type: number diff --git a/packages/irs_soi/state_2023/source_package.yaml b/packages/irs_soi/state_2023/source_package.yaml new file mode 100644 index 00000000..3794862a --- /dev/null +++ b/packages/irs_soi/state_2023/source_package.yaml @@ -0,0 +1,228 @@ +schema_version: ledger.source_package.v1 +package_id: soi-state-2023 +label: IRS SOI 2023 state-data United States totals +dimension_labels: + 'filing_status': 'Filing status' + 'income_range': 'Income range' + 'irs_soi.state_table_measure': 'State table measure' + 'us.tax.earned_income_credit_qualifying_children': 'EITC qualifying children' +dimension_value_labels: + 'filing_status': + 'all': 'All filing statuses' + 'income_range': + 'all': 'All income ranges' +artifact: + source_name: irs_soi + source_table: Historic Table 2 state data, United States total + resource_package: db + resource_directory: data/irs_soi/state_2023 + manifest: manifest.yaml + vintage: tax_year_2023 + extracted_at: "2026-09-27" + extraction_method: xlsx whole-workbook used-range cell parse + parser: xlsx_used_range + artifact_year: 2023 +record_sets: + - record_set_id: irs_soi.ty2023.state_2023.us.return_count + provenance_class: administrative + record_set_spec_id: irs_soi.state_2022.us.return_count.v1 + source_record_id_prefix: irs_soi.ty2023.state_2023.us.return_count + sheet_name: Sheet1 + period_type: tax_year + period: "2023" + geography_id: 0100000US + geography_level: country + geography_name: United States + geography_vintage: 2020_census + entity: tax_unit + entity_role: filing_unit + domain: all_individual_income_tax_returns + groupby_dimension: irs_soi.state_table_measure + rows: + - value_id: all_returns + label: All returns + ordinal: 0 + row_number: 9 + expected_row_header_column: A + expected_row_header: Number of returns [1] + filters: + filing_status: all + income_range: all + guard_cells: + - column: B + row: 3 + expected_value: "UNITED STATES " + label: geography section + table_record_kind: total + measures: + - measure_id: return_count + label: Number of returns + ordinal: 0 + column: B + source_column_id: all_returns + expected_column_header_row: 4 + expected_column_header: All returns + concept: irs_soi.individual_income_tax_returns + unit: count + aggregation: sum + expected_cell_type: number + - record_set_id: irs_soi.ty2023.state_2023.us.adjusted_gross_income + provenance_class: administrative + record_set_spec_id: irs_soi.state_2022.us.adjusted_gross_income.v1 + source_record_id_prefix: irs_soi.ty2023.state_2023.us.adjusted_gross_income + sheet_name: Sheet1 + period_type: tax_year + period: "2023" + geography_id: 0100000US + geography_level: country + geography_name: United States + geography_vintage: 2020_census + entity: tax_unit + entity_role: filing_unit + domain: all_individual_income_tax_returns + groupby_dimension: irs_soi.state_table_measure + rows: + - value_id: all_returns + label: All returns + ordinal: 0 + row_number: 26 + expected_row_header_column: A + expected_row_header: Adjusted gross income (AGI) [6] + filters: + filing_status: all + income_range: all + guard_cells: + - column: B + row: 3 + expected_value: "UNITED STATES " + label: geography section + table_record_kind: total + measures: + - measure_id: amount + label: Adjusted gross income + ordinal: 0 + column: B + source_column_id: all_returns + expected_column_header_row: 4 + expected_column_header: All returns + concept: us:statutes/26/62#adjusted_gross_income + source_concept: irs_soi.adjusted_gross_income + concept_relation: exact + concept_authority: ledger-us + concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) + concept_evidence_notes: > + IRS SOI Historic Table 2 state-data files report adjusted gross + income for individual income tax returns; IRC section 62 defines + adjusted gross income. This Ledger assertion treats the SOI AGI row + as exactly adopting that legal concept for the tax-year source + record. + legal_vintage: tax_year_2023 + unit: usd + aggregation: sum + value_scale: 1000 + expected_cell_type: number + - record_set_id: irs_soi.ty2023.state_2023.us.eitc_three_or_more_children_returns + provenance_class: administrative + record_set_spec_id: irs_soi.state_2022.us.eitc_three_or_more_children_returns.v1 + source_record_id_prefix: irs_soi.ty2023.state_2023.us.eitc_three_or_more_children_returns + sheet_name: Sheet1 + period_type: tax_year + period: "2023" + geography_id: 0100000US + geography_level: country + geography_name: United States + geography_vintage: 2020_census + entity: tax_unit + entity_role: filing_unit + domain: all_individual_income_tax_returns + groupby_dimension: us.tax.earned_income_credit_qualifying_children + rows: + - value_id: three_or_more_qualifying_children + label: Three or more qualifying children + ordinal: 0 + row_number: 146 + expected_row_header_column: A + expected_row_header: " Earned income credit with three or more qualifying children: Number" + filters: + filing_status: all + income_range: all + constraints: + - variable: us.tax.earned_income_credit_qualifying_children + operator: ">=" + value: 3 + unit: count + label: Earned income credit qualifying children lower bound + guard_cells: + - column: B + row: 3 + expected_value: "UNITED STATES " + label: geography section + - column: A + row: 138 + expected_value: "Earned income credit: [10] Number" + label: earned income credit section + measures: + - measure_id: return_count + label: Returns with earned income credit + ordinal: 0 + column: B + source_column_id: all_returns + expected_column_header_row: 4 + expected_column_header: All returns + concept: irs_soi.returns_with_earned_income_credit + unit: count + aggregation: sum + expected_cell_type: number + - record_set_id: irs_soi.ty2023.state_2023.us.eitc_three_or_more_children_amount + provenance_class: administrative + record_set_spec_id: irs_soi.state_2022.us.eitc_three_or_more_children_amount.v1 + source_record_id_prefix: irs_soi.ty2023.state_2023.us.eitc_three_or_more_children_amount + sheet_name: Sheet1 + period_type: tax_year + period: "2023" + geography_id: 0100000US + geography_level: country + geography_name: United States + geography_vintage: 2020_census + entity: tax_unit + entity_role: filing_unit + domain: all_individual_income_tax_returns + groupby_dimension: us.tax.earned_income_credit_qualifying_children + rows: + - value_id: three_or_more_qualifying_children + label: Three or more qualifying children + ordinal: 0 + row_number: 147 + expected_row_header_column: A + expected_row_header: " Amount" + filters: + filing_status: all + income_range: all + constraints: + - variable: us.tax.earned_income_credit_qualifying_children + operator: ">=" + value: 3 + unit: count + label: Earned income credit qualifying children lower bound + guard_cells: + - column: B + row: 3 + expected_value: "UNITED STATES " + label: geography section + - column: A + row: 146 + expected_value: " Earned income credit with three or more qualifying children: Number" + label: earned income credit qualifying children row + measures: + - measure_id: amount + label: Earned income credit + ordinal: 0 + column: B + source_column_id: all_returns + expected_column_header_row: 4 + expected_column_header: All returns + concept: irs_soi.earned_income_credit + unit: usd + aggregation: sum + value_scale: 1000 + expected_cell_type: number diff --git a/tests/test_chronicle_bundle.py b/tests/test_chronicle_bundle.py index 4d6fcd36..a8866dd6 100644 --- a/tests/test_chronicle_bundle.py +++ b/tests/test_chronicle_bundle.py @@ -136,13 +136,13 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "aggregate_duplicate_key_count": 0, "entity_count": 12, "error_count": 0, - "fact_count": 350893, + "fact_count": 350901, "geography_count": 12592, "period_count": 494, "semantic_duplicate_key_count": 467, "skipped_source_count": 10, "source_count": 50, - "source_package_count": 227, + "source_package_count": 230, # 1 semantic-duplicate warning, plus the publisher wording Chronicle # keeps as published: values two packages word differently, groupby # rows that drift inside one package (chronicle#265, #266), and the @@ -150,7 +150,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): # it does has stopped warning (chronicle#281). "warning_count": 76, } - assert len(rows) == 350893 + assert len(rows) == 350901 assert {row["provenance_class"] for row in rows} <= { "administrative", "census", @@ -168,7 +168,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): ) assert rows[0]["aggregate_fact_key"].startswith("ledger.aggregate_fact.v2:") assert rows[0]["semantic_fact_key"].startswith("ledger.semantic_fact.v2:") - assert source_packages["source_package_count"] == 227 + assert source_packages["source_package_count"] == 230 assert source_packages["skipped_source_count"] == 10 assert sorted(item["source"] for item in source_packages["skipped_sources"]) == [ "census-acs-s0101-congressional-district-age-2024", @@ -182,7 +182,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "jct-obbba-revenue-estimates-2025", "jct-tax-expenditures-2024", ] - assert coverage["fact_count"] == 350893 + assert coverage["fact_count"] == 350901 assert coverage["counts"]["by_source"] == { "bea": 445, "bfp_economic_outlook": 5, @@ -207,7 +207,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "hhs_acf_tanf": 110, "hmrc": 31343, "ici": 12, - "irs_soi": 40063, + "irs_soi": 40071, "isc": 2, "jrc_euromod_be": 90, "kff": 52, @@ -1190,6 +1190,8 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): issue_280_period_increments[f"month:{month}"] = 23 for key, count in issue_280_period_increments.items(): expected_period_counts[key] = expected_period_counts.get(key, 0) + count + # IRS SOI TY2023 state-data US totals (4) and IRA Tables 5/6 (2 + 2). + expected_period_counts["tax_year:2023"] += 8 assert coverage["counts"]["by_period"] == expected_period_counts assert coverage["counts"]["by_geography"]["country:BE"] == 4888 assert coverage["counts"]["by_geography"]["country:DE"] == 36 @@ -1198,7 +1200,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): assert coverage["counts"]["by_geography"]["nuts1:BE2"] == 3673 assert coverage["counts"]["by_geography"]["nuts1:BE3"] == 3662 assert coverage["counts"]["by_geography"]["commune:11002"] == 1 - assert coverage["counts"]["by_geography"]["country:0100000US"] == 2109 + assert coverage["counts"]["by_geography"]["country:0100000US"] == 2117 assert coverage["counts"]["by_geography"]["state:0400000US06"] == 229 assert ( coverage["counts"]["by_geography"]["congressional_district:5001700US0601"] == 56 @@ -1220,7 +1222,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "person": 74808, "return": 14600, "social_protection_scheme": 36, - "tax_unit": 41368, + "tax_unit": 41376, } assert not coverage["duplicates"]["aggregate_fact_keys"] assert len(coverage["duplicates"]["semantic_fact_keys"]) == 467 diff --git a/tests/test_chronicle_soi_state_ira_2023.py b/tests/test_chronicle_soi_state_ira_2023.py new file mode 100644 index 00000000..f57d876b --- /dev/null +++ b/tests/test_chronicle_soi_state_ira_2023.py @@ -0,0 +1,269 @@ +"""IRS SOI TY2023 state-data US totals and IRA Tables 5/6. + +``soi-state-2023`` reads the United States workbook of the TY2023 state data +(``23in54us.xlsx``); ``soi-ira-traditional-contributions-2023`` and +``soi-ira-roth-contributions-2023`` read IRA Tables 5 and 6 of the June 2026 +IRA study (``23in05ira.xlsx``, ``23in06ira.xlsx``). Each mirrors its TY2022 +twin with literal TY2023 labels. The workbooks' layouts moved, and the +packages move with them: + +- 23in54us.xlsx: the geography label moved from A8 to B3, the "All returns" + header from row 3 to row 4, and the EITC block two rows down. +- The IRA tables use an "Estimates" sheet with "All taxpayers" on row 7. + +The expected values are read with openpyxl directly, not through Chronicle's +cell parser. +""" + +from __future__ import annotations + +import csv +import hashlib +import json +from pathlib import Path + +import openpyxl +import pytest +import yaml + +from chronicle.core import validate_facts +from chronicle.source_package import SOURCE_PACKAGE_ALIASES, load_source_package +from chronicle.sources.cells import validate_source_cells +from chronicle.suite import build_source_suite + +REPO_ROOT = Path(__file__).resolve().parents[1] +DATA = REPO_ROOT / "db" / "data" / "irs_soi" +PACKAGES_ROOT = REPO_ROOT / "packages" + +# package -> (TY2022 twin, manifest, workbook, sha256, size, sheet, +# {source_record_id suffix: (cell, scale)}) +PACKAGES = { + "soi-state-2023": ( + "soi-state-2022", + DATA / "state_2023" / "manifest.yaml", + DATA / "state_2023" / "23in54us.xlsx", + "4bc45f2586fb924c1e0e7818773096bdcf71223b6407120ca058efaed728607a", + 265_549, + "Sheet1", + { + "state_2023.us.return_count.all_returns.return_count": ("B9", 1), + "state_2023.us.adjusted_gross_income.all_returns.amount": ("B26", 1000), + ( + "state_2023.us.eitc_three_or_more_children_returns." + "three_or_more_qualifying_children.return_count" + ): ("B146", 1), + ( + "state_2023.us.eitc_three_or_more_children_amount." + "three_or_more_qualifying_children.amount" + ): ("B147", 1000), + }, + ), + "soi-ira-traditional-contributions-2023": ( + "soi-ira-traditional-contributions-2022", + DATA / "ira_contributions" / "manifest_traditional_2023_source_package.yaml", + DATA / "ira_contributions" / "23in05ira.xlsx", + "17b02670eeb35dab5d3f8be6a26333d527993f8b22933fa77bb3a06f3c8ad143", + 11_345, + "Estimates", + { + "traditional_ira_contributions.all_taxpayers.all_taxpayers.taxpayer_count": ( + "B7", + 1, + ), + "traditional_ira_contributions.all_taxpayers.all_taxpayers.amount": ( + "C7", + 1000, + ), + }, + ), + "soi-ira-roth-contributions-2023": ( + "soi-ira-roth-contributions-2022", + DATA / "ira_contributions" / "manifest_roth_2023_source_package.yaml", + DATA / "ira_contributions" / "23in06ira.xlsx", + "b526e4d4163172ed9992d16375cbf601b1c8b4d5bcf66f86e038c0e91c420418", + 11_380, + "Estimates", + { + "roth_ira_contributions.all_taxpayers.all_taxpayers.taxpayer_count": ( + "B7", + 1, + ), + "roth_ira_contributions.all_taxpayers.all_taxpayers.amount": ("C7", 1000), + }, + ), +} +PUBLISHED = { + # Spot values from the publisher files, as printed. + "irs_soi.ty2023.state_2023.us.return_count.all_returns.return_count": 159_949_000, + "irs_soi.ty2023.traditional_ira_contributions.all_taxpayers.all_taxpayers." + "taxpayer_count": 5_873_053, + "irs_soi.ty2023.roth_ira_contributions.all_taxpayers.all_taxpayers." + "taxpayer_count": 9_488_414, +} + + +def _build(package_id: str, year: int): + package = load_source_package(package_id) + rows = package.build_source_rows(year) + cells = package.build_source_cells(year, source_rows=rows) + return cells, package.build_facts(year, cells=cells, source_rows=rows) + + +@pytest.fixture(scope="module") +def built(): + facts_by_package = {} + for package_id in PACKAGES: + cells, facts = _build(package_id, 2023) + assert validate_source_cells(cells).valid + assert validate_facts(facts).valid + facts_by_package[package_id] = facts + return facts_by_package + + +def _cell(workbook: Path, sheet: str, address: str): + book = openpyxl.load_workbook(workbook, read_only=True, data_only=True) + try: + return book[sheet][address].value + finally: + book.close() + + +@pytest.mark.parametrize("package_id", sorted(PACKAGES)) +def test_manifest_registers_the_publisher_file(package_id): + _, manifest_path, workbook, sha256, size, _, _ = PACKAGES[package_id] + entry = yaml.safe_load(manifest_path.read_text())["files"][2023] + + assert hashlib.sha256(workbook.read_bytes()).hexdigest() == sha256 + assert entry["filename"] == workbook.name + assert entry["source_url"] == f"https://www.irs.gov/pub/irs-soi/{workbook.name}" + assert entry["sha256"] == sha256 + assert entry["size_bytes"] == size == workbook.stat().st_size + assert entry["storage"]["r2"]["key"].endswith(f"/2023/{sha256}/{workbook.name}") + + +@pytest.mark.parametrize("package_id", sorted(PACKAGES)) +def test_each_fact_is_its_publisher_cell(built, package_id): + _, _, workbook, sha256, _, sheet, expected = PACKAGES[package_id] + facts = {fact.source_record_id: fact for fact in built[package_id]} + + assert set(facts) == {f"irs_soi.ty2023.{suffix}" for suffix in expected} + for suffix, (address, scale) in expected.items(): + fact = facts[f"irs_soi.ty2023.{suffix}"] + assert fact.value == _cell(workbook, sheet, address) * scale, suffix + assert fact.period.type == "tax_year" + assert fact.period.value == 2023 + assert fact.source.source_sha256 == sha256 + assert fact.source.vintage == "tax_year_2023" + assert fact.geography.id == "0100000US" + if fact.measure.legal_vintage is not None: + assert fact.measure.legal_vintage == "tax_year_2023" + for record_id, value in PUBLISHED.items(): + if record_id.startswith(tuple(f"irs_soi.ty2023.{s}" for s in expected)): + assert facts[record_id].value == value + + +def test_state_totals_equal_the_historic_table_2_us_row(built): + """Differential across two publisher files: the TY2023 United States + workbook and the TY2023 Historic Table 2 CSV (US, AGI stub 0) report the + same totals.""" + csv_path = DATA / "historic_table_2" / "23in55cmcsv.csv" + if not csv_path.exists(): + pytest.skip("23in55cmcsv.csv arrives with chronicle#291") + with csv_path.open(newline="", encoding="utf-8-sig") as handle: + us = next( + row + for row in csv.DictReader(handle) + if row["STATE"] == "US" and row["AGI_STUB"] == "0" + ) + by_measure = { + fact.source_record_id.split(".")[4]: fact.value + for fact in built["soi-state-2023"] + } + variables = { + "return_count": ("N1", 1), + "adjusted_gross_income": ("A00100", 1000), + "eitc_three_or_more_children_returns": ("N59664", 1), + "eitc_three_or_more_children_amount": ("A59664", 1000), + } + for measure, (variable, scale) in variables.items(): + assert by_measure[measure] == int(us[variable].replace(",", "")) * scale + + +def test_every_published_spot_value_belongs_to_a_package(): + ids = { + f"irs_soi.ty2023.{suffix}" + for package in PACKAGES.values() + for suffix in package[6] + } + assert set(PUBLISHED) <= ids + + +@pytest.mark.parametrize("build_year", [2022, 2024]) +def test_building_at_another_year_does_not_relabel_the_2023_files(built, build_year): + for package_id in PACKAGES: + _, other = _build(package_id, build_year) + assert sorted((f.source_record_id, f.value, f.period.value) for f in other) == ( + sorted( + (f.source_record_id, f.value, f.period.value) for f in built[package_id] + ) + ) + + +def _declarations(package_id: str) -> dict: + path = PACKAGES_ROOT / SOURCE_PACKAGE_ALIASES[package_id] / "source_package.yaml" + return yaml.safe_load(path.read_text()) + + +@pytest.mark.parametrize("package_id", sorted(PACKAGES)) +def test_packages_mirror_their_2022_twins(package_id): + """Differential: apart from year labels, file locations and the cell + positions the TY2023 layout moved, the declarations equal TY2022's.""" + text = ( + PACKAGES_ROOT / SOURCE_PACKAGE_ALIASES[package_id] / "source_package.yaml" + ).read_text() + assert "{year}" not in text + + def normalised(spec: dict) -> dict: + spec = json.loads(json.dumps(spec)) + for key in ("package_id", "label"): + spec.pop(key) + for key in ( + "vintage", + "extracted_at", + "artifact_year", + "manifest", + "resource_directory", + ): + spec["artifact"].pop(key, None) + for record_set in spec["record_sets"]: + for key in ( + "record_set_id", + "source_record_id_prefix", + "period", + "sheet_name", + ): + record_set.pop(key) + for row in record_set["rows"]: + row.pop("row_number") + for guard in row.get("guard_cells", ()): + guard.pop("row", None) + guard.pop("column", None) + for measure in record_set["measures"]: + for key in ( + "legal_vintage", + "expected_column_header_row", + "expected_column_header", + ): + measure.pop(key, None) + return spec + + twin = PACKAGES[package_id][0] + assert normalised(_declarations(package_id)) == normalised(_declarations(twin)) + + +@pytest.mark.parametrize("package_id", sorted(PACKAGES)) +def test_source_suite_passes_agent_acceptance(package_id, tmp_path): + suite = build_source_suite(package_id, tmp_path / package_id, year=2023) + + assert suite.agent_acceptance.valid + assert suite.agent_acceptance.counts["row_semantic_error_count"] == 0 diff --git a/tests/test_chronicle_source_cells.py b/tests/test_chronicle_source_cells.py index 1d4603f6..648d5b0b 100644 --- a/tests/test_chronicle_source_cells.py +++ b/tests/test_chronicle_source_cells.py @@ -421,3 +421,111 @@ def test_source_cells_from_xlsx_rejects_a_sheet_the_workbook_does_not_carry(): artifact, sheets=("UKPC", "Renamed_by_publisher"), ) + + +def _test_artifact() -> SourceArtifactMetadata: + return SourceArtifactMetadata( + source_name="irs_soi", + source_table="test", + source_file="test.xlsx", + url="https://example.test/test.xlsx", + vintage="test", + sha256="abc123", + size_bytes=10, + extracted_at="2026-09-27", + extraction_method="test", + ) + + +def _workbook_bytes(data_extent: tuple[int, int], merged: list[str]) -> bytes: + workbook = openpyxl.Workbook() + sheet = workbook.active + for row in range(1, data_extent[0] + 1): + for column in range(1, data_extent[1] + 1): + sheet.cell(row=row, column=column, value=row * 100 + column) + for cell_range in merged: + sheet.merge_cells(cell_range) + buffer = BytesIO() + workbook.save(buffer) + return buffer.getvalue() + + +def test_source_cells_from_xlsx_ignores_a_row_merged_to_the_sheet_edge(): + """23in54us.xlsx (TY2023) merges its footnote row A179:XFD179. openpyxl + fills all 16,384 columns with MergedCell placeholders, so the used range + grew to ~70 million cells; the parse now stops at the file's own cells.""" + content = _workbook_bytes((4, 5), ["A6:XFD6"]) + + cells = source_cells_from_xlsx(content, _test_artifact()) + + assert {(cell.row_number, cell.column_number) for cell in cells} == { + (row, column) for row in range(1, 7) for column in range(1, 6) + } + assert {cell.address: cell.raw_value for cell in cells}["C2"] == 203 + + +def test_xlsx_used_range_bounds_equal_openpyxl_without_an_edge_merge(): + """Invariant: a sheet with no merged range reaching column XFD or the last + row keeps openpyxl's (max_row, max_column), whether its merged ranges sit + inside the data or run past it; with one, only its placeholders stop + counting.""" + from chronicle.sources.cells import _xlsx_used_range_bounds + + layouts = [ + ((rows, columns), merged) + for rows in (1, 3) + for columns in (1, 4) + for merged in ( + [], + ["A1:B1"], + [f"A{rows + 2}:F{rows + 2}"], + [f"B{rows + 1}:B{rows + 4}"], + ["C2:H9"], + ) + ] + for extent, merged in layouts: + sheet = openpyxl.load_workbook(BytesIO(_workbook_bytes(extent, merged))).active + assert _xlsx_used_range_bounds(sheet) == (sheet.max_row, sheet.max_column), ( + extent, + merged, + ) + + edge = openpyxl.load_workbook( + BytesIO(_workbook_bytes(extent, [*merged, "A20:XFD20"])) + ).active + assert edge.max_column == 16_384 + assert _xlsx_used_range_bounds(edge) == ( + max(sheet.max_row, 20), + sheet.max_column, + ), (extent, merged) + + +def test_xlsx_used_range_bounds_ignore_a_column_merged_to_the_last_row(): + """The row edge works like the column edge. A stub stands in for the + sheet: openpyxl would materialise over a million MergedCells for a real + A1:A1048576 merge.""" + from types import SimpleNamespace + + from openpyxl.cell.cell import MergedCell + from openpyxl.worksheet.cell_range import CellRange + + from chronicle.sources.cells import _xlsx_used_range_bounds + + real = openpyxl.Workbook().active + cells = { + (row, column): real.cell(row=row, column=column) + for row in (1, 2, 3) + for column in (1, 2) + } + cells.update( + {(row, 4): MergedCell(real, row=row, column=4) for row in range(5, 60)} + ) + cells[(4, 4)] = real.cell(row=4, column=4) # the merge's anchor + sheet = SimpleNamespace( + merged_cells=SimpleNamespace(ranges=[CellRange("D4:D1048576")]), + _cells=cells, + max_row=1_048_576, + max_column=4, + ) + + assert _xlsx_used_range_bounds(sheet) == (4, 4) diff --git a/tests/test_chronicle_source_package.py b/tests/test_chronicle_source_package.py index 6c7963b5..fd9083dd 100644 --- a/tests/test_chronicle_source_package.py +++ b/tests/test_chronicle_source_package.py @@ -1657,6 +1657,20 @@ def test_national_soi_source_package_aliases_validate_fixture_counts(): "source_record_count": 2, "source_region_count": 1, }, + "soi-ira-traditional-contributions-2023": { + "record_set_count": 1, + "row_count": 1, + "measure_count": 2, + "source_record_count": 2, + "source_region_count": 1, + }, + "soi-ira-roth-contributions-2023": { + "record_set_count": 1, + "row_count": 1, + "measure_count": 2, + "source_record_count": 2, + "source_region_count": 1, + }, } for package_id, counts in expected_counts.items():