From 9e9ba6822cc706bf12afa3814f50d7a86ef6738c Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 25 Sep 2026 18:29:18 -0400 Subject: [PATCH 1/6] Pin label years in pinned SOI and ICI packages; make BEA regional year-selecting These packages pin artifact_year, so every --year reads the same file, but their labels rendered {year} from --year. A --year 2023 build therefore stamped TY2022 and TY2020 IRS tables as ty2023 (chronicle#117 item 1) and stamped CY2024 BEA regional values as cy2023. - soi-w2-statistics-2020: period, record ids, vintage and legal_vintage are literal 2020. - soi-congressional-district-2022, soi-state-2022 and both IRA 2022 packages: the same labels are literal 2022. - The three Historic Table 2 packages: legal_vintage is literal tax_year_2022. - ici-fact-book-table-30: vintage is literal calendar_year_2026, which is what its artifact-year build emits. - bea-regional-state-personal-income-components-2024 now selects SAINC5N__ALL_AREAS_1998_2025.csv's own year column with column_by_year (1998 = I ... 2023 = AH, 2024 = AI, 2025 = AJ), and source_column_id follows the year. --year 2023 now reads the 2023 column: US personal income is 23,577,208,000 thousand, not the 2024 cell's 24,897,613,000. Every artifact-year build is byte-identical to origin/main (facts and consumer rows for all ten packages, BEA regional at 2024 included). Co-Authored-By: Claude Opus 5.5 --- .../source_package.yaml | 58 +++++++--- .../fact_book_table_30/source_package.yaml | 2 +- .../source_package.yaml | 10 +- .../historic_table_2/source_package.yaml | 4 +- .../source_package.yaml | 102 +++++++++--------- .../source_package.yaml | 4 +- .../source_package.yaml | 8 +- .../source_package.yaml | 8 +- .../irs_soi/state_2022/source_package.yaml | 28 ++--- .../w2_statistics_2020/source_package.yaml | 22 ++-- tests/test_chronicle_source_package.py | 30 ++++++ 11 files changed, 168 insertions(+), 108 deletions(-) diff --git a/packages/bea/regional_personal_income_state/source_package.yaml b/packages/bea/regional_personal_income_state/source_package.yaml index 788245d1..69731286 100644 --- a/packages/bea/regional_personal_income_state/source_package.yaml +++ b/packages/bea/regional_personal_income_state/source_package.yaml @@ -3165,8 +3165,38 @@ record_sets: - measure_id: amount label: Personal income ordinal: 0 - column: AI - source_column_id: '2024' + # SAINC5N__ALL_AREAS_1998_2025.csv holds one column per calendar + # year, 1998 (I) through 2025 (AJ); --year selects that column. + column_by_year: &sainc5n_year_columns + 1998: I + 1999: J + 2000: K + 2001: L + 2002: M + 2003: N + 2004: O + 2005: P + 2006: Q + 2007: R + 2008: S + 2009: T + 2010: U + 2011: V + 2012: W + 2013: X + 2014: Y + 2015: Z + 2016: AA + 2017: AB + 2018: AC + 2019: AD + 2020: AE + 2021: AF + 2022: AG + 2023: AH + 2024: AI + 2025: AJ + source_column_id: '{year}' concept: bea_regional.personal_income source_concept: bea_regional.sainc5n_line_10_personal_income concept_relation: source_label @@ -5070,8 +5100,8 @@ record_sets: - measure_id: amount label: Dividends, interest, and rent ordinal: 0 - column: AI - source_column_id: '2024' + column_by_year: *sainc5n_year_columns + source_column_id: '{year}' concept: bea_regional.dividends_interest_and_rent source_concept: bea_regional.sainc5n_line_46_dividends_interest_and_rent concept_relation: source_label @@ -6975,8 +7005,8 @@ record_sets: - measure_id: amount label: Personal current transfer receipts ordinal: 0 - column: AI - source_column_id: '2024' + column_by_year: *sainc5n_year_columns + source_column_id: '{year}' concept: bea_regional.personal_current_transfer_receipts source_concept: bea_regional.sainc5n_line_47_personal_current_transfer_receipts concept_relation: source_label @@ -8880,8 +8910,8 @@ record_sets: - measure_id: amount label: Wages and salaries ordinal: 0 - column: AI - source_column_id: '2024' + column_by_year: *sainc5n_year_columns + source_column_id: '{year}' concept: bea_regional.wages_and_salaries source_concept: bea_regional.sainc5n_line_50_wages_and_salaries concept_relation: source_label @@ -10785,8 +10815,8 @@ record_sets: - measure_id: amount label: Supplements to wages and salaries ordinal: 0 - column: AI - source_column_id: '2024' + column_by_year: *sainc5n_year_columns + source_column_id: '{year}' concept: bea_regional.supplements_to_wages_and_salaries source_concept: bea_regional.sainc5n_line_60_supplements_to_wages_and_salaries concept_relation: source_label @@ -12690,8 +12720,8 @@ record_sets: - measure_id: amount label: Proprietors' income ordinal: 0 - column: AI - source_column_id: '2024' + column_by_year: *sainc5n_year_columns + source_column_id: '{year}' concept: bea_regional.proprietors_income source_concept: bea_regional.sainc5n_line_70_proprietors_income concept_relation: source_label @@ -14595,7 +14625,7 @@ record_sets: - measure_id: amount label: Contributions for government social insurance ordinal: 0 - column: AI + column_by_year: *sainc5n_year_columns source_column_id: value concept: bea_regional.contributions_for_government_social_insurance source_concept: bea_regional.sainc5n_line_36_contributions_for_government_social_insurance @@ -16500,7 +16530,7 @@ record_sets: - measure_id: amount label: Residence adjustment ordinal: 0 - column: AI + column_by_year: *sainc5n_year_columns source_column_id: value concept: bea_regional.residence_adjustment source_concept: bea_regional.sainc5n_line_42_residence_adjustment diff --git a/packages/ici/fact_book_table_30/source_package.yaml b/packages/ici/fact_book_table_30/source_package.yaml index 2aa10364..26688487 100644 --- a/packages/ici/fact_book_table_30/source_package.yaml +++ b/packages/ici/fact_book_table_30/source_package.yaml @@ -9,7 +9,7 @@ artifact: resource_package: db resource_directory: data/ici/fact_book_table_30 manifest: manifest.yaml - vintage: calendar_year_{year} + vintage: calendar_year_2026 extracted_at: "2026-07-02" extraction_method: xlsx whole-workbook used-range cell parse parser: xlsx_used_range diff --git a/packages/irs_soi/congressional_district_2022/source_package.yaml b/packages/irs_soi/congressional_district_2022/source_package.yaml index 5fd3dbdd..ef37a997 100644 --- a/packages/irs_soi/congressional_district_2022/source_package.yaml +++ b/packages/irs_soi/congressional_district_2022/source_package.yaml @@ -19,7 +19,7 @@ artifact: resource_package: db resource_directory: data/irs_soi/congressional_district_2022 manifest: manifest.yaml - vintage: tax_year_{year} + vintage: tax_year_2022 extracted_at: '2026-05-11' extraction_method: full CSV row parse with all-return national, state, and congressional district facts parser: delimited_text_full_rows @@ -1467,13 +1467,13 @@ artifact: CONG_DISTRICT: 0 agi_stub: 0 record_sets: -- record_set_id: irs_soi.ty{year}.congressional_district_2022.all_returns +- record_set_id: irs_soi.ty2022.congressional_district_2022.all_returns provenance_class: administrative record_set_spec_id: irs_soi.congressional_district_2022.all_returns.v1 - source_record_id_prefix: irs_soi.ty{year}.congressional_district_2022.all_returns + source_record_id_prefix: irs_soi.ty2022.congressional_district_2022.all_returns sheet_name: incd2022 period_type: tax_year - period: '{year}' + period: '2022' geography_id: 0100000US geography_level: country geography_name: United States @@ -14003,7 +14003,7 @@ record_sets: concept_evidence_notes: IRS SOI congressional district files report adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 diff --git a/packages/irs_soi/historic_table_2/source_package.yaml b/packages/irs_soi/historic_table_2/source_package.yaml index 4a5104b4..723ae241 100644 --- a/packages/irs_soi/historic_table_2/source_package.yaml +++ b/packages/irs_soi/historic_table_2/source_package.yaml @@ -297,7 +297,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -431,7 +431,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://www.irs.gov/statistics/soi-tax-stats-historic-table-2 concept_evidence_notes: IRS SOI Historic Table 2 labels this measure salaries and wages for individual income tax returns. Ledger treats the SOI wage measure as a broad match to the wage input of IRC section 62 adjusted gross income. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 - measure_id: taxable_interest_returns label: Returns with taxable interest diff --git a/packages/irs_soi/historic_table_2_state_agi_2022/source_package.yaml b/packages/irs_soi/historic_table_2_state_agi_2022/source_package.yaml index 5de0b0cb..135c2a70 100644 --- a/packages/irs_soi/historic_table_2_state_agi_2022/source_package.yaml +++ b/packages/irs_soi/historic_table_2_state_agi_2022/source_package.yaml @@ -1296,7 +1296,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -1561,7 +1561,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -1826,7 +1826,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -2091,7 +2091,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -2356,7 +2356,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -2621,7 +2621,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -2886,7 +2886,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -3151,7 +3151,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -3416,7 +3416,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -3681,7 +3681,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -3946,7 +3946,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -4211,7 +4211,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -4476,7 +4476,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -4741,7 +4741,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -5006,7 +5006,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -5271,7 +5271,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -5536,7 +5536,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -5801,7 +5801,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -6066,7 +6066,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -6331,7 +6331,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -6596,7 +6596,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -6861,7 +6861,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -7126,7 +7126,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -7391,7 +7391,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -7656,7 +7656,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -7921,7 +7921,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -8186,7 +8186,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -8451,7 +8451,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -8716,7 +8716,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -8981,7 +8981,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -9246,7 +9246,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -9511,7 +9511,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -9776,7 +9776,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -10041,7 +10041,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -10306,7 +10306,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -10571,7 +10571,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -10836,7 +10836,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -11101,7 +11101,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -11366,7 +11366,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -11631,7 +11631,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -11896,7 +11896,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -12161,7 +12161,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -12426,7 +12426,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -12691,7 +12691,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -12956,7 +12956,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -13221,7 +13221,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -13486,7 +13486,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -13751,7 +13751,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -14016,7 +14016,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -14281,7 +14281,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 @@ -14546,7 +14546,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 diff --git a/packages/irs_soi/historic_table_2_state_broad_2022/source_package.yaml b/packages/irs_soi/historic_table_2_state_broad_2022/source_package.yaml index e876dcb0..297a6a01 100644 --- a/packages/irs_soi/historic_table_2_state_broad_2022/source_package.yaml +++ b/packages/irs_soi/historic_table_2_state_broad_2022/source_package.yaml @@ -199,7 +199,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://uscode.house.gov/view.xhtml?req=(title:26%20section:62%20edition:prelim) concept_evidence_notes: IRS SOI Historic Table 2 reports adjusted gross income for individual income tax returns; IRC section 62 defines adjusted gross income. This Ledger assertion treats the SOI AGI column as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 - measure_id: total_income_returns label: Returns with total income ordinal: 3 @@ -251,7 +251,7 @@ record_sets: concept_authority: ledger-us concept_evidence_url: https://www.irs.gov/statistics/soi-tax-stats-historic-table-2 concept_evidence_notes: IRS SOI Historic Table 2 labels this measure salaries and wages for individual income tax returns. Ledger treats the SOI wage measure as a broad match to the wage input of IRC section 62 adjusted gross income. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 - measure_id: taxable_interest_returns label: Returns with taxable interest ordinal: 7 diff --git a/packages/irs_soi/ira_roth_contributions_2022/source_package.yaml b/packages/irs_soi/ira_roth_contributions_2022/source_package.yaml index 650d1527..cf6c8b1c 100644 --- a/packages/irs_soi/ira_roth_contributions_2022/source_package.yaml +++ b/packages/irs_soi/ira_roth_contributions_2022/source_package.yaml @@ -9,19 +9,19 @@ artifact: resource_package: db resource_directory: data/irs_soi/ira_contributions manifest: manifest_roth_source_package.yaml - vintage: tax_year_{year} + vintage: tax_year_2022 extracted_at: '2026-05-08' extraction_method: xlsx whole-workbook used-range cell parse parser: xlsx_used_range artifact_year: 2022 record_sets: -- record_set_id: irs_soi.ty{year}.roth_ira_contributions.all_taxpayers +- record_set_id: irs_soi.ty2022.roth_ira_contributions.all_taxpayers provenance_class: administrative record_set_spec_id: irs_soi.roth_ira_contributions.all_taxpayers.v1 - source_record_id_prefix: irs_soi.ty{year}.roth_ira_contributions.all_taxpayers + source_record_id_prefix: irs_soi.ty2022.roth_ira_contributions.all_taxpayers sheet_name: Sheet1 period_type: tax_year - period: '{year}' + period: '2022' geography_id: 0100000US geography_level: country geography_name: United States diff --git a/packages/irs_soi/ira_traditional_contributions_2022/source_package.yaml b/packages/irs_soi/ira_traditional_contributions_2022/source_package.yaml index a8cf5b5c..aa588fd6 100644 --- a/packages/irs_soi/ira_traditional_contributions_2022/source_package.yaml +++ b/packages/irs_soi/ira_traditional_contributions_2022/source_package.yaml @@ -9,19 +9,19 @@ artifact: resource_package: db resource_directory: data/irs_soi/ira_contributions manifest: manifest_traditional_source_package.yaml - vintage: tax_year_{year} + vintage: tax_year_2022 extracted_at: '2026-05-08' extraction_method: xlsx whole-workbook used-range cell parse parser: xlsx_used_range artifact_year: 2022 record_sets: -- record_set_id: irs_soi.ty{year}.traditional_ira_contributions.all_taxpayers +- record_set_id: irs_soi.ty2022.traditional_ira_contributions.all_taxpayers provenance_class: administrative record_set_spec_id: irs_soi.traditional_ira_contributions.all_taxpayers.v1 - source_record_id_prefix: irs_soi.ty{year}.traditional_ira_contributions.all_taxpayers + source_record_id_prefix: irs_soi.ty2022.traditional_ira_contributions.all_taxpayers sheet_name: Sheet1 period_type: tax_year - period: '{year}' + period: '2022' geography_id: 0100000US geography_level: country geography_name: United States diff --git a/packages/irs_soi/state_2022/source_package.yaml b/packages/irs_soi/state_2022/source_package.yaml index 6306c117..0642c4d0 100644 --- a/packages/irs_soi/state_2022/source_package.yaml +++ b/packages/irs_soi/state_2022/source_package.yaml @@ -17,19 +17,19 @@ artifact: resource_package: db resource_directory: data/irs_soi/state_2022 manifest: manifest.yaml - vintage: tax_year_{year} + vintage: tax_year_2022 extracted_at: "2026-05-11" extraction_method: xlsx whole-workbook used-range cell parse parser: xlsx_used_range artifact_year: 2022 record_sets: - - record_set_id: irs_soi.ty{year}.state_2022.us.return_count + - record_set_id: irs_soi.ty2022.state_2022.us.return_count provenance_class: administrative record_set_spec_id: irs_soi.state_2022.us.return_count.v1 - source_record_id_prefix: irs_soi.ty{year}.state_2022.us.return_count + source_record_id_prefix: irs_soi.ty2022.state_2022.us.return_count sheet_name: Sheet1 period_type: tax_year - period: "{year}" + period: "2022" geography_id: 0100000US geography_level: country geography_name: United States @@ -66,13 +66,13 @@ record_sets: unit: count aggregation: sum expected_cell_type: number - - record_set_id: irs_soi.ty{year}.state_2022.us.adjusted_gross_income + - record_set_id: irs_soi.ty2022.state_2022.us.adjusted_gross_income provenance_class: administrative record_set_spec_id: irs_soi.state_2022.us.adjusted_gross_income.v1 - source_record_id_prefix: irs_soi.ty{year}.state_2022.us.adjusted_gross_income + source_record_id_prefix: irs_soi.ty2022.state_2022.us.adjusted_gross_income sheet_name: Sheet1 period_type: tax_year - period: "{year}" + period: "2022" geography_id: 0100000US geography_level: country geography_name: United States @@ -116,18 +116,18 @@ record_sets: adjusted gross income. This Ledger assertion treats the SOI AGI row as exactly adopting that legal concept for the tax-year source record. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2022 unit: usd aggregation: sum value_scale: 1000 expected_cell_type: number - - record_set_id: irs_soi.ty{year}.state_2022.us.eitc_three_or_more_children_returns + - record_set_id: irs_soi.ty2022.state_2022.us.eitc_three_or_more_children_returns provenance_class: administrative record_set_spec_id: irs_soi.state_2022.us.eitc_three_or_more_children_returns.v1 - source_record_id_prefix: irs_soi.ty{year}.state_2022.us.eitc_three_or_more_children_returns + source_record_id_prefix: irs_soi.ty2022.state_2022.us.eitc_three_or_more_children_returns sheet_name: Sheet1 period_type: tax_year - period: "{year}" + period: "2022" geography_id: 0100000US geography_level: country geography_name: United States @@ -173,13 +173,13 @@ record_sets: unit: count aggregation: sum expected_cell_type: number - - record_set_id: irs_soi.ty{year}.state_2022.us.eitc_three_or_more_children_amount + - record_set_id: irs_soi.ty2022.state_2022.us.eitc_three_or_more_children_amount provenance_class: administrative record_set_spec_id: irs_soi.state_2022.us.eitc_three_or_more_children_amount.v1 - source_record_id_prefix: irs_soi.ty{year}.state_2022.us.eitc_three_or_more_children_amount + source_record_id_prefix: irs_soi.ty2022.state_2022.us.eitc_three_or_more_children_amount sheet_name: Sheet1 period_type: tax_year - period: "{year}" + period: "2022" geography_id: 0100000US geography_level: country geography_name: United States diff --git a/packages/irs_soi/w2_statistics_2020/source_package.yaml b/packages/irs_soi/w2_statistics_2020/source_package.yaml index 4352948e..fdde375a 100644 --- a/packages/irs_soi/w2_statistics_2020/source_package.yaml +++ b/packages/irs_soi/w2_statistics_2020/source_package.yaml @@ -9,19 +9,19 @@ artifact: resource_package: db resource_directory: data/irs_soi/w2_statistics manifest: manifest_2020_source_package.yaml - vintage: tax_year_{year} + vintage: tax_year_2020 extracted_at: '2026-05-08' extraction_method: xlsx whole-workbook used-range cell parse parser: xlsx_used_range artifact_year: 2020 record_sets: -- record_set_id: irs_soi.ty{year}.form_w2_social_security_tips +- record_set_id: irs_soi.ty2020.form_w2_social_security_tips provenance_class: administrative record_set_spec_id: irs_soi.form_w2_social_security_tips.v1 - source_record_id_prefix: irs_soi.ty{year}.form_w2_social_security_tips + source_record_id_prefix: irs_soi.ty2020.form_w2_social_security_tips sheet_name: Table 4.B period_type: tax_year - period: '{year}' + period: '2020' geography_id: 0100000US geography_level: country geography_name: United States @@ -88,16 +88,16 @@ record_sets: item 3). Box 7 covers only tips reported to employers and excludes unreported tips picked up on Form 4137, so it is a conservative floor for tip_income. - legal_vintage: tax_year_{year} + legal_vintage: tax_year_2020 value_scale: 1000 expected_cell_type: number -- record_set_id: irs_soi.ty{year}.form_w2_401k_elective_deferrals +- record_set_id: irs_soi.ty2020.form_w2_401k_elective_deferrals provenance_class: administrative record_set_spec_id: irs_soi.form_w2_401k_elective_deferrals.v1 - source_record_id_prefix: irs_soi.ty{year}.form_w2_401k_elective_deferrals + source_record_id_prefix: irs_soi.ty2020.form_w2_401k_elective_deferrals sheet_name: Table 4.B period_type: tax_year - period: '{year}' + period: '2020' geography_id: 0100000US geography_level: country geography_name: United States @@ -132,13 +132,13 @@ record_sets: aggregation: sum value_scale: 1000 expected_cell_type: number -- record_set_id: irs_soi.ty{year}.form_w2_designated_roth_401k_contributions +- record_set_id: irs_soi.ty2020.form_w2_designated_roth_401k_contributions provenance_class: administrative record_set_spec_id: irs_soi.form_w2_designated_roth_401k_contributions.v1 - source_record_id_prefix: irs_soi.ty{year}.form_w2_designated_roth_401k_contributions + source_record_id_prefix: irs_soi.ty2020.form_w2_designated_roth_401k_contributions sheet_name: Table 4.B period_type: tax_year - period: '{year}' + period: '2020' geography_id: 0100000US geography_level: country geography_name: United States diff --git a/tests/test_chronicle_source_package.py b/tests/test_chronicle_source_package.py index 60332610..6c7963b5 100644 --- a/tests/test_chronicle_source_package.py +++ b/tests/test_chronicle_source_package.py @@ -1392,6 +1392,36 @@ def test_bea_regional_state_personal_income_components_build_2024_facts(): assert validate_consumer_fact_contract(facts).valid +def test_bea_regional_state_personal_income_components_read_the_requested_year(): + # SAINC5N__ALL_AREAS_1998_2025.csv carries one column per year, 1998 (I) + # to 2025 (AJ). A --year 2023 build must read the 2023 column, not + # restamp the 2024 column as 2023 (chronicle#117). + package = load_source_package("bea-regional-state-personal-income-components-2024") + facts_2023 = {fact.source_record_id: fact for fact in package.build_facts(2023)} + facts_2025 = {fact.source_record_id: fact for fact in package.build_facts(2025)} + + assert len(facts_2023) == len(facts_2025) == 416 + assert {fact.period.value for fact in facts_2023.values()} == {2023} + us_income = facts_2023["bea_regional.cy2023.state_personal_income.us.amount"] + assert us_income.value == 23_577_208_000_000 + assert us_income.layout.source_column_id == "2023" + assert ( + facts_2023["bea_regional.cy2023.state_wages_salaries.ca.amount"].value + == 1_666_632_928_000 + ) + assert ( + facts_2023["bea_regional.cy2023.state_residence_adjustment.ca.amount"].value + == -2_381_165_000 + ) + assert ( + facts_2025["bea_regional.cy2025.state_personal_income.us.amount"].value + == 26_109_831_238_000 + ) + for year in (1997, 2026): + with pytest.raises(ValueError, match=f"No source artifact for year {year}"): + package.build_facts(year) + + def test_bea_regional_state_personal_income_components_validate_counts(): report = validate_source_package( "bea-regional-state-personal-income-components-2024", From b97ee94cfc63f00a41037ca7b752c4d111ca1ff1 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 25 Sep 2026 18:29:29 -0400 Subject: [PATCH 2/6] Refuse pinned-artifact builds that would only relabel another year artifact_year pins which file a package reads, while each {year} label renders from --year. Add a structural guard so a pinned build can no longer stamp one year's publisher data as another year. artifact_year_restamp_issues(package, year) compares the build at Y with the build at the artifact year A. A record set whose selection is the same at Y and A while any other rendered field differs is a restamp. Selection means the artifact's parser, sheet, archive member and rendered selected_rows, plus the record set's sheet, row and column addresses, value scaling and guard cells. The other fields include period, record ids, vintage, source_table, legal_vintage, filters, constraints and notes. - SourcePackage.build_source_rows, build_source_cells and build_source_record_set_specs raise ArtifactYearRestampError(ValueError). Every other build path goes through them. - validate_source_package reports error code artifact_year_restamp. - A default build-bundle treats the code as "no data for this year": it skips the package and gives the reason in skipped_sources. An explicit --source fails the bundle (source_suite_build_failed, exit 1). Year-selecting pinned packages still build off-year. That covers selected_rows on {year}, column_by_year, sheet_name_by_year and guards that follow the year. The check compiles specs only and never parses the artifact. Over all 219 pinned packages at A-1, A+1, 2023 and every microcosm scope year (585 pairs), it flags the same 22 pairs in 10 packages as the design prototype on origin/main's package YAML, and 0 pairs after the package fix. Tests: synthetic refuse/allow/literal cases, validate and CLI exit codes, default-bundle skip and explicit-source failure, an exhaustive property test over 1,280 synthetic cases (Hypothesis is not a dependency), a real-tree guard check, and a real-bytes differential: origin/main's IRA and BEA regional YAML restamp without the guard and are refused with it. Co-Authored-By: Claude Opus 5.5 --- chronicle/bundle.py | 20 +- chronicle/source_package.py | 289 ++++++- docs/agent-source-package-harness.md | 33 + tests/test_chronicle_artifact_year_restamp.py | 798 ++++++++++++++++++ 4 files changed, 1136 insertions(+), 4 deletions(-) create mode 100644 tests/test_chronicle_artifact_year_restamp.py diff --git a/chronicle/bundle.py b/chronicle/bundle.py index 649eebec..1d1fab7e 100644 --- a/chronicle/bundle.py +++ b/chronicle/bundle.py @@ -12,6 +12,7 @@ from chronicle.dimension_labels import dimension_label_issues, dimension_labels_by_id from chronicle.epoch import canonicalize_key, schema_id from chronicle.source_package import ( + ARTIFACT_YEAR_RESTAMP_CODE, SOURCE_PACKAGE_ALIASES, assert_alias_map_covers_packages, validate_source_package, @@ -801,10 +802,18 @@ def _resolve_bundle_sources( if report.valid or explicit or not _source_unavailable_for_year(report, year): build_sources.append(source) continue + restamp = any( + error.code == ARTIFACT_YEAR_RESTAMP_CODE for error in report.errors + ) skipped_sources.append( SkippedSourceReport( source=source, - reason="source package is not available for requested year", + reason=( + "source package pins another year's artifact and would only " + f"relabel it as {year} ({ARTIFACT_YEAR_RESTAMP_CODE})" + if restamp + else "source package is not available for requested year" + ), validation=report.to_dict(), ) ) @@ -812,13 +821,20 @@ def _resolve_bundle_sources( def _source_unavailable_for_year(report: Any, year: int) -> bool: + """Whether every validation error says the package has no data for ``year``. + + That is either no artifact or column for the year, or a pinned artifact that + the year would only relabel (``artifact_year_restamp``): the package holds + no facts for the requested year either way. + """ unavailable_messages = {repr(str(year)), f"No source artifact for year {year}"} unavailable_codes = { "record_set_compile_failed", "source_artifact_unavailable", } return bool(report.errors) and all( - error.code in unavailable_codes and error.message in unavailable_messages + error.code == ARTIFACT_YEAR_RESTAMP_CODE + or (error.code in unavailable_codes and error.message in unavailable_messages) for error in report.errors ) diff --git a/chronicle/source_package.py b/chronicle/source_package.py index f411de37..d2c76ef9 100644 --- a/chronicle/source_package.py +++ b/chronicle/source_package.py @@ -4,7 +4,7 @@ import hashlib import os -from dataclasses import dataclass, replace +from dataclasses import asdict, dataclass, field, replace from importlib.resources import files from io import BytesIO from pathlib import Path @@ -1227,9 +1227,18 @@ class SourcePackage: # (chronicle#261); see chronicle.dimension_labels for the other sources. dimension_labels: dict[str, str] | None = None dimension_value_labels: dict[str, dict[str, str]] | None = None + # Memo of artifact_year_restamp_issues by build year. The verdict depends + # only on the package's declarations, which a loaded package never changes. + _restamp_issues_by_year: dict[Any, tuple[SourcePackageIssue, ...]] = field( + default_factory=dict, + init=False, + repr=False, + compare=False, + ) def build_source_rows(self, year: int) -> list[SourceRow]: """Build full source rows for row-oriented artifacts.""" + self._require_year_reads_its_own_data(year) return self.artifact.build_source_rows(year) def build_source_cells( @@ -1239,6 +1248,7 @@ def build_source_cells( source_rows: list[SourceRow] | None = None, ) -> list[SourceCell]: """Build whole-artifact source cells for this package.""" + self._require_year_reads_its_own_data(year) return self.artifact.build_source_cells(year, source_rows=source_rows) def build_source_record_set_specs( @@ -1246,8 +1256,19 @@ def build_source_record_set_specs( year: int, ) -> list[SourceRecordSetSpec]: """Build compact record-set specs for this package.""" + self._require_year_reads_its_own_data(year) + return self._compile_record_set_specs(year) + + def _compile_record_set_specs(self, year: int) -> list[SourceRecordSetSpec]: + """Compile record-set specs without the artifact-year restamp guard.""" return [record_set.to_record_set_spec(year) for record_set in self.record_sets] + def _require_year_reads_its_own_data(self, year: int) -> None: + """Refuse a build whose year would only relabel the pinned artifact.""" + issues = artifact_year_restamp_issues(self, year) + if issues: + raise ArtifactYearRestampError(self, year, issues) + def build_source_regions(self, year: int): """Build source-region specs implied by the package record sets.""" regions = [] @@ -1322,6 +1343,267 @@ def build_facts( ) +ARTIFACT_YEAR_RESTAMP_CODE = "artifact_year_restamp" + +# The declarations that decide what a build reads: which cells, the value taken +# from them, and what each guard cell must hold. A pinned package always reads +# the file of its ``artifact_year``, so when these render the same at two build +# years, both builds read the same cells. Every other compiled field (period, +# record ids, legal_vintage, filters, constraints, geography, concept and layout +# labels) only names or dates the facts built from those cells. +_ARTIFACT_SELECTION_FIELDS = ( + "parser", + "archive_member", + "sheets", + "delimiter", + "header_row", +) +_ARTIFACT_LABEL_FIELDS = ("vintage", "source_table") +_RECORD_SET_SELECTION_FIELDS = ("sheet_name",) +# A row's label doubles as its row-header guard when expected_row_header is +# unset. It counts as a label here, so a {year} row label over a fixed row is +# refused as a restamp rather than left to fail that header check. +_ROW_SELECTION_FIELDS = ( + "row_number", + "row_end_number", + "column", + "value_scale", + "expected_row_header", + "expected_row_header_column", + "expected_column_header_row", + "expected_column_header", +) +_MEASURE_SELECTION_FIELDS = ( + "column", + "divisor_column", + "value_scale", + "round_to", + "expected_cell_type", + "expected_column_header_row", + "expected_column_header", +) +_RESTAMP_MESSAGE_MOVES = 6 + + +class ArtifactYearRestampError(ValueError): + """A pinned-artifact build whose year would only relabel another year's data. + + ``artifact_year`` pins which file a package reads. When a build at another + year selects exactly the cells the ``artifact_year`` build selects, every + ``{year}`` label it renders (period, record ids, vintage, legal_vintage) + would restamp that file's facts as the requested year. + """ + + def __init__( + self, + package: SourcePackage, + year: int, + issues: list[SourcePackageIssue] | tuple[SourcePackageIssue, ...], + ) -> None: + self.package_id = package.package_id + self.artifact_year = package.artifact.artifact_year + self.year = year + self.issues = tuple(issues) + more = len(self.issues) - 1 + suffix = f" ({more} more record set(s) affected.)" if more else "" + super().__init__(f"{self.issues[0].message}{suffix}") + + +def artifact_year_restamp_issues( + package: SourcePackage, + year: int, +) -> list[SourcePackageIssue]: + """Return restamp issues for building a pinned package at ``year``. + + A package pinned to ``artifact_year`` A reads A's file at every build year. + A build at year Y != A is a restamp when a record set selects the same cells + at Y as at A (same artifact and record-set selection) while any label the + build renders differs, such as the period, record ids, artifact vintage or + legal_vintage. Packages whose selection follows the year + (``column_by_year``, ``sheet_name_by_year``, ``selected_rows`` that filter + on ``{year}``, or a guard cell whose expected value follows the year) pass; + a build at a year their file does not cover then fails with its own error. + The check compiles record-set specs only and never parses the artifact. + """ + artifact_year = package.artifact.artifact_year + if ( + artifact_year is None + or not isinstance(year, int) + or isinstance(year, bool) + or year == artifact_year + ): + return [] + cached = package._restamp_issues_by_year.get(year) + if cached is None: + cached = tuple(_restamp_issues(package, year, artifact_year)) + package._restamp_issues_by_year[year] = cached + return list(cached) + + +def _restamp_issues( + package: SourcePackage, + year: int, + artifact_year: int, +) -> list[SourcePackageIssue]: + if not _declarations_depend_on_year(package): + return [] # every declaration renders the same at every year + artifact = package.artifact + try: + if _artifact_selection(artifact, year) != _artifact_selection( + artifact, artifact_year + ): + return [] + specs_at_year = package._compile_record_set_specs(year) + specs_at_artifact_year = package._compile_record_set_specs(artifact_year) + except (KeyError, TypeError, ValueError): + # A declaration that renders or compiles at only one of the two years + # selects by year (for example a column_by_year without that year). A + # build at a year that cannot compile fails with its own error. + return [] + artifact_moves = [] + for name in _ARTIFACT_LABEL_FIELDS: + at_artifact_year = _render_string(getattr(artifact, name), year=artifact_year) + at_year = _render_string(getattr(artifact, name), year=year) + if at_artifact_year != at_year: + artifact_moves.append((f"artifact.{name}", at_artifact_year, at_year)) + issues = [] + for spec_at_year, spec_at_artifact_year in zip( + specs_at_year, + specs_at_artifact_year, + strict=True, + ): + if _record_set_selection(spec_at_year) != _record_set_selection( + spec_at_artifact_year + ): + continue + moves = [*artifact_moves] + if spec_at_year != spec_at_artifact_year: + moves.extend( + _changed_leaves(asdict(spec_at_artifact_year), asdict(spec_at_year)) + ) + if not moves: + continue + issues.append( + SourcePackageIssue( + code=ARTIFACT_YEAR_RESTAMP_CODE, + message=_restamp_message( + package, + year=year, + artifact_year=artifact_year, + record_set_id=spec_at_artifact_year.record_set_id, + moves=moves, + ), + record_set_id=spec_at_year.record_set_id, + ) + ) + return issues + + +def _declarations_depend_on_year(package: SourcePackage) -> bool: + """Whether any declaration renders ``{year}``/``{filing_year}`` or is by-year.""" + return _mentions_year(asdict(package.artifact)) or any( + _mentions_year(record_set.payload) for record_set in package.record_sets + ) + + +def _mentions_year(value: Any) -> bool: + if isinstance(value, str): + return "{year" in value or "{filing_year" in value + if isinstance(value, dict): + return any( + (isinstance(key, str) and key.endswith("_by_year")) or _mentions_year(item) + for key, item in value.items() + ) + if isinstance(value, list | tuple): + return any(_mentions_year(item) for item in value) + return False + + +def _artifact_selection(artifact: SourceArtifactSpec, year: int) -> tuple[Any, ...]: + return ( + tuple(getattr(artifact, name) for name in _ARTIFACT_SELECTION_FIELDS), + _render_string(artifact.sheet_name, year=year) if artifact.sheet_name else None, + tuple( + {key: str(_render_value(value, year=year)) for key, value in row.items()} + for row in artifact.selected_rows + ), + ) + + +def _record_set_selection(spec: SourceRecordSetSpec) -> tuple[Any, ...]: + return ( + tuple(getattr(spec, name) for name in _RECORD_SET_SELECTION_FIELDS), + tuple( + ( + tuple(getattr(row, name) for name in _ROW_SELECTION_FIELDS), + tuple( + (guard.column, guard.expected_value, guard.row) + for guard in row.guard_cells + ), + tuple( + (guard.column, guard.expected_values) + for guard in row.range_label_guards + ), + ) + for row in spec.rows + ), + tuple( + tuple(getattr(measure, name) for name in _MEASURE_SELECTION_FIELDS) + for measure in spec.measures + ), + ) + + +def _changed_leaves( + before: Any, + after: Any, + path: str = "", +) -> list[tuple[str, Any, Any]]: + if isinstance(before, dict) and isinstance(after, dict): + changes = [] + for key in dict.fromkeys([*before, *after]): + child = f"{path}.{key}" if path else str(key) + changes.extend(_changed_leaves(before.get(key), after.get(key), child)) + return changes + if ( + isinstance(before, list | tuple) + and isinstance(after, list | tuple) + and len(before) == len(after) + ): + changes = [] + for index, (item_before, item_after) in enumerate( + zip(before, after, strict=True) + ): + changes.extend(_changed_leaves(item_before, item_after, f"{path}[{index}]")) + return changes + return [] if before == after else [(path, before, after)] + + +def _restamp_message( + package: SourcePackage, + *, + year: int, + artifact_year: int, + record_set_id: str, + moves: list[tuple[str, Any, Any]], +) -> str: + shown = "; ".join( + f"{path} {before!r} -> {after!r}" + for path, before, after in moves[:_RESTAMP_MESSAGE_MOVES] + ) + if len(moves) > _RESTAMP_MESSAGE_MOVES: + shown += f"; and {len(moves) - _RESTAMP_MESSAGE_MOVES} more" + return ( + f"Source package {package.package_id!r} pins artifact_year " + f"{artifact_year}, so a build at year {year} reads the same cells as " + f"the {artifact_year} build of record set {record_set_id!r} but " + f"relabels them: {shown}. Write year labels as literals (for example " + f"period: '{artifact_year}'), or select the year's data with " + "column_by_year, sheet_name_by_year or a {year}-templated " + "selected_rows filter." + ) + + def load_source_package(source: str | Path) -> SourcePackage: """Load a declarative source package from an alias, directory, or YAML file.""" path = resolve_source_package_path(source) @@ -1454,7 +1736,7 @@ def validate_source_package( ) try: - record_sets = package.build_source_record_set_specs(year) + record_sets = package._compile_record_set_specs(year) except (KeyError, TypeError, ValueError) as exc: errors.append( SourcePackageIssue( @@ -1471,6 +1753,9 @@ def validate_source_package( warnings=tuple(warnings), ) + # A pinned package that would only relabel its artifact at this year is + # invalid here; the build methods refuse it with ArtifactYearRestampError. + errors.extend(artifact_year_restamp_issues(package, year)) counts["record_set_count"] = len(record_sets) record_set_ids: dict[str, list[int]] = {} source_record_ids: dict[str, list[int]] = {} diff --git a/docs/agent-source-package-harness.md b/docs/agent-source-package-harness.md index d01b5e81..bdcf7e76 100644 --- a/docs/agent-source-package-harness.md +++ b/docs/agent-source-package-harness.md @@ -825,6 +825,39 @@ county names to twenty characters, and ONS writes "Yorkshire and The Humber" where HMRC writes "the" — Chronicle keeps each publisher's text and warns, the same way it does for a groupby value's row labels. +### Pinned artifacts and year labels + +`artifact.artifact_year` pins the file. A package with `artifact_year: 2022` +reads the manifest's 2022 file at every `--year`, but each `{year}` and +`{filing_year}` in the package still renders from `--year`. So a pinned package +writes its labels as literals (`period: '2022'`, `record_set_id: +irs_soi.ty2022.…`, `vintage: tax_year_2022`, `legal_vintage: tax_year_2022`) +unless `--year` also changes what it reads, through `column_by_year`, +`sheet_name_by_year` or a `selected_rows` filter on `{year}`. Otherwise a +`--year 2023` build would stamp the 2022 file's facts as 2023 (chronicle#117). +A consumer that wants a value at another period declares that alignment +itself; Chronicle records the publisher's period. + +The harness enforces the rule. For a build at year Y other than the artifact +year A, it compares what each record set selects at Y and at A: the artifact's +parser, sheet, archive member and rendered `selected_rows`, and the record +set's sheet, rows, columns, value scaling and guard cells. If a record set +selects the same cells at both years while any other rendered field differs +(period, record ids, `vintage`, `source_table`, `legal_vintage`, filters, +constraints or notes), then: + +- the build refuses with `ArtifactYearRestampError`; +- `validate-package` reports `artifact_year_restamp`; +- a default `build-bundle` skips the package for that year and gives the reason + under `skipped_sources`; +- an explicit `build-bundle --source` fails the bundle. + +The check compiles record-set specs only and never parses the artifact. +Year-selecting pinned packages still build at other years and read that +year's column, sheet or rows. A year the file does not cover fails with its +own error, such as `No source artifact for year 2027` or a selected-row +mismatch. + Agents may add new package directories and YAML specs. They should not modify `chronicle.core`, `chronicle.database`, or `chronicle.suite` unless the package cannot be expressed in the current contract and the failure is documented in the build diff --git a/tests/test_chronicle_artifact_year_restamp.py b/tests/test_chronicle_artifact_year_restamp.py new file mode 100644 index 00000000..2081e176 --- /dev/null +++ b/tests/test_chronicle_artifact_year_restamp.py @@ -0,0 +1,798 @@ +"""The artifact-year restamp guard (PolicyEngine/chronicle#117). + +``artifact_year`` pins which file a source package reads: every build year reads +the artifact year's file. A build at another year that selects exactly the cells +the artifact-year build selects, but renders different labels (period, record +ids, vintage, legal_vintage, ...), would restamp one year's data as another +year. The guard refuses those builds and lets year-selecting packages through. + +Invariant under test: for a package pinned to artifact year A and any build +year Y, the build at Y either raises, or emits no fact whose value, +``source_cell_keys`` and ``source_row_keys`` equal a fact of the A build while +any other field differs. +""" + +from __future__ import annotations + +import functools +import hashlib +import itertools +import json +from io import BytesIO +from pathlib import Path + +import openpyxl +import pytest +import yaml + +import chronicle.bundle as bundle_module +import chronicle.source_package as source_package_module +from chronicle.bundle import build_bundle +from chronicle.harness import main as harness_main +from chronicle.source_package import ( + ARTIFACT_YEAR_RESTAMP_CODE, + ArtifactYearRestampError, + DeclarativeRecordSet, + SourceArtifactSpec, + SourcePackage, + artifact_year_restamp_issues, + discover_source_package_dirs, + load_source_package, + validate_source_package, +) +from chronicle.store import fact_to_mapping + +REPO_ROOT = Path(__file__).resolve().parents[1] +ARTIFACT_YEAR = 2021 +BUILD_YEARS = (2020, 2021, 2022, 2023, 2024) +# Years the synthetic publisher file covers. 2023 is deliberately absent, so a +# year-selecting package has nothing to read at 2023 and must fail loudly. +FILE_YEARS = (2020, 2021, 2022, 2024) +VALUES = {2020: 100, 2021: 110, 2022: 120, 2024: 140} +WIDE_COLUMNS = {2020: "C", 2021: "D", 2022: "E", 2024: "F"} +# Every label a synthetic package can template with {year}. +LABEL_FIELDS = ( + "period", + "record_ids", + "legal_vintage", + "artifact_vintage", + "source_table", + "filter", +) +# How a synthetic package reads the year: not at all (a fixed column of the +# pinned file), or by column, by source row or by sheet. +MECHANISMS = ("none", "column_by_year", "selected_rows", "sheet_name_by_year") + + +# --- synthetic packages ----------------------------------------------------- + + +def _wide_csv() -> bytes: + header = ",".join(["Item", "Unit", *(str(year) for year in FILE_YEARS)]) + row = ",".join(["Returns", "count", *(str(VALUES[y]) for y in FILE_YEARS)]) + return f"{header}\n{row}\n".encode() + + +def _long_csv() -> bytes: + lines = ["Item,Period,Value"] + lines.extend(f"Returns,{year},{VALUES[year]}" for year in FILE_YEARS) + return ("\n".join(lines) + "\n").encode() + + +@functools.cache +def _sheets_xlsx() -> bytes: + workbook = openpyxl.Workbook() + workbook.remove(workbook.active) + for year in FILE_YEARS: + sheet = workbook.create_sheet(f"TY{year}") + sheet.append(["Item", "Unit", "Value"]) + sheet.append(["Returns", "count", VALUES[year]]) + buffer = BytesIO() + workbook.save(buffer) + return buffer.getvalue() + + +def _artifact_bytes(mechanism: str) -> tuple[bytes, str]: + if mechanism in {"none", "column_by_year"}: + return _wide_csv(), "synthetic_wide.csv" + if mechanism == "selected_rows": + return _long_csv(), "synthetic_long.csv" + return _sheets_xlsx(), "synthetic_sheets.xlsx" + + +def _payload(templated: frozenset[str], mechanism: str) -> dict: + """A source-package payload pinned to ARTIFACT_YEAR. + + Each field in ``templated`` renders ``{year}``; every other label is the + literal artifact year. ``mechanism`` decides how the build reads the year. + """ + + def label(field: str, pattern: str) -> str: + year = "{year}" if field in templated else str(ARTIFACT_YEAR) + return pattern.replace("@", year) + + artifact = { + "source_name": "test", + "source_table": label("source_table", "Synthetic returns, tax year @"), + "vintage": label("artifact_vintage", "tax_year_@"), + "extracted_at": "2026-09-25", + "extraction_method": "synthetic test artifact", + "artifact_year": ARTIFACT_YEAR, + } + measure = { + "measure_id": "count", + "label": "Returns", + "ordinal": 0, + "concept": "test.returns", + "unit": "count", + "aggregation": "sum", + "legal_vintage": label("legal_vintage", "tax_year_@"), + } + record_set = { + "record_set_id": label("record_ids", "test.ty@.returns"), + "provenance_class": "administrative", + "record_set_spec_id": "test.returns.v1", + "source_record_id_prefix": label("record_ids", "test.ty@.returns"), + "period_type": "tax_year", + "period": label("period", "@"), + "geography_id": "0100000US", + "geography_level": "country", + "entity": "tax_unit", + "domain": "test", + "groupby_dimension": "test.item", + # A universe filter and its explicit constraint: labels on the fact. + "shared_filters": {"tax_year": label("filter", "@")}, + "shared_constraints": [ + { + "variable": "tax_year", + "operator": "==", + "value": label("filter", "@"), + "label": "Tax year", + } + ], + "rows": [ + { + "value_id": "returns", + "label": "Returns", + "ordinal": 0, + "row_number": 2, + "expected_row_header_column": "A", + "table_record_kind": "total", + } + ], + "measures": [measure], + } + if mechanism in {"none", "column_by_year"}: + artifact.update(parser="delimited_text_full_rows", sheet_name="synthetic") + record_set["sheet_name"] = "synthetic" + if mechanism == "none": + measure["column"] = WIDE_COLUMNS[ARTIFACT_YEAR] + else: + measure["column_by_year"] = dict(WIDE_COLUMNS) + elif mechanism == "selected_rows": + artifact.update( + parser="delimited_text_full_rows", + sheet_name="synthetic", + selected_rows=[{"Item": "Returns", "Period": "{year}"}], + ) + record_set["sheet_name"] = "synthetic" + measure["column"] = "C" + elif mechanism == "sheet_name_by_year": + artifact.update(parser="xlsx_used_range") + record_set["sheet_name_by_year"] = {year: f"TY{year}" for year in FILE_YEARS} + measure["column"] = "C" + else: + raise AssertionError(mechanism) + return { + "schema_version": "ledger.source_package.v1", + "package_id": "test-artifact-year-restamp", + "dimension_labels": {"test.item": "Test item"}, + "artifact": artifact, + "record_sets": [record_set], + } + + +class _PinnedInlineArtifact(SourceArtifactSpec): + """A pinned artifact held in memory: the same bytes at every build year.""" + + def __init__(self, *, content: bytes, filename: str, **fields) -> None: + super().__init__( + resource_package="unused", + resource_directory="unused", + manifest="unused", + **fields, + ) + object.__setattr__(self, "_content", content) + object.__setattr__(self, "_filename", filename) + + def _artifact_content(self, year: int) -> tuple[bytes, str, str, dict[str, str]]: + key = f"raw/test/test-artifact-year-restamp/{self.artifact_year}/x" + return ( + self._content, + self._filename, + f"https://example.test/{self._filename}", + {"bucket": "test", "key": key, "uri": f"r2://test/{key}"}, + ) + + +def _package(templated=frozenset(), mechanism="none") -> SourcePackage: + payload = _payload(frozenset(templated), mechanism) + artifact = dict(payload["artifact"]) + selected_rows = tuple(artifact.pop("selected_rows", ())) + content, filename = _artifact_bytes(mechanism) + return SourcePackage( + package_id=payload["package_id"], + label=None, + artifact=_PinnedInlineArtifact( + content=content, + filename=filename, + selected_rows=selected_rows, + **artifact, + ), + record_sets=tuple(DeclarativeRecordSet(rs) for rs in payload["record_sets"]), + package_path=Path("synthetic"), + dimension_labels=payload["dimension_labels"], + ) + + +def _write_package( + root: Path, + monkeypatch: pytest.MonkeyPatch, + templated=frozenset(), + mechanism: str = "none", + *, + name: str = "package", +) -> Path: + """Write a synthetic package and its pinned artifact to disk; return its dir.""" + resource = ( + "restamp_fixture_" + hashlib.sha256(str(root / name).encode()).hexdigest()[:16] + ) + resource_dir = root / "resources" / resource + data_dir = resource_dir / "data" / "synthetic" + data_dir.mkdir(parents=True) + (resource_dir / "__init__.py").write_text("", encoding="utf-8") + content, filename = _artifact_bytes(mechanism) + (data_dir / filename).write_bytes(content) + manifest = { + "files": { + ARTIFACT_YEAR: { + "filename": filename, + "source_url": f"https://example.test/{filename}", + "sha256": hashlib.sha256(content).hexdigest(), + "storage": { + "r2": { + "provider": "r2", + "bucket": "test", + "key": f"raw/test/{name}/{ARTIFACT_YEAR}/{filename}", + "uri": f"r2://test/raw/test/{name}/{ARTIFACT_YEAR}/{filename}", + } + }, + } + } + } + (data_dir / "manifest.yaml").write_text(yaml.safe_dump(manifest), encoding="utf-8") + monkeypatch.syspath_prepend(str(root / "resources")) + payload = _payload(frozenset(templated), mechanism) + payload["package_id"] = f"test-artifact-year-restamp-{name}" + payload["artifact"].update( + resource_package=resource, + resource_directory="data/synthetic", + manifest="manifest.yaml", + ) + package_dir = root / "packages" / name + package_dir.mkdir(parents=True) + (package_dir / "source_package.yaml").write_text( + yaml.safe_dump(payload, sort_keys=False), + encoding="utf-8", + ) + return package_dir + + +# --- fact comparison --------------------------------------------------------- + + +def _serialized(fact) -> str: + return json.dumps(fact_to_mapping(fact), sort_keys=True, default=str) + + +def _lineage(fact) -> tuple: + return ( + json.dumps(fact.value, default=str), + tuple(fact.source_cell_keys), + tuple(fact.source_row_keys or ()), + ) + + +def _restamped_facts(facts_at_year, facts_at_artifact_year) -> list: + """Facts that repeat an artifact-year fact's value and lineage but not its labels.""" + by_lineage: dict[tuple, set[str]] = {} + for fact in facts_at_artifact_year: + by_lineage.setdefault(_lineage(fact), set()).add(_serialized(fact)) + return [ + fact + for fact in facts_at_year + if _lineage(fact) in by_lineage + and _serialized(fact) not in by_lineage[_lineage(fact)] + ] + + +def _unguarded_facts(package: SourcePackage, year: int): + """build_facts without the guard; None when the build fails for another reason.""" + original = source_package_module.artifact_year_restamp_issues + source_package_module.artifact_year_restamp_issues = lambda package, year: [] + try: + return package.build_facts(year) + except (KeyError, TypeError, ValueError): + return None + finally: + source_package_module.artifact_year_restamp_issues = original + + +# --- unit tests: refuse ------------------------------------------------------ + + +def test_pinned_package_with_year_templated_period_refuses_off_year_builds(): + package = _package({"period", "record_ids", "legal_vintage", "artifact_vintage"}) + + facts = package.build_facts(ARTIFACT_YEAR) + assert [fact.period.value for fact in facts] == [ARTIFACT_YEAR] + assert facts[0].value == VALUES[ARTIFACT_YEAR] + + for build in ( + package.build_facts, + package.build_source_rows, + package.build_source_cells, + package.build_source_record_set_specs, + package.build_source_record_specs, + package.build_source_regions, + package.build_source_records, + ): + with pytest.raises(ArtifactYearRestampError) as raised: + build(ARTIFACT_YEAR + 3) + assert raised.value.package_id == "test-artifact-year-restamp" + assert raised.value.artifact_year == ARTIFACT_YEAR + assert raised.value.year == ARTIFACT_YEAR + 3 + assert [issue.code for issue in raised.value.issues] == [ + ARTIFACT_YEAR_RESTAMP_CODE + ] + # The error is a ValueError, so existing `except ValueError` paths (bundle + # source-suite failures, CLI) report it rather than crash differently. + assert issubclass(ArtifactYearRestampError, ValueError) + + +def test_restamp_message_names_the_package_years_and_moved_labels(): + package = _package({"period", "record_ids", "legal_vintage", "artifact_vintage"}) + + (issue,) = artifact_year_restamp_issues(package, 2024) + + assert issue.code == ARTIFACT_YEAR_RESTAMP_CODE + assert issue.record_set_id == "test.ty2024.returns" + message = issue.message + assert "'test-artifact-year-restamp' pins artifact_year 2021" in message + assert "a build at year 2024 reads the same cells as the 2021 build" in message + assert "artifact.vintage 'tax_year_2021' -> 'tax_year_2024'" in message + assert "record_set_id 'test.ty2021.returns' -> 'test.ty2024.returns'" in message + assert "period 2021 -> 2024" in message + assert "measures[0].legal_vintage 'tax_year_2021' -> 'tax_year_2024'" in message + assert "column_by_year" in message + + +@pytest.mark.parametrize( + "templated", + [{"legal_vintage"}, {"artifact_vintage"}, {"source_table"}, {"filter"}], + ids=lambda fields: "-".join(sorted(fields)), +) +def test_a_single_templated_label_is_enough_to_refuse(templated): + package = _package(templated) + + assert artifact_year_restamp_issues(package, 2024) + with pytest.raises(ArtifactYearRestampError): + package.build_facts(2024) + # The period is literal, so only the templated label would have moved. + unguarded = _unguarded_facts(package, 2024) + assert [fact.period.value for fact in unguarded] == [ARTIFACT_YEAR] + assert _restamped_facts(unguarded, package.build_facts(ARTIFACT_YEAR)) + + +def test_validate_reports_restamp_as_an_error(tmp_path, monkeypatch, capsys): + package_dir = _write_package( + tmp_path, + monkeypatch, + {"period", "record_ids", "legal_vintage", "artifact_vintage"}, + ) + + at_artifact_year = validate_source_package(package_dir, year=ARTIFACT_YEAR) + off_year = validate_source_package(package_dir, year=2024) + + assert at_artifact_year.valid, at_artifact_year.to_dict() + assert not off_year.valid + assert [error.code for error in off_year.errors] == [ARTIFACT_YEAR_RESTAMP_CODE] + # Counts still describe the package, so the report stays useful. + assert off_year.counts == at_artifact_year.counts + assert harness_main(["validate-package", str(package_dir), "--year", "2024"]) == 1 + payload = json.loads(capsys.readouterr().out) + assert payload["errors"][0]["code"] == ARTIFACT_YEAR_RESTAMP_CODE + + +# --- unit tests: allow ------------------------------------------------------- + + +@pytest.mark.parametrize( + "mechanism", + ["column_by_year", "selected_rows", "sheet_name_by_year"], +) +def test_year_selecting_pinned_package_builds_off_year_with_that_years_value( + mechanism, +): + package = _package(set(LABEL_FIELDS), mechanism) + + assert artifact_year_restamp_issues(package, 2024) == [] + facts = package.build_facts(2024) + base = package.build_facts(ARTIFACT_YEAR) + + assert [(fact.period.value, fact.value) for fact in facts] == [(2024, 140)] + assert [(fact.period.value, fact.value) for fact in base] == [(2021, 110)] + # A different column or sheet gives different source cells; a different + # selected source row lands on the same virtual cell but a different + # source row. Either way the lineage differs from the artifact-year fact. + assert _lineage(facts[0]) != _lineage(base[0]) + assert _restamped_facts(facts, base) == [] + + +@pytest.mark.parametrize( + "mechanism", + ["column_by_year", "selected_rows", "sheet_name_by_year"], +) +def test_year_selecting_package_fails_loudly_for_a_year_its_file_lacks(mechanism): + package = _package(set(LABEL_FIELDS), mechanism) + + assert artifact_year_restamp_issues(package, 2023) == [] + with pytest.raises(ValueError) as raised: + package.build_facts(2023) + assert not isinstance(raised.value, ArtifactYearRestampError) + + +def test_literal_labels_build_the_artifact_year_facts_at_any_year(): + package = _package() + base = [_serialized(fact) for fact in package.build_facts(ARTIFACT_YEAR)] + + for year in BUILD_YEARS: + assert artifact_year_restamp_issues(package, year) == [] + assert [_serialized(fact) for fact in package.build_facts(year)] == base + + +def test_guard_ignores_unpinned_packages_and_the_artifact_year_itself(): + package = _package({"period"}) + unpinned = SourcePackage( + package_id=package.package_id, + label=None, + artifact=_PinnedInlineArtifact( + content=package.artifact._content, + filename=package.artifact._filename, + **{ + **{ + name: getattr(package.artifact, name) + for name in ( + "source_name", + "source_table", + "vintage", + "extracted_at", + "extraction_method", + "parser", + "sheet_name", + ) + }, + "artifact_year": None, + }, + ), + record_sets=package.record_sets, + package_path=package.package_path, + ) + + assert artifact_year_restamp_issues(package, ARTIFACT_YEAR) == [] + assert artifact_year_restamp_issues(unpinned, 2024) == [] + assert unpinned.build_facts(2024)[0].period.value == 2024 + # Non-integer build keys (release labels, chronicle#79) render no {year}. + assert artifact_year_restamp_issues(package, "release_2026") == [] + + +def test_guard_verdict_is_cached_per_package_and_year(): + package = _package({"period"}) + + first = artifact_year_restamp_issues(package, 2024) + assert package._restamp_issues_by_year == {2024: tuple(first)} + assert artifact_year_restamp_issues(package, 2024) == first + assert artifact_year_restamp_issues(_package(), 2024) == [] + + +# --- bundles ----------------------------------------------------------------- + + +def _bundle_rows(output_dir: Path) -> list[dict]: + text = (output_dir / "consumer_facts.jsonl").read_text(encoding="utf-8") + return [json.loads(line) for line in text.splitlines() if line.strip()] + + +def test_default_bundle_skips_a_restamping_package_and_says_why( + tmp_path, + monkeypatch, +): + restamp = str( + _write_package(tmp_path, monkeypatch, {"period", "record_ids"}, name="r") + ) + literal = str(_write_package(tmp_path, monkeypatch, name="literal")) + monkeypatch.setattr(bundle_module, "DEFAULT_BUNDLE_SOURCES", (restamp, literal)) + monkeypatch.setattr(bundle_module, "assert_alias_map_covers_packages", lambda: None) + + report = build_bundle(tmp_path / "bundle", year=2024) + + (skipped,) = report.skipped_sources + assert skipped.source == restamp + assert ARTIFACT_YEAR_RESTAMP_CODE in skipped.reason + assert [error["code"] for error in skipped.validation["errors"]] == [ + ARTIFACT_YEAR_RESTAMP_CODE + ] + # The literal-label package still builds, with its own year's period. + assert [source.source for source in report.source_packages] == [literal] + assert "source_suite_build_failed" not in {error.code for error in report.errors} + assert {row["period"]["value"] for row in _bundle_rows(tmp_path / "bundle")} == { + ARTIFACT_YEAR + } + + +def test_explicit_source_that_would_restamp_fails_the_bundle( + tmp_path, + monkeypatch, + capsys, +): + restamp = str( + _write_package(tmp_path, monkeypatch, {"period", "record_ids"}, name="r") + ) + + at_artifact_year = build_bundle( + tmp_path / "at-artifact-year", + year=ARTIFACT_YEAR, + sources=[restamp], + ) + report = build_bundle(tmp_path / "off-year", year=2024, sources=[restamp]) + + assert "source_suite_build_failed" not in { + error.code for error in at_artifact_year.errors + } + assert not report.valid + assert [(error.code, error.source) for error in report.errors] == [ + ("source_suite_build_failed", restamp) + ] + assert "pins artifact_year 2021" in report.errors[0].message + assert _bundle_rows(tmp_path / "off-year") == [] + exit_code = harness_main( + [ + "build-bundle", + "--year", + "2024", + "--source", + restamp, + "--out", + str(tmp_path / "cli"), + ] + ) + payload = json.loads(capsys.readouterr().out) + assert exit_code == 1 + assert payload["valid"] is False + assert payload["errors"][0]["code"] == "source_suite_build_failed" + + +# --- property: exhaustive enumeration ---------------------------------------- + + +def test_guard_matches_the_restamp_invariant_over_every_synthetic_combination(): + """Exhaustive over label subsets x selection mechanisms x build years. + + Hypothesis is not a Chronicle dependency (see pyproject.toml / uv.lock), so + this enumerates the whole space instead of sampling it: 2**6 templated label + subsets x 4 mechanisms x 5 build years = 1,280 cases. For each case it + checks that + + - the guard fires exactly when the year selects nothing (mechanism + "none"), Y != A, and at least one label is templated; + - a refused build raises ArtifactYearRestampError, and the build the guard + refused would have restamped (differential: without the guard the build + emits a fact with A's value and lineage but different labels); + - an allowed build either fails with its own error or emits exactly what + the unguarded build emits, and none of it is a restamp. + """ + cases = 0 + for size in range(len(LABEL_FIELDS) + 1): + for templated in itertools.combinations(LABEL_FIELDS, size): + for mechanism in MECHANISMS: + package = _package(frozenset(templated), mechanism) + base = _unguarded_facts(package, ARTIFACT_YEAR) + assert base, (templated, mechanism) + for year in BUILD_YEARS: + cases += 1 + case = (templated, mechanism, year) + expected = ( + year != ARTIFACT_YEAR + and mechanism == "none" + and bool(templated) + ) + issues = artifact_year_restamp_issues(package, year) + assert bool(issues) is expected, case + unguarded = _unguarded_facts(package, year) + if expected: + with pytest.raises(ArtifactYearRestampError): + package.build_facts(year) + assert unguarded is not None, case + assert _restamped_facts(unguarded, base), case + continue + if unguarded is None: + # Only a year-selecting package at a year its file + # lacks fails; it must fail with its own error. + assert mechanism != "none" and year not in FILE_YEARS, case + with pytest.raises(ValueError) as raised: + package.build_facts(year) + assert not isinstance(raised.value, ArtifactYearRestampError) + continue + guarded = package.build_facts(year) + assert [_serialized(f) for f in guarded] == [ + _serialized(f) for f in unguarded + ], case + assert _restamped_facts(guarded, base) == [], case + assert cases == 2 ** len(LABEL_FIELDS) * len(MECHANISMS) * len(BUILD_YEARS) + + +# --- the real package tree --------------------------------------------------- + + +def _pinned_year_dependent_packages() -> list[str]: + """Pinned packages whose YAML renders {year}/{filing_year} or has *_by_year.""" + packages = [] + for sub in discover_source_package_dirs(REPO_ROOT / "packages"): + path = REPO_ROOT / "packages" / sub / "source_package.yaml" + text = path.read_text(encoding="utf-8") + if "artifact_year:" not in text: + continue + if "{year" in text or "{filing_year" in text or "_by_year:" in text: + packages.append(str(sub)) + return sorted(packages) + + +PINNED_YEAR_DEPENDENT = _pinned_year_dependent_packages() + + +def test_real_tree_has_pinned_year_dependent_packages_to_check(): + # Guards the parametrization below against silently checking nothing. + assert len(PINNED_YEAR_DEPENDENT) >= 10 + assert "bea/regional_personal_income_state" in PINNED_YEAR_DEPENDENT + + +@pytest.mark.parametrize("package_path", PINNED_YEAR_DEPENDENT) +def test_no_pinned_package_restamps_at_neighbouring_or_default_years(package_path): + """Every pinned, year-dependent package passes the guard at A-1, A+1, 2023. + + 2023 is the default bundle year (``build-bundle --year`` default). + """ + package = load_source_package(REPO_ROOT / "packages" / package_path) + artifact_year = package.artifact.artifact_year + for year in sorted({artifact_year - 1, artifact_year + 1, 2023} - {artifact_year}): + assert artifact_year_restamp_issues(package, year) == [], (package_path, year) + + +# A representative subset of pinned packages whose builds take a second or +# two, covering each way a pinned package meets a non-artifact build year: +# literal labels (the W-2, IRA and Historic Table 2 packages and ICI, whose +# labels this change pinned to the artifact year), column_by_year (BEA +# regional, CBO projections, CMS NHE), a column_by_year that lacks the artifact +# year itself (CBO receipts, Federal Reserve Z.1) and selected_rows that filter +# on {year} (Census projections). The slow builds (soi-state-2022 at 50-90 s, +# the BEA NIPA packages at ~6 s, the 26,880-fact congressional-district +# package) stay out of the default suite; the guard-only test above covers +# them, and the PR's before/after build evidence covers their facts. +REPRESENTATIVE_PINNED_PACKAGES = ( + "irs_soi/w2_statistics_2020", + "irs_soi/ira_roth_contributions_2022", + "irs_soi/ira_traditional_contributions_2022", + "irs_soi/historic_table_2", + "ici/fact_book_table_30", + "bea/regional_personal_income_state", + "cbo/revenue_projections_income_by_source_2026_02", + "cbo/individual_income_tax_receipts_2026_02", + "cms_nhe/table_24", + "federal_reserve/z1_household_net_worth", + "census/population_projections_2023", +) + + +@pytest.mark.parametrize("package_path", REPRESENTATIVE_PINNED_PACKAGES) +def test_real_pinned_builds_never_restamp_the_artifact_year_facts(package_path): + """The invariant on real bytes, and the guard agreeing with the built facts.""" + package = load_source_package(REPO_ROOT / "packages" / package_path) + artifact_year = package.artifact.artifact_year + years = (artifact_year - 1, artifact_year + 1) + base = _unguarded_facts(package, artifact_year) + for year in years: + issues = artifact_year_restamp_issues(package, year) + unguarded = _unguarded_facts(package, year) + would_restamp = bool( + base is not None + and unguarded is not None + and _restamped_facts(unguarded, base) + ) + assert bool(issues) is would_restamp, (package_path, year) + try: + facts = package.build_facts(year) + except ValueError: + continue # the build raises: the invariant holds + if base is not None: # else A has no facts (column_by_year lacks A) + assert _restamped_facts(facts, base) == [], (package_path, year) + + +def _as_origin_main_had_it(tmp_path: Path, package_path: str, edit) -> Path: + """Copy a real package into tmp_path with ``edit`` applied to its payload. + + The copy keeps ``resource_package: db``, so it reads the real pinned bytes. + """ + payload = yaml.safe_load( + (REPO_ROOT / "packages" / package_path / "source_package.yaml").read_text( + encoding="utf-8" + ) + ) + edit(payload) + package_dir = tmp_path / package_path.replace("/", "__") + package_dir.mkdir(parents=True) + (package_dir / "source_package.yaml").write_text( + yaml.safe_dump(payload, sort_keys=False), + encoding="utf-8", + ) + return package_dir + + +def _template_ira_labels(payload: dict) -> None: + # soi-ira-roth-contributions-2022 at origin/main: every label follows --year. + payload["artifact"]["vintage"] = "tax_year_{year}" + for record_set in payload["record_sets"]: + record_set["period"] = "{year}" + for key in ("record_set_id", "source_record_id_prefix"): + record_set[key] = record_set[key].replace("ty2022", "ty{year}") + + +def _fix_bea_regional_column(payload: dict) -> None: + # BEA regional at origin/main: the 2024 column under a {year} period, which + # relabelled CY2024 values as CY2023 in a --year 2023 build. + for record_set in payload["record_sets"]: + for measure in record_set["measures"]: + measure.pop("column_by_year") + measure.pop("expected_column_header_row", None) + measure.pop("expected_column_header_by_year", None) + measure["column"] = "AI" + + +@pytest.mark.parametrize( + ("package_path", "edit"), + [ + ("irs_soi/ira_roth_contributions_2022", _template_ira_labels), + ("bea/regional_personal_income_state", _fix_bea_regional_column), + ], + ids=["ira-roth-templated-labels", "bea-regional-fixed-column"], +) +def test_guard_flags_real_bytes_exactly_when_the_unguarded_build_restamps( + tmp_path, + package_path, + edit, +): + """Differential on real bytes: origin/main's YAML restamps; the guard agrees.""" + package = load_source_package(_as_origin_main_had_it(tmp_path, package_path, edit)) + artifact_year = package.artifact.artifact_year + base = _unguarded_facts(package, artifact_year) + assert base + + for year in (artifact_year - 1, artifact_year + 1): + restamped = _restamped_facts(_unguarded_facts(package, year), base) + issues = artifact_year_restamp_issues(package, year) + # Without the guard every artifact-year fact comes back relabelled. + assert len(restamped) == len(base), year + assert {fact.period.value for fact in restamped} == {year} + assert issues + assert {issue.code for issue in issues} == {ARTIFACT_YEAR_RESTAMP_CODE} + with pytest.raises(ArtifactYearRestampError): + package.build_facts(year) From aa3e4de3a9656c3f47d9af1d6bccccd46760ca56 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Fri, 25 Sep 2026 18:29:38 -0400 Subject: [PATCH 3/6] Guard BEA regional year columns by their header column_by_year binds each year to a column letter of SAINC5N__ALL_AREAS_1998_2025.csv. Check the binding at build time: each measure now expects row 1 of its year's column to name that year (expected_column_header_by_year). A refreshed file with shifted columns then fails loudly instead of relabelling one year's values as another year's. A delimited file's header row keeps the text '2024', while package YAML renders a digit-only string to the integer 2024 (_render_value). So no year header guard could match a CSV header. _resolve_guard_cell now matches a text cell against an integer expectation when the text is exactly that integer's digits, and nothing looser ('02024', '2024.0' and booleans still fail). Only builds this used to reject can change; a build that passed before resolves the same cells. Effect on BEA regional: each fact's source_cell_keys gains the row-1 header cell of its column, and layout.record_set_spec_hash changes. Values, periods, record ids, source_row_keys, aggregate_fact_key and semantic_fact_key are unchanged at 2023, 2024 and 2025, and every year 1998-2025 builds with the guard passing. Co-Authored-By: Claude Opus 5.5 --- chronicle/sources/specs.py | 23 ++++++++- docs/agent-source-package-harness.md | 6 +++ .../source_package.yaml | 47 ++++++++++++++++++- tests/test_chronicle_source_cells.py | 45 +++++++++++++++++- 4 files changed, 118 insertions(+), 3 deletions(-) diff --git a/chronicle/sources/specs.py b/chronicle/sources/specs.py index 277df600..bac8c12a 100644 --- a/chronicle/sources/specs.py +++ b/chronicle/sources/specs.py @@ -738,7 +738,7 @@ def _resolve_guard_cell( raise ValueError( f"Selector {spec.selector_id!r} missing {label} {address}" ) from exc - if guard_cell.raw_value != expected_value: + if not _guard_value_matches(guard_cell.raw_value, expected_value): raise ValueError( f"Selector {spec.selector_id!r} expected {label} " f"{expected_value!r}, got {guard_cell.raw_value!r}" @@ -746,6 +746,27 @@ def _resolve_guard_cell( return guard_cell +def _guard_value_matches(raw_value: Scalar, expected_value: Scalar) -> bool: + """Whether a guard cell holds the value a source package expects. + + Package YAML renders a digit-only string such as ``'2024'`` to the integer + 2024, so an author cannot ask for the text ``'2024'`` that a delimited + file's header row keeps. That text matches the integer it renders to, and + nothing looser: not ``'02024'``, ``'2024.0'`` or a boolean. + """ + if raw_value == expected_value: + return True + return ( + isinstance(expected_value, int) + and not isinstance(expected_value, bool) + and isinstance(raw_value, str) + and raw_value.isascii() + and raw_value.isdigit() + and (raw_value == "0" or not raw_value.startswith("0")) + and int(raw_value) == expected_value + ) + + def _scale_value(value: Scalar, scale: int | float) -> int | float | str: if isinstance(value, bool) or value is None: raise ValueError(f"Cannot scale nonnumeric source value {value!r}") diff --git a/docs/agent-source-package-harness.md b/docs/agent-source-package-harness.md index bdcf7e76..c8cec774 100644 --- a/docs/agent-source-package-harness.md +++ b/docs/agent-source-package-harness.md @@ -152,6 +152,12 @@ rows: label: end age ``` +Package YAML renders a digit-only string such as `'2024'` to the integer 2024, +while a delimited file's header row keeps the text `2024`. A guard that expects +an integer therefore also matches a cell holding exactly that integer's digits, +so a year-selecting package can check a CSV year header with +`expected_column_header_by_year`. `02024`, `2024.0` and booleans still fail. + Use `range_label_guards` when a fact sums a dense row range and interior labels are part of the fact definition. Endpoint guards catch off-by-one boundaries, but they do not catch an inserted, duplicated, or shifted interior label. Range diff --git a/packages/bea/regional_personal_income_state/source_package.yaml b/packages/bea/regional_personal_income_state/source_package.yaml index 69731286..f22d7432 100644 --- a/packages/bea/regional_personal_income_state/source_package.yaml +++ b/packages/bea/regional_personal_income_state/source_package.yaml @@ -3166,7 +3166,8 @@ record_sets: label: Personal income ordinal: 0 # SAINC5N__ALL_AREAS_1998_2025.csv holds one column per calendar - # year, 1998 (I) through 2025 (AJ); --year selects that column. + # year, 1998 (I) through 2025 (AJ); --year selects that column, and + # the header guard checks that row 1 of the column names that year. column_by_year: &sainc5n_year_columns 1998: I 1999: J @@ -3196,6 +3197,36 @@ record_sets: 2023: AH 2024: AI 2025: AJ + expected_column_header_row: 1 + expected_column_header_by_year: &sainc5n_year_headers + 1998: 1998 + 1999: 1999 + 2000: 2000 + 2001: 2001 + 2002: 2002 + 2003: 2003 + 2004: 2004 + 2005: 2005 + 2006: 2006 + 2007: 2007 + 2008: 2008 + 2009: 2009 + 2010: 2010 + 2011: 2011 + 2012: 2012 + 2013: 2013 + 2014: 2014 + 2015: 2015 + 2016: 2016 + 2017: 2017 + 2018: 2018 + 2019: 2019 + 2020: 2020 + 2021: 2021 + 2022: 2022 + 2023: 2023 + 2024: 2024 + 2025: 2025 source_column_id: '{year}' concept: bea_regional.personal_income source_concept: bea_regional.sainc5n_line_10_personal_income @@ -5101,6 +5132,8 @@ record_sets: label: Dividends, interest, and rent ordinal: 0 column_by_year: *sainc5n_year_columns + expected_column_header_row: 1 + expected_column_header_by_year: *sainc5n_year_headers source_column_id: '{year}' concept: bea_regional.dividends_interest_and_rent source_concept: bea_regional.sainc5n_line_46_dividends_interest_and_rent @@ -7006,6 +7039,8 @@ record_sets: label: Personal current transfer receipts ordinal: 0 column_by_year: *sainc5n_year_columns + expected_column_header_row: 1 + expected_column_header_by_year: *sainc5n_year_headers source_column_id: '{year}' concept: bea_regional.personal_current_transfer_receipts source_concept: bea_regional.sainc5n_line_47_personal_current_transfer_receipts @@ -8911,6 +8946,8 @@ record_sets: label: Wages and salaries ordinal: 0 column_by_year: *sainc5n_year_columns + expected_column_header_row: 1 + expected_column_header_by_year: *sainc5n_year_headers source_column_id: '{year}' concept: bea_regional.wages_and_salaries source_concept: bea_regional.sainc5n_line_50_wages_and_salaries @@ -10816,6 +10853,8 @@ record_sets: label: Supplements to wages and salaries ordinal: 0 column_by_year: *sainc5n_year_columns + expected_column_header_row: 1 + expected_column_header_by_year: *sainc5n_year_headers source_column_id: '{year}' concept: bea_regional.supplements_to_wages_and_salaries source_concept: bea_regional.sainc5n_line_60_supplements_to_wages_and_salaries @@ -12721,6 +12760,8 @@ record_sets: label: Proprietors' income ordinal: 0 column_by_year: *sainc5n_year_columns + expected_column_header_row: 1 + expected_column_header_by_year: *sainc5n_year_headers source_column_id: '{year}' concept: bea_regional.proprietors_income source_concept: bea_regional.sainc5n_line_70_proprietors_income @@ -14626,6 +14667,8 @@ record_sets: label: Contributions for government social insurance ordinal: 0 column_by_year: *sainc5n_year_columns + expected_column_header_row: 1 + expected_column_header_by_year: *sainc5n_year_headers source_column_id: value concept: bea_regional.contributions_for_government_social_insurance source_concept: bea_regional.sainc5n_line_36_contributions_for_government_social_insurance @@ -16531,6 +16574,8 @@ record_sets: label: Residence adjustment ordinal: 0 column_by_year: *sainc5n_year_columns + expected_column_header_row: 1 + expected_column_header_by_year: *sainc5n_year_headers source_column_id: value concept: bea_regional.residence_adjustment source_concept: bea_regional.sainc5n_line_42_residence_adjustment diff --git a/tests/test_chronicle_source_cells.py b/tests/test_chronicle_source_cells.py index 66e1d237..1d4603f6 100644 --- a/tests/test_chronicle_source_cells.py +++ b/tests/test_chronicle_source_cells.py @@ -32,7 +32,11 @@ source_cells_from_source_rows, source_rows_from_delimited_text, ) -from chronicle.sources.specs import resolve_source_record +from chronicle.sources.specs import ( + CellSelectorSpec, + resolve_cell_selector, + resolve_source_record, +) def test_ods_numeric_text_mode_coerces_formatted_numbers_only(): @@ -151,6 +155,45 @@ def test_source_record_selector_guard_fails_on_changed_row_header(): resolve_source_record(cells, bad_spec) +def test_column_header_guard_matches_a_delimited_year_header_text(): + # Package YAML renders a digit-only string such as '2024' to the integer + # 2024, while a delimited file's header row keeps the text '2024'. A guard + # expecting 2024 matches that text, and nothing looser. + artifact = SourceArtifactMetadata( + source_name="bea", + source_table="test", + source_file="test.csv", + url="https://example.test/test.csv", + vintage="test", + sha256="abc123", + size_bytes=10, + extracted_at="2026-09-25", + extraction_method="test", + ) + rows = source_rows_from_delimited_text( + b"Item,2023,2024,02024,2024.0\nReturns,1,2,3,4\n", + artifact, + sheet_name="test", + ) + cells = source_cells_from_source_rows(rows, selected_rows=()) + + def selector(column: str, header): + return CellSelectorSpec( + selector_id=f"test.{column}", + sheet_name="test", + address=f"{column}2", + expected_cell_type="number", + expected_column_header_address=f"{column}1", + expected_column_header=header, + ) + + assert resolve_cell_selector(cells, selector("C", 2024)).raw_value == 2 + assert resolve_cell_selector(cells, selector("C", "2024")).raw_value == 2 + for column, header in (("B", 2024), ("D", 2024), ("E", 2024), ("C", True)): + with pytest.raises(ValueError, match="expected column header"): + resolve_cell_selector(cells, selector(column, header)) + + def test_delimited_source_row_selection_requires_exact_match(): artifact = SourceArtifactMetadata( source_name="bea", From 74bce58a7d6396b91944016d61fcfa564844dbc6 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 27 Sep 2026 04:11:03 -0400 Subject: [PATCH 4/6] Re-pin the default-bundle snapshot for the literal-year packages The --year 2023 default bundle now emits the congressional-district, state and IRA facts at TY2022 and the W-2 facts at TY2020. tax_year:2023 drops by 26,893, tax_year:2022 gains 26,888 and tax_year:2020 gains 5. At TY2022 the CD state-total and US rows share semantic keys with the Historic Table 2 rows for the same cells, so semantic duplicates go from 467 to 2,022 (CI run 36244118301). Every other count is unchanged. Co-Authored-By: Claude Opus 5.5 --- tests/test_chronicle_bundle.py | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/tests/test_chronicle_bundle.py b/tests/test_chronicle_bundle.py index 4d6fcd36..6fc59b6e 100644 --- a/tests/test_chronicle_bundle.py +++ b/tests/test_chronicle_bundle.py @@ -139,7 +139,12 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "fact_count": 350893, "geography_count": 12592, "period_count": 494, - "semantic_duplicate_key_count": 467, + # 467 before chronicle#292 moved the congressional-district and + # state_2022 rows from their ty2023 restamp to TY2022, where 1,560 CD + # state-total and US rows now share semantic keys with the Historic + # Table 2 rows for the same TY2022 cells (two IRS publications of one + # cell), net of ty2023 collisions the restamp had made. + "semantic_duplicate_key_count": 2022, "skipped_source_count": 10, "source_count": 50, "source_package_count": 227, @@ -948,10 +953,12 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "tax_year:2017": 9, "tax_year:2018": 11, "tax_year:2019": 11, - "tax_year:2020": 11, + "tax_year:2020": 16, "tax_year:2021": 42, - "tax_year:2022": 41270, - "tax_year:2023": 63342, + # chronicle#292: 26,888 CD, state and IRA facts move from their + # ty2023 restamp to TY2022, and the 5 W-2 facts to TY2020. + "tax_year:2022": 68158, + "tax_year:2023": 36449, "tax_year:2024": 328, } for fiscal_year in range(2017, 2026): @@ -1223,7 +1230,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "tax_unit": 41368, } assert not coverage["duplicates"]["aggregate_fact_keys"] - assert len(coverage["duplicates"]["semantic_fact_keys"]) == 467 + assert len(coverage["duplicates"]["semantic_fact_keys"]) == 2022 assert Counter(warning["code"] for warning in summary["warnings"]) == { "conflicting_geography_name_across_packages": 50, "conflicting_groupby_value_label": 16, From 0eb55703750963bb555ca2c1d474a0d5a772e500 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 27 Sep 2026 04:17:00 -0400 Subject: [PATCH 5/6] Check the restamp guard per cell, not per record set Review of #292 found three synthetic restamps the record-set-level selection fingerprint admitted: a fixed selected_rows entry beside a {year} one, a fixed-column measure beside a column_by_year measure in one record set, and a {year} virtual sheet name on a delimited file. A record set now restamps when any of its (row, measure) cells is read alike at both years (same row, measure, guard cells, spreadsheet sheet and selected_rows entries) while anything in the record set differs, since each fact carries a hash of the whole compiled record set. A delimited, JSON, HTML or PDF artifact's sheet name is a label, not a selection. Over the real tree the verdicts are unchanged: 0 of 585 pinned (package, year) pairs flagged on this branch, and the same 22 pairs in the same 10 packages on origin/main's package YAML. Co-Authored-By: Claude Opus 5.5 --- chronicle/source_package.py | 164 +++++++++++++----- tests/test_chronicle_artifact_year_restamp.py | 110 ++++++++++++ 2 files changed, 234 insertions(+), 40 deletions(-) diff --git a/chronicle/source_package.py b/chronicle/source_package.py index d2c76ef9..7f4d65c6 100644 --- a/chronicle/source_package.py +++ b/chronicle/source_package.py @@ -1453,6 +1453,8 @@ def _restamp_issues( artifact, artifact_year ): return [] + entries_at_year = _selected_row_entries(artifact, year) + entries_at_artifact_year = _selected_row_entries(artifact, artifact_year) specs_at_year = package._compile_record_set_specs(year) specs_at_artifact_year = package._compile_record_set_specs(artifact_year) except (KeyError, TypeError, ValueError): @@ -1460,10 +1462,13 @@ def _restamp_issues( # selects by year (for example a column_by_year without that year). A # build at a year that cannot compile fails with its own error. return [] + sheet_selects = _sheet_name_selects(artifact) artifact_moves = [] - for name in _ARTIFACT_LABEL_FIELDS: - at_artifact_year = _render_string(getattr(artifact, name), year=artifact_year) - at_year = _render_string(getattr(artifact, name), year=year) + label_fields = _ARTIFACT_LABEL_FIELDS + (() if sheet_selects else ("sheet_name",)) + for name in label_fields: + value = getattr(artifact, name) + at_artifact_year = _render_string(value, year=artifact_year) if value else value + at_year = _render_string(value, year=year) if value else value if at_artifact_year != at_year: artifact_moves.append((f"artifact.{name}", at_artifact_year, at_year)) issues = [] @@ -1472,15 +1477,16 @@ def _restamp_issues( specs_at_artifact_year, strict=True, ): - if _record_set_selection(spec_at_year) != _record_set_selection( - spec_at_artifact_year - ): + moves = _restamped_cell_moves( + spec_at_year, + spec_at_artifact_year, + entries_at_year=entries_at_year, + entries_at_artifact_year=entries_at_artifact_year, + sheet_selects=sheet_selects, + ) + if moves is None: continue - moves = [*artifact_moves] - if spec_at_year != spec_at_artifact_year: - moves.extend( - _changed_leaves(asdict(spec_at_artifact_year), asdict(spec_at_year)) - ) + moves = [*artifact_moves, *moves] if not moves: continue issues.append( @@ -1499,6 +1505,106 @@ def _restamp_issues( return issues +def _restamped_cell_moves( + spec_at_year: SourceRecordSetSpec, + spec_at_artifact_year: SourceRecordSetSpec, + *, + entries_at_year: tuple[Any, ...], + entries_at_artifact_year: tuple[Any, ...], + sheet_selects: bool, +) -> list[tuple[str, Any, Any]] | None: + """The record set's moved fields when some cell is read alike at both years. + + A record set builds one fact per (row, measure) cell. A cell is read alike + when its row, its measure, its guard cells, the record set's sheet (for a + spreadsheet) and the ``selected_rows`` entries its rows come from all + render the same at both years. Returns ``None`` when no cell is read alike. + Otherwise it returns every field of the record set that differs. Each + fact carries ``layout.record_set_spec_hash``, a hash of the whole compiled + record set, so any difference relabels the alike cells' facts. One + year-selected measure beside a fixed one, or one ``{year}`` selected row + beside a fixed one, therefore no longer hides the fixed cells + (chronicle#292 review). + """ + + if sheet_selects and spec_at_year.sheet_name != spec_at_artifact_year.sheet_name: + return None + rows_alike = any( + _row_selection(row_at_year, entries_at_year) + == _row_selection(row_at_artifact_year, entries_at_artifact_year) + for row_at_year, row_at_artifact_year in zip( + spec_at_year.rows, spec_at_artifact_year.rows, strict=True + ) + ) + measures_alike = any( + _measure_selection(measure_at_year) + == _measure_selection(measure_at_artifact_year) + for measure_at_year, measure_at_artifact_year in zip( + spec_at_year.measures, spec_at_artifact_year.measures, strict=True + ) + ) + if not (rows_alike and measures_alike): + return None + return _changed_leaves(asdict(spec_at_artifact_year), asdict(spec_at_year)) + + +def _row_selection(row: Any, entries: tuple[Any, ...]) -> tuple[Any, ...]: + """What a record-set row reads: its cells, guards and source rows.""" + + referenced = {row.row_number, row.row_end_number or row.row_number} + if row.row_end_number: + referenced.update(range(row.row_number, row.row_end_number + 1)) + referenced.update( + guard.row + for guard in row.guard_cells + if isinstance(guard.row, int) and not isinstance(guard.row, bool) + ) + return ( + tuple(getattr(row, name) for name in _ROW_SELECTION_FIELDS), + tuple( + (guard.column, guard.expected_value, guard.row) for guard in row.guard_cells + ), + tuple( + (guard.column, guard.expected_values) for guard in row.range_label_guards + ), + tuple( + _selected_row_entry(entries, row_number) + for row_number in sorted(number for number in referenced if number) + ), + ) + + +def _measure_selection(measure: Any) -> tuple[Any, ...]: + return tuple(getattr(measure, name) for name in _MEASURE_SELECTION_FIELDS) + + +def _selected_row_entries(artifact: SourceArtifactSpec, year: int) -> tuple[Any, ...]: + return tuple( + tuple((key, str(_render_value(value, year=year))) for key, value in row.items()) + for row in artifact.selected_rows + ) + + +def _selected_row_entry(entries: tuple[Any, ...], row_number: int) -> Any: + # A selected_rows extract numbers its rows from 2 in declaration order + # (row 1 is the header): row r reads entry r - 2. + if not entries: + return None + index = row_number - 2 + return entries[index] if 0 <= index < len(entries) else ("row", row_number) + + +def _sheet_name_selects(artifact: SourceArtifactSpec) -> bool: + """Whether the artifact's sheet name picks data (a spreadsheet) or only labels. + + Delimited text, JSON, HTML and PDF parsers name a virtual sheet; reading the + same bytes under another sheet name reads the same cells. + """ + + parser = str(artifact.parser or "") + return any(kind in parser for kind in ("xls", "ods")) + + def _declarations_depend_on_year(package: SourcePackage) -> bool: """Whether any declaration renders ``{year}``/``{filing_year}`` or is by-year.""" return _mentions_year(asdict(package.artifact)) or any( @@ -1520,37 +1626,15 @@ def _mentions_year(value: Any) -> bool: def _artifact_selection(artifact: SourceArtifactSpec, year: int) -> tuple[Any, ...]: - return ( - tuple(getattr(artifact, name) for name in _ARTIFACT_SELECTION_FIELDS), - _render_string(artifact.sheet_name, year=year) if artifact.sheet_name else None, - tuple( - {key: str(_render_value(value, year=year)) for key, value in row.items()} - for row in artifact.selected_rows - ), - ) + """What the artifact reads apart from ``selected_rows`` (compared per row).""" - -def _record_set_selection(spec: SourceRecordSetSpec) -> tuple[Any, ...]: + sheet_name = None + if artifact.sheet_name and _sheet_name_selects(artifact): + sheet_name = _render_string(artifact.sheet_name, year=year) return ( - tuple(getattr(spec, name) for name in _RECORD_SET_SELECTION_FIELDS), - tuple( - ( - tuple(getattr(row, name) for name in _ROW_SELECTION_FIELDS), - tuple( - (guard.column, guard.expected_value, guard.row) - for guard in row.guard_cells - ), - tuple( - (guard.column, guard.expected_values) - for guard in row.range_label_guards - ), - ) - for row in spec.rows - ), - tuple( - tuple(getattr(measure, name) for name in _MEASURE_SELECTION_FIELDS) - for measure in spec.measures - ), + tuple(getattr(artifact, name) for name in _ARTIFACT_SELECTION_FIELDS), + sheet_name, + len(artifact.selected_rows), ) diff --git a/tests/test_chronicle_artifact_year_restamp.py b/tests/test_chronicle_artifact_year_restamp.py index 2081e176..05409762 100644 --- a/tests/test_chronicle_artifact_year_restamp.py +++ b/tests/test_chronicle_artifact_year_restamp.py @@ -14,6 +14,8 @@ from __future__ import annotations +import copy + import functools import hashlib import itertools @@ -585,6 +587,114 @@ def test_explicit_source_that_would_restamp_fails_the_bundle( # --- property: exhaustive enumeration ---------------------------------------- +def _package_from(payload: dict, content: bytes, filename: str) -> SourcePackage: + artifact = dict(payload["artifact"]) + selected_rows = tuple(artifact.pop("selected_rows", ())) + return SourcePackage( + package_id=payload["package_id"], + label=None, + artifact=_PinnedInlineArtifact( + content=content, + filename=filename, + selected_rows=selected_rows, + **artifact, + ), + record_sets=tuple(DeclarativeRecordSet(rs) for rs in payload["record_sets"]), + package_path=Path("synthetic"), + dimension_labels=payload["dimension_labels"], + ) + + +def _assert_guard_agrees_with_built_facts(package: SourcePackage, year: int) -> None: + restamped = _restamped_facts( + _unguarded_facts(package, year), _unguarded_facts(package, ARTIFACT_YEAR) + ) + assert bool(artifact_year_restamp_issues(package, year)) == bool(restamped) + + +def test_a_fixed_selected_row_beside_a_year_row_is_refused(): + # One selected_rows entry follows {year} and one is fixed; the fixed row's + # facts would be relabelled even though the extract as a whole moves. + payload = _payload(frozenset({"period", "record_ids"}), "selected_rows") + payload["artifact"]["selected_rows"] = [ + {"Item": "Total", "Period": "all"}, + {"Item": "Returns", "Period": "{year}"}, + ] + fixed = copy.deepcopy(payload["record_sets"][0]) + fixed["record_set_id"] = fixed["source_record_id_prefix"] = "test.ty{year}.total" + fixed["rows"][0].update(value_id="total", label="Total", row_number=2) + by_year = copy.deepcopy(payload["record_sets"][0]) + by_year["rows"][0].update(row_number=3) + payload["record_sets"] = [fixed, by_year] + package = _package_from( + payload, (_long_csv().decode() + "Total,all,999\n").encode(), "long.csv" + ) + issues = artifact_year_restamp_issues(package, 2024) + assert [issue.record_set_id for issue in issues] == ["test.ty2024.total"] + _assert_guard_agrees_with_built_facts(package, 2024) + + +def test_a_fixed_measure_beside_a_year_selected_measure_is_refused(): + payload = _payload(frozenset({"period", "record_ids"}), "column_by_year") + by_year = payload["record_sets"][0]["measures"][0] + fixed = copy.deepcopy(by_year) + fixed.pop("column_by_year") + fixed.update( + measure_id="base", label="Base", ordinal=1, column=WIDE_COLUMNS[ARTIFACT_YEAR] + ) + payload["record_sets"][0]["measures"].append(fixed) + package = _package_from(payload, _wide_csv(), "wide.csv") + (issue,) = artifact_year_restamp_issues(package, 2024) + assert "period" in issue.message + _assert_guard_agrees_with_built_facts(package, 2024) + with pytest.raises(ArtifactYearRestampError): + package.build_facts(2024) + + +@pytest.mark.parametrize("templated", [frozenset(), frozenset({"legal_vintage"})]) +def test_a_fixed_cell_is_refused_when_anything_else_in_its_record_set_moves( + templated, +): + # Every fact carries a hash of its whole record set, so a fixed-column + # measure beside a year-selected one is relabelled by the other measure's + # column or label alone. Split fixed and year-selected cells into separate + # record sets. + payload = _payload(templated, "column_by_year") + fixed = copy.deepcopy(payload["record_sets"][0]["measures"][0]) + fixed.pop("column_by_year") + fixed.update( + measure_id="base", + label="Base", + ordinal=1, + column=WIDE_COLUMNS[ARTIFACT_YEAR], + legal_vintage=f"tax_year_{ARTIFACT_YEAR}", + ) + payload["record_sets"][0]["measures"].append(fixed) + package = _package_from(payload, _wide_csv(), "wide.csv") + assert artifact_year_restamp_issues(package, 2024) + _assert_guard_agrees_with_built_facts(package, 2024) + split = copy.deepcopy(payload) + alone = copy.deepcopy(split["record_sets"][0]) + alone["record_set_id"] = alone["source_record_id_prefix"] = "test.ty2021.base" + alone["measures"] = [split["record_sets"][0]["measures"].pop()] + split["record_sets"].append(alone) + split_package = _package_from(split, _wide_csv(), "wide.csv") + assert artifact_year_restamp_issues(split_package, 2024) == [] + _assert_guard_agrees_with_built_facts(split_package, 2024) + + +def test_a_delimited_files_virtual_sheet_name_is_a_label(): + # A delimited file has no sheets: a {year} sheet name reads the same cells + # under a new name, so it is a relabel, not a selection. + payload = _payload(frozenset({"period", "record_ids"}), "none") + payload["artifact"]["sheet_name"] = "synthetic {year}" + payload["record_sets"][0]["sheet_name"] = "synthetic {year}" + package = _package_from(payload, _wide_csv(), "wide.csv") + assert artifact_year_restamp_issues(package, 2024) + with pytest.raises(ArtifactYearRestampError): + package.build_facts(2024) + + def test_guard_matches_the_restamp_invariant_over_every_synthetic_combination(): """Exhaustive over label subsets x selection mechanisms x build years. From 364f6fdfcd4efaac0a76223311e7279009650195 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 27 Sep 2026 04:29:14 -0400 Subject: [PATCH 6/6] Compare each cell's effective selection and each parser's sheet role Final review of #292 found two more synthetic admits, neither used by any package: - A row's column overrides its measure's (row.column or measure.column), so a fixed row column beside a column_by_year measure is read alike. The guard now compares each (row, measure) cell resolved as the compiler resolves it: column, header row and header with row precedence, both value scales, divisor, rounding, cell type, guards and selected-row entries. - The artifact-level sheet_name picks a sheet only for xlsx_table_full_rows, is a virtual label for delimited and JSON parsers, and is never read by the used-range spreadsheet parsers, whose record sets name the sheet. The guard now treats it that way instead of letting an unused {year} sheet name disguise a fixed sheet as year-selecting. Real tree unchanged: 0 of 585 pairs flagged here, the same 22 in 10 packages on origin/main's YAML. Co-Authored-By: Claude Opus 5.5 --- chronicle/source_package.py | 108 ++++++++++-------- tests/test_chronicle_artifact_year_restamp.py | 23 ++++ tests/test_chronicle_bundle.py | 7 +- 3 files changed, 90 insertions(+), 48 deletions(-) diff --git a/chronicle/source_package.py b/chronicle/source_package.py index 7f4d65c6..9f9adbbf 100644 --- a/chronicle/source_package.py +++ b/chronicle/source_package.py @@ -1345,12 +1345,13 @@ def build_facts( ARTIFACT_YEAR_RESTAMP_CODE = "artifact_year_restamp" -# The declarations that decide what a build reads: which cells, the value taken -# from them, and what each guard cell must hold. A pinned package always reads +# The declarations that decide what a build reads: which file, sheet and +# selected rows, and per (row, measure) cell the column, rows, header and guard +# expectations and value scaling (_cell_selection). A pinned package always reads # the file of its ``artifact_year``, so when these render the same at two build # years, both builds read the same cells. Every other compiled field (period, # record ids, legal_vintage, filters, constraints, geography, concept and layout -# labels) only names or dates the facts built from those cells. +# labels) names or dates the facts built from those cells. _ARTIFACT_SELECTION_FIELDS = ( "parser", "archive_member", @@ -1359,29 +1360,6 @@ def build_facts( "header_row", ) _ARTIFACT_LABEL_FIELDS = ("vintage", "source_table") -_RECORD_SET_SELECTION_FIELDS = ("sheet_name",) -# A row's label doubles as its row-header guard when expected_row_header is -# unset. It counts as a label here, so a {year} row label over a fixed row is -# refused as a restamp rather than left to fail that header check. -_ROW_SELECTION_FIELDS = ( - "row_number", - "row_end_number", - "column", - "value_scale", - "expected_row_header", - "expected_row_header_column", - "expected_column_header_row", - "expected_column_header", -) -_MEASURE_SELECTION_FIELDS = ( - "column", - "divisor_column", - "value_scale", - "round_to", - "expected_cell_type", - "expected_column_header_row", - "expected_column_header", -) _RESTAMP_MESSAGE_MOVES = 6 @@ -1464,7 +1442,9 @@ def _restamp_issues( return [] sheet_selects = _sheet_name_selects(artifact) artifact_moves = [] - label_fields = _ARTIFACT_LABEL_FIELDS + (() if sheet_selects else ("sheet_name",)) + label_fields = _ARTIFACT_LABEL_FIELDS + ( + ("sheet_name",) if _artifact_sheet_role(artifact) == "label" else () + ) for name in label_fields: value = getattr(artifact, name) at_artifact_year = _render_string(value, year=artifact_year) if value else value @@ -1529,29 +1509,35 @@ def _restamped_cell_moves( if sheet_selects and spec_at_year.sheet_name != spec_at_artifact_year.sheet_name: return None - rows_alike = any( - _row_selection(row_at_year, entries_at_year) - == _row_selection(row_at_artifact_year, entries_at_artifact_year) + alike = any( + _cell_selection(row_at_year, measure_at_year, entries_at_year) + == _cell_selection( + row_at_artifact_year, measure_at_artifact_year, entries_at_artifact_year + ) for row_at_year, row_at_artifact_year in zip( spec_at_year.rows, spec_at_artifact_year.rows, strict=True ) - ) - measures_alike = any( - _measure_selection(measure_at_year) - == _measure_selection(measure_at_artifact_year) for measure_at_year, measure_at_artifact_year in zip( spec_at_year.measures, spec_at_artifact_year.measures, strict=True ) ) - if not (rows_alike and measures_alike): + if not alike: return None return _changed_leaves(asdict(spec_at_artifact_year), asdict(spec_at_year)) -def _row_selection(row: Any, entries: tuple[Any, ...]) -> tuple[Any, ...]: - """What a record-set row reads: its cells, guards and source rows.""" +def _cell_selection( + row: Any, measure: Any, entries: tuple[Any, ...] +) -> tuple[Any, ...]: + """What one (row, measure) cell reads, resolved as the compiler resolves it. + + ``compile_source_record_set_specs`` reads ``row.column or measure.column`` + and lets a row's column-header expectation override the measure's, so a + fixed row column is read alike even when the measure's column follows the + year. + """ - referenced = {row.row_number, row.row_end_number or row.row_number} + referenced = {row.row_number} if row.row_end_number: referenced.update(range(row.row_number, row.row_end_number + 1)) referenced.update( @@ -1560,7 +1546,26 @@ def _row_selection(row: Any, entries: tuple[Any, ...]) -> tuple[Any, ...]: if isinstance(guard.row, int) and not isinstance(guard.row, bool) ) return ( - tuple(getattr(row, name) for name in _ROW_SELECTION_FIELDS), + row.column or measure.column, + ( + row.expected_column_header_row + if row.expected_column_header_row is not None + else measure.expected_column_header_row + ), + ( + row.expected_column_header + if row.expected_column_header is not None + else measure.expected_column_header + ), + row.row_number, + row.row_end_number, + row.value_scale, + row.expected_row_header, + row.expected_row_header_column, + measure.divisor_column, + measure.value_scale, + measure.round_to, + measure.expected_cell_type, tuple( (guard.column, guard.expected_value, guard.row) for guard in row.guard_cells ), @@ -1574,10 +1579,6 @@ def _row_selection(row: Any, entries: tuple[Any, ...]) -> tuple[Any, ...]: ) -def _measure_selection(measure: Any) -> tuple[Any, ...]: - return tuple(getattr(measure, name) for name in _MEASURE_SELECTION_FIELDS) - - def _selected_row_entries(artifact: SourceArtifactSpec, year: int) -> tuple[Any, ...]: return tuple( tuple((key, str(_render_value(value, year=year))) for key, value in row.items()) @@ -1595,7 +1596,7 @@ def _selected_row_entry(entries: tuple[Any, ...], row_number: int) -> Any: def _sheet_name_selects(artifact: SourceArtifactSpec) -> bool: - """Whether the artifact's sheet name picks data (a spreadsheet) or only labels. + """Whether a record set's sheet name picks data (a spreadsheet) or only labels. Delimited text, JSON, HTML and PDF parsers name a virtual sheet; reading the same bytes under another sheet name reads the same cells. @@ -1605,6 +1606,23 @@ def _sheet_name_selects(artifact: SourceArtifactSpec) -> bool: return any(kind in parser for kind in ("xls", "ods")) +def _artifact_sheet_role(artifact: SourceArtifactSpec) -> str: + """What the artifact-level ``sheet_name`` does for this parser. + + ``xlsx_table_full_rows`` reads the named sheet ("selects"). The used-range + spreadsheet parsers never read it; their record sets name the sheet + ("unused"). Every other parser names a virtual sheet with it, a label on + each cell ("label"). + """ + + parser = str(artifact.parser or "") + if parser == "xlsx_table_full_rows": + return "selects" + if "used_range" in parser: + return "unused" + return "label" + + def _declarations_depend_on_year(package: SourcePackage) -> bool: """Whether any declaration renders ``{year}``/``{filing_year}`` or is by-year.""" return _mentions_year(asdict(package.artifact)) or any( @@ -1629,7 +1647,7 @@ def _artifact_selection(artifact: SourceArtifactSpec, year: int) -> tuple[Any, . """What the artifact reads apart from ``selected_rows`` (compared per row).""" sheet_name = None - if artifact.sheet_name and _sheet_name_selects(artifact): + if artifact.sheet_name and _artifact_sheet_role(artifact) == "selects": sheet_name = _render_string(artifact.sheet_name, year=year) return ( tuple(getattr(artifact, name) for name in _ARTIFACT_SELECTION_FIELDS), diff --git a/tests/test_chronicle_artifact_year_restamp.py b/tests/test_chronicle_artifact_year_restamp.py index 05409762..4e759b1a 100644 --- a/tests/test_chronicle_artifact_year_restamp.py +++ b/tests/test_chronicle_artifact_year_restamp.py @@ -683,6 +683,29 @@ def test_a_fixed_cell_is_refused_when_anything_else_in_its_record_set_moves( _assert_guard_agrees_with_built_facts(split_package, 2024) +def test_a_row_column_that_overrides_a_year_selected_measure_is_refused(): + # The compiler reads row.column before measure.column, so a fixed row + # column is read alike even though the measure's column follows the year. + payload = _payload(frozenset({"period", "record_ids"}), "column_by_year") + payload["record_sets"][0]["rows"][0]["column"] = WIDE_COLUMNS[ARTIFACT_YEAR] + package = _package_from(payload, _wide_csv(), "wide.csv") + assert artifact_year_restamp_issues(package, 2024) + _assert_guard_agrees_with_built_facts(package, 2024) + + +def test_an_unused_artifact_sheet_name_on_a_used_range_parser_is_ignored(): + # Used-range spreadsheet parsers read the record set's sheet; a {year} + # artifact sheet_name selects nothing and must not hide a fixed sheet. + payload = _payload(frozenset({"period", "record_ids"}), "sheet_name_by_year") + record_set = payload["record_sets"][0] + record_set.pop("sheet_name_by_year") + record_set["sheet_name"] = f"TY{ARTIFACT_YEAR}" + payload["artifact"]["sheet_name"] = "TY{year}" + package = _package_from(payload, *_artifact_bytes("sheet_name_by_year")) + assert artifact_year_restamp_issues(package, 2024) + _assert_guard_agrees_with_built_facts(package, 2024) + + def test_a_delimited_files_virtual_sheet_name_is_a_label(): # A delimited file has no sheets: a {year} sheet name reads the same cells # under a new name, so it is a relabel, not a selection. diff --git a/tests/test_chronicle_bundle.py b/tests/test_chronicle_bundle.py index 6fc59b6e..9b7cb177 100644 --- a/tests/test_chronicle_bundle.py +++ b/tests/test_chronicle_bundle.py @@ -140,10 +140,11 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "geography_count": 12592, "period_count": 494, # 467 before chronicle#292 moved the congressional-district and - # state_2022 rows from their ty2023 restamp to TY2022, where 1,560 CD - # state-total and US rows now share semantic keys with the Historic + # state_2022 rows from their ty2023 restamp to TY2022. There the CD + # file's state-total and US rows share semantic keys with the Historic # Table 2 rows for the same TY2022 cells (two IRS publications of one - # cell), net of ty2023 collisions the restamp had made. + # cell): 1,560 new duplicate keys among the changed packages' own + # builds, for a bundle-wide net of +1,555. "semantic_duplicate_key_count": 2022, "skipped_source_count": 10, "source_count": 50,