From 0f343249fd9a53816f336b9d34b811f38b650727 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 27 Sep 2026 04:55:28 -0400 Subject: [PATCH 1/3] Choose the capital-gains rebase control by concept and period, not feed order Historic Table 2 and congressional-district N01000 count Form 1040 line 7 returns (gain, loss-limited loss or distributions only); Table 1.4 col 37, the population capital_gains_gross measures, counts Schedule D gain returns only. The rebase control is now the latest Table 1.4 fact not after the build period, CD and HT2 rows never qualify, a tie between matched records refuses the compile, and a line-7 row without a usable control is dropped instead of shipping as a level. Rebased rows declare the bridge in soi_source_concept / soi_control_concept. On the pinned feed the 51 national_state HT2 state return counts go from 29,599,604 (2.39x the national 12,392,020 target) back to Build P's 12,289,836; amounts keep their values. national_state registry d315c75804ef -> 65e4dde11c83 (microcosm#1035). Co-Authored-By: Claude Opus 5.5 --- ...035-capital-gains-control-concept.fixed.md | 1 + docs/us-chronicle-feed-repin.md | 10 + docs/us-soi-capital-gains-concepts.md | 120 +++++ .../build/us_runtime/fiscal_targets.py | 139 ++++- .../tests/test_us_fiscal_targets.py | 484 +++++++++++++++++- 5 files changed, 728 insertions(+), 26 deletions(-) create mode 100644 changelog.d/1035-capital-gains-control-concept.fixed.md create mode 100644 docs/us-soi-capital-gains-concepts.md diff --git a/changelog.d/1035-capital-gains-control-concept.fixed.md b/changelog.d/1035-capital-gains-control-concept.fixed.md new file mode 100644 index 000000000..0e190d99d --- /dev/null +++ b/changelog.d/1035-capital-gains-control-concept.fixed.md @@ -0,0 +1 @@ +Rebase Historic Table 2 capital-gains rows only onto the Table 1.4 Schedule D taxable-net-gain control, the concept `capital_gains_gross` measures. Feed order no longer picks the control: congressional-district and Historic Table 2 rows (Form 1040 line 7, gain or loss) never qualify, a tie between matched records refuses the compile, and a line-7 row with no usable control is dropped instead of shipping as a level. On the pinned feed the 51 `national_state` state return-count targets move from 29,599,604 (2.39x the national target) to 12,289,836 (microcosm#1035). diff --git a/docs/us-chronicle-feed-repin.md b/docs/us-chronicle-feed-repin.md index 14aee6d60..8bf63b09c 100644 --- a/docs/us-chronicle-feed-repin.md +++ b/docs/us-chronicle-feed-repin.md @@ -171,6 +171,16 @@ compiled register goes from 32,867 to 32,843 targets, and the count, the only change: 32,842 compiled targets and 5,694 in `national_state` (`d315c75804ef`). +**Erratum (27 September 2026, microcosm#1035).** Row order changed too. The +new feed is sorted by the re-derived `aggregate_fact_key`s, and the +capital-gains rebase kept the first of two equal-period controls. The +congressional-district US row (TY2022 data stamped ty2023) now sorted ahead of +Table 1.4 ty2023. It took over the returns control and raised the 51 Historic +Table 2 state return counts from 12,289,836 to 29,599,604. The control is now +chosen by concept and period alone +([us-soi-capital-gains-concepts.md](us-soi-capital-gains-concepts.md)). The +counts are unchanged and `national_state` is `65e4dde11c83`. + The labelled feed exposed one latent defect in microcosm, fixed in this change: `apply_us_medicaid_enrollment_substitutions` built Rhode Island's substituted spec by cloning a neighbouring state's spec and kept that state's diff --git a/docs/us-soi-capital-gains-concepts.md b/docs/us-soi-capital-gains-concepts.md new file mode 100644 index 000000000..841d6fdf8 --- /dev/null +++ b/docs/us-soi-capital-gains-concepts.md @@ -0,0 +1,120 @@ +# SOI capital-gains concepts and the rebase control + +`capital_gains_gross` targets come from three IRS SOI products that count +different populations under the same measure ids (`net_capital_gains_returns`, +`net_capital_gains_amount`). This page records what each one counts, which +one the model measures, and how `fiscal_targets.py` combines them +(microcosm#1035). + +## What each table counts + +| Source | Columns | What it counts | +|---|---|---| +| Table 1.4 (`23in14ar.xls`) | col 37 / 38, "Sales of capital assets reported on Form 1040, Schedule D: Taxable net gain" | Schedule D returns that net to a gain, and that gain | +| Table 1.4 | col 39 / 40, "Taxable net loss" | Schedule D returns that net to a loss, and the loss after the $3,000 limit | +| Table 1.4 | col 35 / 36, "Capital gain distributions reported on Form 1040" | Returns that report only distributions, without Schedule D | +| Historic Table 2 (`22in55cmcsv.csv`), congressional-district file (`22incd.csv`) | N01000 / A01000 | "Number of returns with net capital gain (less loss)" and its amount, at Form 1040 line 7 (both documentation guides) | + +Line 7 carries a Schedule D gain, a loss-limited Schedule D loss, or +distributions reported without Schedule D (2022 Form 1040 instructions, line +7, Exception 1). Columns 35, 37 and 39 are disjoint: Table 1.3 row 18, "Sales +of capital assets net gain", equals col 35 + col 37 exactly in TY2022 and +TY2023. + +N01000 is the union of the three Table 1.4 populations: + +| Tax year | HT2 US N01000 | Table 1.4 col 35 + 37 + 39 | Difference | HT2 A01000 vs col 36 + 38 − 40 | +|---|---:|---:|---:|---:| +| 2020 | 29,008,620 | 29,003,885 | +0.016% | −0.63% | +| 2021 | 32,996,180 | 33,076,998 | −0.244% | −0.29% | +| 2022 | 30,465,850 | 30,461,045 | +0.016% | −0.17% | +| 2023 | 29,481,840 | 29,434,188 | +0.162% | −0.28% | + +In TY2022, col 37 (12,915,122) is 42.4% of N01000. The amounts are close +(A01000 is 1.4% below col 38), because distributions and capped losses are +small in dollars. The counts differ by 2.36x. + +The congressional-district file covers a subset of HT2: it leaves out +territories, APO/FPO and foreign addresses and returns without a matched ZIP +code. Its US N01000 is 98.0% of HT2's. + +## What the model measures + +`capital_gains_gross` maps to PE-US `capital_gains`, which is short-term plus +long-term gains, i.e. Schedule D before the loss limit. Distributions reported +without Schedule D are the separate `non_sch_d_capital_gains`, with its own +Table 1.4 col 35 target. The SOI materializer counts a tax unit when the +unit's summed `capital_gains` is positive and sums its positive part, so the +count is col 37 and the sum is col 38. The Build P release met col 37 at +12,391,446 against 12,392,020. + +No released record has negative capital gains, so the model has no Schedule D +loss returns. An N01000 count therefore has no model counterpart at any +level. + +## How the targets combine + +- **Controls.** Only a family registered with the model's concept + (`_SOI_CAPITAL_GAINS_FAMILY_CONCEPTS`, today Table 1.4 alone) can supply the + national level, and the latest national all-AGI fact not after the build + period wins. HT2 and congressional-district rows never do, whatever their + period stamp. Two different records at the winning period raise + `AmbiguousSoiCapitalGainsControlError`, so feed order never decides. +- **Historic Table 2 rows** enter only as shares. Each state row is its share + of the HT2 US total, scaled to the control, landed at the control's period, + and tagged `soi_source_concept = form_1040_line_7_net_gain_or_loss` and + `soi_control_concept = schedule_d_taxable_net_gain`. A row with no control + at or after its own period is dropped, because a line-7 level would be a + 2.36x mismatch. The HT2 national all-AGI rows retire because the control's + own row owns the national concept. +- **Congressional-district rows** are not controls. Their own targets, which + are on the `full` surface only, are not reconciled here. + +The share step assumes that each state's share of line-7 returns equals its +share of Schedule D gain returns. No IRS state product publishes a gain-only +count, so this cannot be checked directly. AGI composition is one measurable +source of difference: weighting each state's HT2 N01000 by AGI class with +Table 1.4's gain share per class would move TY2022 state targets by −2.9% (WV) +to +4.3% (DC), with a median of 1.6% and no state beyond 5%. + +## How the returns control drifted + +The rebase used to keep the first fact among equal-period candidates. The +July pin listed Table 1.4 ty2023 first, so Build P rebased the 51 state +return counts onto 12,392,020 (factor 0.406751165649407; states summing to +12,289,836). The September re-pin (`consumer_facts_us_c5e5bf8`) sorts rows by +`aggregate_fact_key`, a content hash. There, the congressional-district US row +sorts first; it carries TY2022 data stamped ty2023 (PolicyEngine/chronicle#117). +It became the control (factor 0.97964), and the states summed to 29,599,604, +2.39x the national target of the same model quantity on the same surface. The +amount control happened to stay on Table 1.4, but reversing the feed would +have moved it to the congressional-district row (factor 0.92455 instead of +0.77190). + +## Effect on the pinned feed + +Compiled at 2024 with target aging and the packaged congressional-district +crosswalk, then narrowed to `national_state`: + +| Targets | Before | After | +|---|---:|---:| +| 51 HT2 state return counts, sum | 29,599,604 | 12,289,836 | +| California | 3,774,052 | 1,566,997 | +| Texas | 2,108,705 | 875,540 | +| Florida | 2,093,099 | 869,060 | +| New York | 1,974,435 | 819,791 | +| National Table 1.4 return count (unchanged) | 12,392,020 | 12,392,020 | +| `national_state` registry | `d315c75804ef` | `65e4dde11c83` | + +- All 60 HT2 return-count specs move by the same factor, 0.415202720927. That + is the 51 state rows plus 9 congressional-district proxies, which are on + `full` only. +- The 60 matching amount specs keep their values and gain only the two + concept keys. +- No spec enters or leaves either surface, which keeps 32,842 compiled and + 5,694 on `national_state`. +- The state return counts are back at the Build P values; California's + 1,566,996.66 equals the July scorecard's. + +`test_pinned_feed_no_soi_state_family_sums_past_its_national_target` holds +the invariant across all 63 SOI state families on the surface. diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py index 1cf1d0ec5..4b5334f83 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py @@ -45,6 +45,7 @@ from microcosm.calibrate.geography_constants import US_STATE_FIPS_TO_POSTAL __all__ = [ + "AmbiguousSoiCapitalGainsControlError", "US_FISCAL_MACRO_REALISM_BANDS", "US_FISCAL_TARGET_REGISTRY", "US_FISCAL_TARGET_SPECS", @@ -1471,13 +1472,62 @@ class _SoiTotalControl: period_key: tuple[int, int, str] +#: What an SOI record-set family's ``net_capital_gains_*`` columns count, read +#: from the published IRS workbooks and documentation guides, TY2020-TY2023 +#: (microcosm#1035): +#: +#: - Table 1.4 cols 37/38, "Sales of capital assets reported on Form 1040, +#: Schedule D: Taxable net gain", cover Schedule D returns that net to a +#: gain. Returns reporting only capital gain distributions on Form 1040 +#: (col 35) and Schedule D loss returns (col 39) are outside them: Table 1.3 +#: row 18 "Sales of capital assets net gain" equals col 35 + col 37 exactly. +#: - Historic Table 2 and the congressional-district file carry N01000/A01000, +#: which both documentation guides define as "Number of returns with net +#: capital gain (less loss)" / "Net capital gain (less loss) amount" at Form +#: 1040 line 7: a Schedule D gain, a loss-limited Schedule D loss, or +#: distributions only. HT2 US N01000 equals Table 1.4 col 35 + col 37 + +#: col 39 within 0.25% in each of TY2020-TY2023, and is 2.36x col 37 in +#: TY2022. Its amount, net of losses, is 1.4% below col 38 in TY2022. +#: +#: A family absent from this register has no reviewed concept and is never a +#: capital-gains control. +_SOI_SCHEDULE_D_TAXABLE_NET_GAIN = "schedule_d_taxable_net_gain" +_SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS = "form_1040_line_7_net_gain_or_loss" +_SOI_CAPITAL_GAINS_FAMILY_CONCEPTS: dict[str, str] = { + "table_1_4": _SOI_SCHEDULE_D_TAXABLE_NET_GAIN, + "historic_table_2": _SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS, + "congressional_district": _SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS, +} +#: The concept ``capital_gains_gross`` measures. It is the positive part of +#: PE-US ``capital_gains`` (short-term plus long-term, i.e. Schedule D, before +#: the loss limit); distributions reported without Schedule D are the separate +#: ``non_sch_d_capital_gains``. Its indicator therefore counts Table 1.4 col 37 +#: returns and its sum is col 38. The model carries no Schedule D loss returns, +#: so a line-7 count has no model counterpart at any level. +_SOI_CAPITAL_GAINS_MODEL_CONCEPT = _SOI_SCHEDULE_D_TAXABLE_NET_GAIN + + +class AmbiguousSoiCapitalGainsControlError(ValueError): + """Two different concept-matched records tie for one capital-gains control.""" + + def _rebase_stale_soi_capital_gains_distributions( registry: TargetRegistry, facts: tuple[object, ...], *, target_period: int | str, ) -> TargetRegistry: - """Use stale SOI capital-gains rows as shares, not hard old-year totals.""" + """Use Historic Table 2 capital-gains rows as shares of a Table 1.4 level. + + HT2 rows count Form 1040 line 7 (gain or loss), a population + ``capital_gains_gross`` cannot measure, so they never ship as levels. Each + one ships as its share of the HT2 national total, scaled to the + concept-matched control of ``_soi_capital_gains_active_totals`` at or after + its own period, and declares the bridge in ``soi_source_concept`` / + ``soi_control_concept``. Without such a control, the row is dropped. The + national all-AGI HT2 rows retire because the control's own row owns the + national concept. + """ controls = _soi_capital_gains_active_totals( facts, @@ -1494,15 +1544,11 @@ def _rebase_stale_soi_capital_gains_distributions( key = _soi_capital_gains_control_key_from_spec(spec) control = controls.get(key) source_total = stale_national_totals.get((*key, spec.metadata["source_period"])) - if control is None: - specs.append(spec) - continue - if source_total in (None, 0): - continue - if not _period_not_before( + if control is None or not _period_not_before( control.period_key, _period_key_from_value(spec.metadata["source_period"]) ): - specs.append(spec) + continue + if source_total in (None, 0): continue if _is_national_all_agi_spec(spec): @@ -1525,6 +1571,8 @@ def _rebase_stale_soi_capital_gains_distributions( "uprating_index_source_record_id": control.source_record_id, "uprating_factor": _format_float(factor), "stale_distribution_rebased_to_active_total": "true", + "soi_source_concept": _SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS, + "soi_control_concept": _SOI_CAPITAL_GAINS_MODEL_CONCEPT, }, ) ) @@ -1536,13 +1584,25 @@ def _soi_capital_gains_active_totals( *, target_period: int | str, ) -> dict[tuple[str, str, str], _SoiTotalControl]: + """The capital-gains control per (measure, filing status, universe). + + Only a national full-AGI fact whose family counts what + ``capital_gains_gross`` measures qualifies, so HT2 and + congressional-district rows never do, whatever their period stamp. The + latest qualifying period not after the build period wins, and the choice + depends on the fact set alone: two different records at the winning + period refuse the compile. Feed order used to break that tie, and the + September re-pin, which re-sorted the feed by content hash, handed the + returns control to the congressional-district US row (microcosm#1035). + """ controls: dict[tuple[str, str, str], _SoiTotalControl] = {} + tied_record_ids: dict[tuple[str, str, str], set[str]] = {} target_period_key = _period_key_from_value(target_period) for fact in facts: key = _soi_capital_gains_control_key_from_fact(fact) if key is None: continue - if _is_stale_soi_historic_capital_gains_fact(fact): + if _soi_capital_gains_concept(fact) != _SOI_CAPITAL_GAINS_MODEL_CONCEPT: continue period_key = _period_key(fact) if not _not_after_target_period(period_key, target_period_key): @@ -1550,22 +1610,59 @@ def _soi_capital_gains_active_totals( source_record_id = _source_record_id(fact) if not source_record_id: continue - candidate = _SoiTotalControl( - value=_numeric_value(fact), - source_period=str(_period_value(fact)), - source_record_id=source_record_id, - period_key=period_key, - ) current = controls.get(key) - if current is None or _prefer_candidate( - candidate.period_key, - current.period_key, - target_period_key=target_period_key, - ): - controls[key] = candidate + if current is not None and period_key[:2] == current.period_key[:2]: + tied_record_ids[key].add(source_record_id) + continue + if current is None or period_key[:2] > current.period_key[:2]: + controls[key] = _SoiTotalControl( + value=_numeric_value(fact), + source_period=str(_period_value(fact)), + source_record_id=source_record_id, + period_key=period_key, + ) + tied_record_ids[key] = {source_record_id} + ambiguous = { + key: sorted(record_ids) + for key, record_ids in tied_record_ids.items() + if len(record_ids) > 1 + } + if ambiguous: + raise AmbiguousSoiCapitalGainsControlError( + "SOI capital-gains control is ambiguous: different records tie at " + f"the latest period for {ambiguous}. Deduplicate them upstream; " + "feed order must not pick a rebase control (microcosm#1035)." + ) return controls +def _soi_record_set_family(record_set_id: str) -> str: + """An SOI record set's table family, without period or data vintage. + + ``irs_soi.ty2023.table_1_4`` is ``table_1_4``, + ``irs_soi.ty2022.historic_table_2.state_broad`` is ``historic_table_2`` and + ``irs_soi.ty2023.congressional_district_2022.all_returns`` is + ``congressional_district``. + """ + parts = record_set_id.split(".") + if parts[0] != "irs_soi": + return "" + rest = parts[2:] if len(parts) > 1 and _is_period_token(parts[1]) else parts[1:] + if not rest: + return "" + family = rest[0] + if family.startswith("congressional_district"): + return "congressional_district" + return family + + +def _soi_capital_gains_concept(fact: object) -> str | None: + """The reviewed IRS concept of a capital-gains fact's family, if any.""" + return _SOI_CAPITAL_GAINS_FAMILY_CONCEPTS.get( + _soi_record_set_family(_str_at(fact, "layout", "record_set_id")) + ) + + def _soi_capital_gains_stale_national_totals( facts: tuple[object, ...], ) -> dict[tuple[str, str, str, str], float]: diff --git a/packages/microcosm-build/tests/test_us_fiscal_targets.py b/packages/microcosm-build/tests/test_us_fiscal_targets.py index 3559eb845..97b1d992a 100644 --- a/packages/microcosm-build/tests/test_us_fiscal_targets.py +++ b/packages/microcosm-build/tests/test_us_fiscal_targets.py @@ -1179,15 +1179,16 @@ def test_pinned_feed_national_state_surface_restores_the_fences( ) -> None: """The release surface on the pinned feed: 32,842 compiled targets after Medicaid substitution, 5,694 national_state targets at registry - d315c75804ef, 32 CHIP rows none of them for an M-CHIP state, no + 65e4dde11c83, 32 CHIP rows none of them for an M-CHIP state, no other-income row and no tips return count. Before microcosm#956 it was - 32,867 / 5,719 at d5f9d854fe11, and before decision d179 dropped the - ty2020 tips return count it was 32,843 / 5,695 at 386fac439e77 - (docs/us-chronicle-feed-repin.md).""" + 32,867 / 5,719 at d5f9d854fe11, before decision d179 dropped the ty2020 + tips return count it was 32,843 / 5,695 at 386fac439e77, and before + microcosm#1035 moved the capital-gains control back to Table 1.4 it was + d315c75804ef at the same counts (docs/us-chronicle-feed-repin.md).""" registry, surface, _ = pinned_feed_national_state_surface assert len(registry.specs) == 32_842 assert len(surface.specs) == 5_694 - assert surface.version == "d315c75804ef" + assert surface.version == "65e4dde11c83" chip = [ spec for spec in surface.specs @@ -1214,6 +1215,101 @@ def test_pinned_feed_national_state_surface_restores_the_fences( ) +def test_pinned_feed_capital_gains_rows_take_the_table_1_4_level( + pinned_feed_national_state_surface, +) -> None: + """The 51 Historic Table 2 state capital-gains rows take their level from + Table 1.4 ty2023, the concept capital_gains_gross measures, and not from + the congressional-district US row that shares its period stamp and sorts + first in the feed (microcosm#1035). The return counts sum to 12,289,836, + 99.2% of the national 12,392,020 target, as in the Build P release; with + the CD row as control they summed to 29,599,604.""" + _, surface, _ = pinned_feed_national_state_surface + expected = { + "net_capital_gains_returns": ( + "irs_soi.ty2023.table_1_4.all.net_capital_gains_returns", + "0.406751165649407", + 12_289_835.97216556, + ), + "net_capital_gains_amount": ( + "irs_soi.ty2023.table_1_4.all.net_capital_gains_amount", + "0.771900044145164", + None, + ), + } + for measure, (control_id, factor, total) in expected.items(): + rows = [ + spec + for spec in surface.specs + if ".historic_table_2.state_broad." in spec.name + and spec.metadata.get("source_measure_id") == measure + ] + assert len(rows) == 51 + national = next(spec for spec in surface.specs if spec.name == control_id) + for spec in rows: + assert spec.metadata["uprating_index_source_record_id"] == control_id + assert spec.metadata["uprating_factor"] == factor + assert spec.metadata["soi_source_concept"] == ( + "form_1040_line_7_net_gain_or_loss" + ) + assert spec.metadata["soi_control_concept"] == ( + "schedule_d_taxable_net_gain" + ) + state_total = sum(spec.value for spec in rows) + assert state_total < national.value + if total is not None: + assert math.isclose(state_total, total, rel_tol=1e-12) + + +def test_pinned_feed_no_soi_state_family_sums_past_its_national_target( + pinned_feed_national_state_surface, +) -> None: + """Invariant: the states partition the nation (HT2 state rows omit only + other areas and Puerto Rico), so on national_state no SOI state family may + sum past every national target of the same model quantity and filters. + Before microcosm#1035 the capital_gains_gross return counts, one of the 63 + compared families, summed to 2.39x theirs. Some quantities still carry + both a stale HT2 US row and a newer Table 1.x row, a few percent apart, so + the bound is the larger of them.""" + _, surface, _ = pinned_feed_national_state_surface + + def quantity(spec) -> tuple[object, ...]: + metadata = spec.metadata + return ( + metadata.get("variable"), + metadata.get("measure_mode"), + metadata.get("agi_lower_bound"), + metadata.get("agi_upper_bound"), + metadata.get("filing_status"), + metadata.get("soi_return_universe", "all_returns"), + metadata.get("itemized_only"), + tuple( + sorted( + (key, value) + for key, value in metadata.items() + if key.startswith("ledger_filter_") + ) + ), + ) + + national: dict[tuple[object, ...], list[float]] = {} + states: dict[tuple[object, ...], float] = {} + for spec in surface.specs: + if spec.family != "irs_soi": + continue + if spec.metadata.get("state_fips"): + states[quantity(spec)] = states.get(quantity(spec), 0.0) + spec.value + else: + national.setdefault(quantity(spec), []).append(spec.value) + compared = {key: total for key, total in states.items() if key in national} + assert len(compared) == 63 + assert { + key[:2]: total / max(national[key]) + for key, total in compared.items() + if total > max(national[key]) + } == {} + + def test_reviewed_zero_support_facts_are_not_active_targets() -> None: excluded_source_record_id = ( "hhs_acf_tanf.fy2024.cash_assistance.ar." @@ -4506,6 +4602,384 @@ def test_stale_soi_capital_gains_without_source_total_is_dropped() -> None: assert state_record_id not in source_record_ids +_CG_T14_RETURNS = "irs_soi.ty2023.table_1_4.all.net_capital_gains_returns" +_CG_T14_AMOUNT = "irs_soi.ty2023.table_1_4.all.net_capital_gains_amount" +_CG_CD_RECORD_SET = "irs_soi.ty2023.congressional_district_2022.all_returns" +_CG_HT2_CA_RETURNS = ( + "irs_soi.ty2022.historic_table_2.state_broad.ca.all.net_capital_gains_returns" +) +_CG_HT2_CA_AMOUNT = ( + "irs_soi.ty2022.historic_table_2.state_broad.ca.all.net_capital_gains_amount" +) + + +def _capital_gains_ht2_facts() -> list[dict[str, object]]: + """TY2022 Historic Table 2 line-7 rows: US 100 / 1,000 and CA 25 / 250.""" + return [ + _soi_capital_gains_fact( + 2022, + source_record_id=( + "irs_soi.ty2022.historic_table_2.us.all.net_capital_gains_returns" + ), + measure_id="net_capital_gains_returns", + value=100.0, + ), + _soi_capital_gains_fact( + 2022, + source_record_id=( + "irs_soi.ty2022.historic_table_2.us.all.net_capital_gains_amount" + ), + value=1_000.0, + ), + _soi_capital_gains_fact( + 2022, + source_record_id=_CG_HT2_CA_RETURNS, + measure_id="net_capital_gains_returns", + geography_level="state", + geography_id="0400000US06", + value=25.0, + ), + _soi_capital_gains_fact( + 2022, + source_record_id=_CG_HT2_CA_AMOUNT, + geography_level="state", + geography_id="0400000US06", + value=250.0, + ), + ] + + +def _capital_gains_cd_us_facts() -> list[dict[str, object]]: + """The congressional-district file's US row: TY2022 line-7 data stamped + ty2023, as on the pinned feed (PolicyEngine/chronicle#117).""" + return [ + _soi_capital_gains_fact( + 2023, + source_record_id=f"{_CG_CD_RECORD_SET}.us.net_capital_gains_returns", + measure_id="net_capital_gains_returns", + layout_record_set_id=_CG_CD_RECORD_SET, + value=98.0, + ), + _soi_capital_gains_fact( + 2023, + source_record_id=f"{_CG_CD_RECORD_SET}.us.net_capital_gains_amount", + layout_record_set_id=_CG_CD_RECORD_SET, + value=925.0, + ), + ] + + +def _capital_gains_t14_facts() -> list[dict[str, object]]: + """Table 1.4 ty2023 Schedule D taxable net gain: 40 returns / 800.""" + return [ + _soi_capital_gains_fact( + 2023, + source_record_id=_CG_T14_RETURNS, + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=40.0, + ), + _soi_capital_gains_fact( + 2023, + source_record_id=_CG_T14_AMOUNT, + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=800.0, + ), + ] + + +@pytest.mark.parametrize("cd_first", [True, False], ids=["cd-first", "t14-first"]) +def test_capital_gains_controls_are_table_1_4_in_either_feed_order( + cd_first: bool, +) -> None: + """A congressional-district US row that shares Table 1.4's period never + becomes the control, whichever comes first in the feed (microcosm#1035). + On the pinned feed it sorted first and set the HT2 state returns rows at + 2.39x the national Table 1.4 target of the same model quantity.""" + cd, t14 = _capital_gains_cd_us_facts(), _capital_gains_t14_facts() + facts = [ + *packaged_reference_facts(), + *_capital_gains_ht2_facts(), + *(cd + t14 if cd_first else t14 + cd), + ] + + controls = fiscal_targets._soi_capital_gains_active_totals( + tuple(facts), target_period=2024 + ) + assert {key[0]: control.source_record_id for key, control in controls.items()} == { + "net_capital_gains_returns": _CG_T14_RETURNS, + "net_capital_gains_amount": _CG_T14_AMOUNT, + } + + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + specs = {spec.metadata["ledger_source_record_id"]: spec for spec in registry.specs} + ca_returns = specs[_CG_HT2_CA_RETURNS] + ca_amount = specs[_CG_HT2_CA_AMOUNT] + assert ca_returns.value == 10.0 + assert ca_returns.metadata["uprating_factor"] == "0.4" + assert ca_returns.metadata["uprating_index_source_record_id"] == _CG_T14_RETURNS + assert ca_amount.value == 200.0 + assert ca_amount.metadata["uprating_index_source_record_id"] == _CG_T14_AMOUNT + for spec in (ca_returns, ca_amount): + # The bridge is declared: the row supplies a line-7 share, the level + # comes from the Schedule D concept capital_gains_gross measures. + assert spec.metadata["soi_source_concept"] == ( + "form_1040_line_7_net_gain_or_loss" + ) + assert spec.metadata["soi_control_concept"] == "schedule_d_taxable_net_gain" + + +def test_capital_gains_rows_without_a_concept_matched_control_are_dropped() -> None: + """With only a congressional-district row stamped later, no control + exists, and a line-7 HT2 row cannot stand as a capital_gains_gross level.""" + facts = [ + *packaged_reference_facts(), + *_capital_gains_ht2_facts(), + *_capital_gains_cd_us_facts(), + ] + + assert ( + fiscal_targets._soi_capital_gains_active_totals( + tuple(facts), target_period=2024 + ) + == {} + ) + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + source_record_ids = { + spec.metadata["ledger_source_record_id"] for spec in registry.specs + } + assert _CG_HT2_CA_RETURNS not in source_record_ids + assert _CG_HT2_CA_AMOUNT not in source_record_ids + + +def test_capital_gains_rows_whose_control_predates_them_are_dropped() -> None: + """A control older than the HT2 row would reverse its period, so the row + drops rather than ship its line-7 level.""" + facts = [ + *packaged_reference_facts(), + *_capital_gains_ht2_facts(), + _soi_capital_gains_fact( + 2021, + source_record_id="irs_soi.ty2021.table_1_4.all.net_capital_gains_returns", + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2021.table_1_4", + value=41.0, + ), + ] + + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + source_record_ids = { + spec.metadata["ledger_source_record_id"] for spec in registry.specs + } + assert _CG_HT2_CA_RETURNS not in source_record_ids + + +def test_capital_gains_control_refuses_two_matched_records_at_one_period() -> None: + duplicate = _soi_capital_gains_fact( + 2023, + source_record_id="irs_soi.ty2023.table_1_4.all_returns.net_capital_gains_returns", + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=41.0, + ) + facts = (*_capital_gains_t14_facts(), duplicate) + + for ordered in (facts, tuple(reversed(facts))): + with pytest.raises( + fiscal_targets.AmbiguousSoiCapitalGainsControlError, + match="all_returns.net_capital_gains_returns", + ): + fiscal_targets._soi_capital_gains_active_totals(ordered, target_period=2024) + # The same record twice is one candidate, not a tie. + same = (*_capital_gains_t14_facts(), *_capital_gains_t14_facts()) + assert ( + len(fiscal_targets._soi_capital_gains_active_totals(same, target_period=2024)) + == 2 + ) + + +@pytest.mark.parametrize( + ("record_set_id", "family"), + [ + ("irs_soi.ty2023.table_1_4", "table_1_4"), + ("irs_soi.table_1_4", "table_1_4"), + ("irs_soi.ty2022.historic_table_2.us", "historic_table_2"), + ("irs_soi.ty2022.historic_table_2.state_broad", "historic_table_2"), + (_CG_CD_RECORD_SET, "congressional_district"), + ( + "irs_soi.ty2024.congressional_district_2024.all_returns", + "congressional_district", + ), + ("irs_soi.congressional_district_2022.all_returns", "congressional_district"), + ("irs_soi.ty2023.table_4_3.all_returns_excluding_dependents", "table_4_3"), + ("irs_soi.ty2023", ""), + ("cbo.revenue_projection.ty2023", ""), + ("", ""), + ], +) +def test_soi_record_set_family_drops_period_and_data_vintage( + record_set_id: str, family: str +) -> None: + assert fiscal_targets._soi_record_set_family(record_set_id) == family + + +def test_capital_gains_control_is_order_free_latest_and_concept_matched() -> None: + """Property (microcosm#1035): for any set of national capital-gains facts + and any build period, the control per measure is the latest Table 1.4 + fact not after the build period, whatever the feed order, and no + Historic Table 2, congressional-district or unreviewed-family fact is ever + chosen. With no such Table 1.4 fact there is no control.""" + pytest.importorskip("hypothesis") + from hypothesis import given, settings + from hypothesis import strategies as st + + record_sets = { + "table_1_4": "irs_soi.ty{period}.table_1_4", + "historic_table_2": "irs_soi.ty{period}.historic_table_2.us", + "congressional_district": ( + "irs_soi.ty{period}.congressional_district_2022.all_returns" + ), + "unreviewed": "irs_soi.ty{period}.table_4_3", + } + measures = ("net_capital_gains_returns", "net_capital_gains_amount") + candidate = st.tuples( + st.sampled_from(sorted(record_sets)), + st.integers(min_value=2018, max_value=2027), + st.sampled_from(measures), + st.floats(min_value=1.0, max_value=1e12, allow_nan=False), + ) + + @settings(max_examples=300, deadline=None) + @given( + st.lists(candidate, max_size=14, unique_by=lambda item: item[:3]), + st.integers(min_value=2019, max_value=2026), + st.randoms(use_true_random=False), + ) + def check(candidates, target_period, rng) -> None: + facts = [ + _soi_capital_gains_fact( + period, + source_record_id=( + f"{record_sets[family].format(period=period)}.us.{measure}" + ), + measure_id=measure, + layout_record_set_id=record_sets[family].format(period=period), + value=value, + ) + for family, period, measure, value in candidates + ] + shuffled = list(facts) + rng.shuffle(shuffled) + + controls = fiscal_targets._soi_capital_gains_active_totals( + tuple(facts), target_period=target_period + ) + assert controls == fiscal_targets._soi_capital_gains_active_totals( + tuple(shuffled), target_period=target_period + ) + chosen = {key[0]: control for key, control in controls.items()} + for measure in measures: + eligible = [ + (period, value) + for family, period, candidate_measure, value in candidates + if family == "table_1_4" + and candidate_measure == measure + and period <= target_period + ] + if not eligible: + assert measure not in chosen + continue + period, value = max(eligible) + control = chosen[measure] + assert control.source_period == str(period) + assert control.value == value + assert control.source_record_id == ( + f"irs_soi.ty{period}.table_1_4.us.{measure}" + ) + + check() + + +def test_rebased_capital_gains_states_never_outgrow_their_control() -> None: + """Property: whatever the HT2 state rows and national total, rebased + state rows keep their HT2 shares and so sum to the control times the + states' share of the HT2 nation; when the HT2 states sum to at most the + HT2 nation, they sum to at most the control.""" + pytest.importorskip("hypothesis") + from hypothesis import given, settings + from hypothesis import strategies as st + + states = (("ca", "06"), ("ny", "36"), ("tx", "48")) + + @settings(max_examples=25, deadline=None) + @given( + st.lists( + st.integers(min_value=1, max_value=10_000_000), + min_size=len(states), + max_size=len(states), + ), + st.integers(min_value=0, max_value=5_000_000), + st.integers(min_value=1, max_value=50_000_000), + ) + def check(state_values, other_areas, control_value) -> None: + national = sum(state_values) + other_areas + facts = [ + *packaged_reference_facts(), + _soi_capital_gains_fact( + 2022, + source_record_id=( + "irs_soi.ty2022.historic_table_2.us.all.net_capital_gains_returns" + ), + measure_id="net_capital_gains_returns", + value=float(national), + ), + *( + _soi_capital_gains_fact( + 2022, + source_record_id=( + "irs_soi.ty2022.historic_table_2.state_broad." + f"{postal}.all.net_capital_gains_returns" + ), + measure_id="net_capital_gains_returns", + geography_level="state", + geography_id=f"0400000US{fips}", + value=float(value), + ) + for (postal, fips), value in zip(states, state_values, strict=True) + ), + _soi_capital_gains_fact( + 2023, + source_record_id=_CG_T14_RETURNS, + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=float(control_value), + ), + ] + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + rebased = [ + spec + for spec in registry.specs + if ".historic_table_2.state_broad." in spec.name + and spec.metadata.get("source_measure_id") == "net_capital_gains_returns" + ] + assert len(rebased) == len(states) + total = sum(spec.value for spec in rebased) + assert math.isclose( + total, control_value * sum(state_values) / national, rel_tol=1e-12 + ) + assert total <= control_value * (1 + 1e-12) + + check() + + def test_cross_period_soi_taxable_interest_agi_slice_without_total_is_dropped() -> None: source_record_id = ( "irs_soi.ty2022.historic_table_2.us.200k_to_500k.taxable_interest_amount" From 6ebfcf7f4f90bc30fbea63cddb40eff67ed97f2d Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 27 Sep 2026 05:22:06 -0400 Subject: [PATCH 2/3] Refuse any ambiguous capital-gains control; qualify two claims (#1036 review) The chooser now treats one record id carrying two values or period labels at the winning period as ambiguous too, so no feed order can pick the value. Tests cover that, an equivalent period label, and a tie at a superseded period. The model-concept comment scopes "no loss returns" to the Build P release data, and the doc's AGI-mix median is the median absolute change. Co-Authored-By: Claude Opus 5.5 --- docs/us-soi-capital-gains-concepts.md | 2 +- .../build/us_runtime/fiscal_targets.py | 39 +++++++------- .../tests/test_us_fiscal_targets.py | 51 +++++++++++++++++++ 3 files changed, 74 insertions(+), 18 deletions(-) diff --git a/docs/us-soi-capital-gains-concepts.md b/docs/us-soi-capital-gains-concepts.md index 841d6fdf8..4475329c5 100644 --- a/docs/us-soi-capital-gains-concepts.md +++ b/docs/us-soi-capital-gains-concepts.md @@ -75,7 +75,7 @@ share of Schedule D gain returns. No IRS state product publishes a gain-only count, so this cannot be checked directly. AGI composition is one measurable source of difference: weighting each state's HT2 N01000 by AGI class with Table 1.4's gain share per class would move TY2022 state targets by −2.9% (WV) -to +4.3% (DC), with a median of 1.6% and no state beyond 5%. +to +4.3% (DC), with a median absolute change of 1.6% and no state beyond 5%. ## How the returns control drifted diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py index 4b5334f83..acd4751dd 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py @@ -1502,8 +1502,10 @@ class _SoiTotalControl: #: PE-US ``capital_gains`` (short-term plus long-term, i.e. Schedule D, before #: the loss limit); distributions reported without Schedule D are the separate #: ``non_sch_d_capital_gains``. Its indicator therefore counts Table 1.4 col 37 -#: returns and its sum is col 38. The model carries no Schedule D loss returns, -#: so a line-7 count has no model counterpart at any level. +#: returns and its sum is col 38; a line-7 count, which also counts losses and +#: distributions-only returns, is a different population. (PE-US allows +#: negative gains, but the Build P release data carries no tax unit with a +#: net loss, so it could not reach a line-7 count at any level.) _SOI_CAPITAL_GAINS_MODEL_CONCEPT = _SOI_SCHEDULE_D_TAXABLE_NET_GAIN @@ -1590,13 +1592,14 @@ def _soi_capital_gains_active_totals( ``capital_gains_gross`` measures qualifies, so HT2 and congressional-district rows never do, whatever their period stamp. The latest qualifying period not after the build period wins, and the choice - depends on the fact set alone: two different records at the winning - period refuse the compile. Feed order used to break that tie, and the + depends on the fact set alone: two different candidates at the winning + period (different records, or one record with different values or period + labels) refuse the compile. Feed order used to break that tie, and the September re-pin, which re-sorted the feed by content hash, handed the returns control to the congressional-district US row (microcosm#1035). """ controls: dict[tuple[str, str, str], _SoiTotalControl] = {} - tied_record_ids: dict[tuple[str, str, str], set[str]] = {} + tied: dict[tuple[str, str, str], set[tuple[str, float, str]]] = {} target_period_key = _period_key_from_value(target_period) for fact in facts: key = _soi_capital_gains_control_key_from_fact(fact) @@ -1610,26 +1613,28 @@ def _soi_capital_gains_active_totals( source_record_id = _source_record_id(fact) if not source_record_id: continue + candidate = _SoiTotalControl( + value=_numeric_value(fact), + source_period=str(_period_value(fact)), + source_record_id=source_record_id, + period_key=period_key, + ) + identity = (source_record_id, candidate.value, candidate.source_period) current = controls.get(key) if current is not None and period_key[:2] == current.period_key[:2]: - tied_record_ids[key].add(source_record_id) + tied[key].add(identity) continue if current is None or period_key[:2] > current.period_key[:2]: - controls[key] = _SoiTotalControl( - value=_numeric_value(fact), - source_period=str(_period_value(fact)), - source_record_id=source_record_id, - period_key=period_key, - ) - tied_record_ids[key] = {source_record_id} + controls[key] = candidate + tied[key] = {identity} ambiguous = { - key: sorted(record_ids) - for key, record_ids in tied_record_ids.items() - if len(record_ids) > 1 + key: sorted(candidates) + for key, candidates in tied.items() + if len(candidates) > 1 } if ambiguous: raise AmbiguousSoiCapitalGainsControlError( - "SOI capital-gains control is ambiguous: different records tie at " + "SOI capital-gains control is ambiguous: different candidates tie at " f"the latest period for {ambiguous}. Deduplicate them upstream; " "feed order must not pick a rebase control (microcosm#1035)." ) diff --git a/packages/microcosm-build/tests/test_us_fiscal_targets.py b/packages/microcosm-build/tests/test_us_fiscal_targets.py index 97b1d992a..d6508f484 100644 --- a/packages/microcosm-build/tests/test_us_fiscal_targets.py +++ b/packages/microcosm-build/tests/test_us_fiscal_targets.py @@ -4804,6 +4804,57 @@ def test_capital_gains_control_refuses_two_matched_records_at_one_period() -> No ) +def _t14_returns_fact(period: object, *, record_id: str, value: float) -> dict: + fact = _soi_capital_gains_fact( + 2023, + source_record_id=record_id, + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=value, + ) + fact["period"] = {**fact["period"], "value": period} + return fact + + +@pytest.mark.parametrize( + "rival", + [ + # One record id carrying two values: order would pick the value. + {"period": 2023, "record_id": _CG_T14_RETURNS, "value": 41.0}, + # Another record at the same period under an equivalent label. + { + "period": "tax_year_2023", + "record_id": "irs_soi.ty2023.table_1_4.v2.net_capital_gains_returns", + "value": 40.0, + }, + ], + ids=["same-id-other-value", "equivalent-period-label"], +) +def test_capital_gains_control_refuses_any_ambiguous_candidate(rival) -> None: + control = _t14_returns_fact(2023, record_id=_CG_T14_RETURNS, value=40.0) + other = _t14_returns_fact( + rival["period"], record_id=rival["record_id"], value=rival["value"] + ) + for ordered in ((control, other), (other, control)): + with pytest.raises(fiscal_targets.AmbiguousSoiCapitalGainsControlError): + fiscal_targets._soi_capital_gains_active_totals(ordered, target_period=2024) + + +def test_capital_gains_tie_at_a_superseded_period_is_not_ambiguous() -> None: + """Only a tie at the winning period matters: a later control supersedes + an older tie, in any order.""" + older = ( + _t14_returns_fact(2022, record_id="irs_soi.ty2022.table_1_4.a", value=1.0), + _t14_returns_fact(2022, record_id="irs_soi.ty2022.table_1_4.b", value=2.0), + ) + latest = _t14_returns_fact(2023, record_id=_CG_T14_RETURNS, value=40.0) + for ordered in ((*older, latest), (latest, *older), (older[0], latest, older[1])): + (control,) = fiscal_targets._soi_capital_gains_active_totals( + ordered, target_period=2024 + ).values() + assert (control.source_record_id, control.value) == (_CG_T14_RETURNS, 40.0) + + @pytest.mark.parametrize( ("record_set_id", "family"), [ From 011a63b1ad45da8364d7e463fabf327c76d757b9 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 27 Sep 2026 17:02:20 -0400 Subject: [PATCH 3/3] Rebase congressional-district capital gains onto Table 1.4; drop mislabeled CD columns (#1038) On the full target surface the congressional-district file's capital-gains rows compiled as Form 1040 line-7 levels onto capital_gains_gross (Schedule D gain): its states and districts each summed to 29,845,710 returns, 2.41x the Table 1.4 target, and $1,522.2B, 1.198x. They now enter as shares of the file's own US row scaled to the Table 1.4 control, as Historic Table 2 rows do since #1036; the US row retires and every share row names its denominator in soi_share_total_source_record_id. The same package reads N18425/A18425 (state and local income taxes) as the limited SALT deduction and N85530/A85530 (additional Medicare tax) as the premium tax credit. US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS drops facts by (measure id, IRS column), replaces the hard-coded CD PTC-amount check, and is listed in the exclusion receipt and the release coverage report. On the pinned feed: 31,376 compiled targets (was 32,842), no CD-file family past its national target on full (was 3), national_state values and membership unchanged (version moves only for the new provenance key). Co-Authored-By: Claude Opus 5.5 --- .../1038-cd-capital-gains-shares.fixed.md | 1 + docs/us-chronicle-feed-repin.md | 11 + docs/us-soi-capital-gains-concepts.md | 109 ++- .../microcosm/build/us_runtime/__init__.py | 2 + .../build/us_runtime/fiscal_targets.py | 239 ++++-- .../engine_free/us/test_us_fiscal_targets.py | 742 +++++++++++++++++- .../microcosm_build/us_fiscal_targets.py | 52 +- tools/build_us_fiscal_refresh_release.py | 11 + 8 files changed, 1069 insertions(+), 98 deletions(-) create mode 100644 changelog.d/1038-cd-capital-gains-shares.fixed.md diff --git a/changelog.d/1038-cd-capital-gains-shares.fixed.md b/changelog.d/1038-cd-capital-gains-shares.fixed.md new file mode 100644 index 000000000..07af399c0 --- /dev/null +++ b/changelog.d/1038-cd-capital-gains-shares.fixed.md @@ -0,0 +1 @@ +On the `full` target surface, congressional-district capital-gains rows now enter as shares of the file's own US row scaled to the Table 1.4 Schedule D control, as Historic Table 2 rows do; the file's US row retires. Every capital-gains share row names its denominator in `soi_share_total_source_record_id`. A new reviewed source-column exclusion, `US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS`, drops congressional-district facts read from IRS columns their measure id does not mean: N18425/A18425 (state and local income taxes) read as the limited SALT deduction, and N85530/A85530 (additional Medicare tax) read as the premium tax credit. It replaces the hard-coded congressional-district PTC-amount check and is listed in the exclusion receipt. On the pinned feed, the congressional-district capital-gains states and districts each move from 29,845,710 returns (2.41x the national target) and $1,522.2B (1.198x) to 12,392,020 and $1,270.9B, and no congressional-district family sums past its national target on `full` (microcosm#1038). diff --git a/docs/us-chronicle-feed-repin.md b/docs/us-chronicle-feed-repin.md index 907fbefc7..8d2dab4ce 100644 --- a/docs/us-chronicle-feed-repin.md +++ b/docs/us-chronicle-feed-repin.md @@ -181,6 +181,17 @@ chosen by concept and period alone ([us-soi-capital-gains-concepts.md](us-soi-capital-gains-concepts.md)). The counts are unchanged and `national_state` is `65e4dde11c83`. +**Follow-up (27 September 2026, microcosm#1038).** On the `full` surface, +the congressional-district file's capital-gains rows are now shares of its US +row scaled to the same control, and the US row retires. The package's four +mislabeled columns are dropped: N18425/A18425 (state and local income taxes), +read as the limited SALT deduction, and N85530/A85530 (additional Medicare +tax), read as the premium tax credit. That leaves 31,376 compiled targets, all of them on +`full` (`059dc56d78db`). `national_state` keeps its 5,694 targets and values. +It moves to `47dce807b412` only because its 102 Historic Table 2 +capital-gains rows now name their share denominator in +`soi_share_total_source_record_id`. + The labelled feed exposed one latent defect in microcosm, fixed in this change: `apply_us_medicaid_enrollment_substitutions` built Rhode Island's substituted spec by cloning a neighbouring state's spec and kept that state's diff --git a/docs/us-soi-capital-gains-concepts.md b/docs/us-soi-capital-gains-concepts.md index 4475329c5..784cdecd7 100644 --- a/docs/us-soi-capital-gains-concepts.md +++ b/docs/us-soi-capital-gains-concepts.md @@ -67,8 +67,13 @@ level. at or after its own period is dropped, because a line-7 level would be a 2.36x mismatch. The HT2 national all-AGI rows retire because the control's own row owns the national concept. -- **Congressional-district rows** are not controls. Their own targets, which - are on the `full` surface only, are not reconciled here. +- **Congressional-district rows** are not controls either, and their own + targets (on the `full` surface only) enter the same way: each state and + district row is its share of the congressional-district file's US row, + scaled to the control and tagged with the same two concept keys, and the + file's US row retires (microcosm#1038, below). Every share row names its + denominator, the HT2 or congressional-district US row, in + `soi_share_total_source_record_id`. The share step assumes that each state's share of line-7 returns equals its share of Schedule D gain returns. No IRS state product publishes a gain-only @@ -118,3 +123,103 @@ crosswalk, then narrowed to `national_state`: `test_pinned_feed_no_soi_state_family_sums_past_its_national_target` holds the invariant across all 63 SOI state families on the surface. + +## Congressional-district rows on the full surface (microcosm#1038) + +The release tool's default `--target-surface` is `full`, which also carries +the congressional-district file's rows: a US row, 51 state rows and 436 +current districts per measure. Before microcosm#1038 they compiled as line-7 +levels onto `capital_gains_gross`. On the pinned feed, their states and their +districts each summed to 29,845,710 returns, 2.41x the national Table 1.4 +target of 12,392,020 on the same surface. Their amounts summed to $1,522.2B +against $1,270.9B, 1.198x. The amount gap is mostly the file's TY2022 data +stamped ty2023 and aged from 2023. The count gap is the concept. + +There were two ways to fix this. Option (a) rebases the rows as shares, like +the HT2 rows. Option (b) drops them with a reviewed exclusion. Microcosm#1038 +takes (a): + +- **The share assumption is already the HT2 one.** Each district's share of + line-7 returns stands in for its share of Schedule D gain returns. AGI + composition is the measurable source of difference. Reweighting each of the + 428 numbered districts' N01000 by AGI class with Table 1.4's per-class gain + share moves its share by a median 2.1% (95th percentile 4.7%, largest + CA-18 +9.7%). Amounts move by a median 0.5% (95th percentile 2.4%). +- **Dropping the rows would discard a much larger signal.** District + capital-gains incidence varies far more than state incidence. N01000 per + return runs from 8.3% to 31.5% between the 5th and 95th percentile + districts, against 13.7% to 23.4% across states. Within a state, the + district share left over after the district's AGI distribution has a log + standard deviation of 0.227 for counts and 0.223 for amounts. The + AGI-composition error of the share proxy is 0.024 and 0.009. +- **The rebased rows agree with the HT2 state rows about as well as other + pairs on `full` already do.** The congressional-district and HT2 state + return counts for capital gains differ by a median 0.6% (95th percentile + 2.1%). The `return_count` pair differs by 1.6% (2.4%). + - Amounts differ more: a median 2.8% (95th percentile 15.4%, Montana + −27.0%). The cause is item suppression in the congressional-district + file's A01000. Montana's is 32.5% below HT2's at the state level. That is + still inside the qualified-dividends amount pair, 11.6% (21.7%). + +The congressional-district file has no rows for other areas, so its US row +equals the sum of its 51 states. The rebased states and districts therefore +each sum to the control itself. HT2's US row includes other areas, so its +rebased states sum to 99.2% of the control for returns and 98.0% for +amounts. + +**Vintage.** A share divides a row by the US row of the same package, so the +package's period stamp cancels. The value lands at the control's period +(2023) and ages from there. It is the same whether 22incd.csv is stamped +ty2023, as on the pinned feed, or truthfully at 2022 +(`test_congressional_district_capital_gains_ignore_the_package_stamp`). + +microcosm#1030 corrects restamped facts to age from their data year, but it +refuses any restamped fact that was rebased. Before it merges with this change, +that refusal must admit a share whose `soi_share_total_source_record_id` is a +restamp of the same package, and leave `uprating_to_period` at the control's +period. Starting the aging at 2022 instead would apply the Table 1.4 +2022-to-2023 fall twice (x0.761). + +### Effect on the pinned feed + +Release compile at 2024, as above: + +| `full` surface | Before | After | +|---|---:|---:| +| Congressional-district capital-gains returns, states and districts (each sum) | 29,845,710 | 12,392,020 | +| Congressional-district capital-gains amounts, states and districts (each sum) | $1,522.19B | $1,270.86B | +| California returns / amount | 3,808,930 / $214.08B | 1,581,478 / $178.73B | +| New York returns / amount | 2,015,540 / $127.96B | 836,858 / $106.83B | +| Compiled targets, all on `full` | 32,842 (`e02123644d42`) | 31,376 (`059dc56d78db`) | +| `national_state` targets | 5,694 (`65e4dde11c83`) | 5,694 (`47dce807b412`) | + +- The 974 congressional-district capital-gains rows move by one factor per + measure: 0.415202720927 for returns and 0.834893818418 for amounts. +- The two US rows retire. +- The Historic Table 2 rows keep their values. The 120 of them (102 on + `national_state`) gain only `soi_share_total_source_record_id`. + +### Two mislabeled columns in the same package + +On the same surface, Chronicle's `congressional_district_2022` package reads +four IRS columns under measure ids whose concept they do not carry. They are +the only differences between its column mapping and the Historic Table 2 +package's (all 56 measures compared): + +| Measure id | Congressional-district column | Doc-guide definition | Column the measure means | +|---|---|---|---| +| `limited_state_local_taxes_returns` / `_amount` | N18425 / A18425 | State and local income taxes (Schedule A line 5a) | N18460 / A18460, limited state and local taxes (line 5e) | +| `premium_tax_credit_returns` / `_amount` | N85530 / A85530 | Additional Medicare tax (Form 8959 line 24) | N85770 / A85770, total premium tax credit | + +The SALT amount rows summed to 1.98x their national target on `full`. The +premium-tax-credit amount was already dropped by a hard-coded check, but the +returns rows compiled as `assigned_aca_ptc` recipients. + +`US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS` in `fiscal_targets.py` now drops +all four, keyed by (measure id, `layout.source_column_id`). It lists them in +the release's exclusion receipt. A corrected package that reads the right +columns passes untouched. + +`test_pinned_feed_no_congressional_district_family_sums_past_its_national_target` +holds the invariant for every congressional-district family on `full`. + diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/__init__.py b/packages/microcosm-build/src/microcosm/build/us_runtime/__init__.py index 74e4757cb..ac03bb85c 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/__init__.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/__init__.py @@ -291,6 +291,7 @@ US_FISCAL_TARGET_LEDGER_REFERENCES, US_FISCAL_TARGET_REFERENCES, US_FISCAL_TARGET_REGISTRY, + US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS, US_FISCAL_TARGET_SPECS, US_FISCAL_TARGET_SUPPORT_EXCLUSIONS, US_JCT_TAX_EXPENDITURE_REFORMS, @@ -1165,6 +1166,7 @@ "US_FISCAL_TARGET_SUPPORT_EXCLUSIONS", "US_FISCAL_TARGET_ALL_VINTAGE_SUPPORT_EXCLUSIONS", "US_FISCAL_TARGET_EXCLUSION_VINTAGE_BYPASSES", + "US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS", "US_FISCAL_TARGET_COVERAGE_REQUIREMENTS", "US_FISCAL_TARGET_LEDGER_REFERENCES", "US_JCT_TAX_EXPENDITURE_REFORMS", diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py index acd4751dd..648759293 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py @@ -55,6 +55,7 @@ "US_FISCAL_TARGET_SUPPORT_EXCLUSIONS", "US_FISCAL_TARGET_ALL_VINTAGE_SUPPORT_EXCLUSIONS", "US_FISCAL_TARGET_EXCLUSION_VINTAGE_BYPASSES", + "US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS", "US_FISCAL_LEDGER_PARITY_REGISTRY", "US_FISCAL_LEDGER_PARITY_REPORT", "US_JCT_TAX_EXPENDITURE_REFORMS", @@ -854,6 +855,61 @@ def _us_hierarchy_seed( # every vintage. US_FISCAL_TARGET_EXCLUSION_VINTAGE_BYPASSES: dict[str, str] = {} +# Reviewed source-column concept exclusions (microcosm#1038). Each entry is an +# IRS column that a Chronicle package publishes under a measure id whose +# concept the column does not carry, so no fact read from that column +# compiles, whatever its package, geography or vintage, and a package +# corrected to read the right column passes untouched. The key is (measure_id, +# layout.source_column_id) because the IRS column is the definition. The +# congressional-district documentation guide (22incddocguide.docx) defines: +# +# - N18425/A18425 as "State and local income taxes" (Schedule A line 5a) and +# N18460/A18460 as "Limited state and local taxes" (line 5e). salt_deduction, +# like the Historic Table 2 rows of these measure ids, is the limited line 5e +# deduction. +# - N85530/A85530 as "Additional Medicare tax" (Form 8959 line 24) and +# N85770/A85770 as "Total premium tax credit" (Form 8962 line 24; the guide +# prints "8926:24"). N85770/A85770 is what the Historic Table 2 rows of +# these measure ids read and what assigned_aca_ptc measures. The amount was +# dropped before by a hard-coded congressional-district check; the returns +# row compiled as premium-tax-credit recipients. +# +# Chronicle's congressional_district_2022 package reads these four columns +# under the limited_state_local_taxes_* and premium_tax_credit_* measure ids; +# every other measure it shares with Historic Table 2 reads the same column. +US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS: dict[tuple[str, str], str] = { + ("limited_state_local_taxes_returns", "N18425"): ( + "N18425 counts returns with state and local income taxes (Schedule A " + "line 5a), not returns with the limited state and local tax deduction " + "(line 5e, N18460). Chronicle's congressional_district_2022 package " + "reads it under this measure id: TY2022 US 10,842,580 returns against " + "14,777,040 for N18460 in the same file and 14,968,720 in Historic " + "Table 2." + ), + ("limited_state_local_taxes_amount", "A18425"): ( + "A18425 is state and local income taxes before the $10,000 limit " + "(Schedule A line 5a), not the limited deduction (line 5e, A18460) " + "salt_deduction measures. Chronicle's congressional_district_2022 " + "package reads it under this measure id: TY2022 US $250.4B against " + "$121.2B for A18460 in the same file, so on the full surface the " + "congressional-district rows summed to 1.98x the national Historic " + "Table 2 target." + ), + ("premium_tax_credit_returns", "N85530"): ( + "N85530 counts returns with additional Medicare tax (Form 8959 line 24), " + "not returns with the premium tax credit (Form 8962 line 24, N85770) " + "assigned_aca_ptc measures. Chronicle's congressional_district_2022 " + "package reads it under this measure id: TY2022 US 6,848,330 returns " + "against 7,713,000 for N85770 in the same file." + ), + ("premium_tax_credit_amount", "A85530"): ( + "A85530 is additional Medicare tax (Form 8959 line 24), not the premium " + "tax credit (Form 8962 line 24, A85770). Chronicle's " + "congressional_district_2022 package reads it under this measure id: " + "TY2022 US $14.2B against $53.1B for A85770 in the same file." + ), +} + @dataclass(frozen=True) class SimpleTaxExpenditureReform: @@ -1090,6 +1146,9 @@ def us_fiscal_target_exclusion_receipt( vintage included (dropped). - ``m_chip_state_chip_enrollment``: CMS CHIP enrollment facts for an ``_M_CHIP_STATE_FIPS`` state, at every period (dropped). + - ``source_column_concept_exclusion``: SOI facts read from an IRS column + that ``US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS`` lists for their + measure id, at every vintage (dropped). - ``allowlisted_vintage_bypass``: facts latest-vintage selection activates at ``target_period`` that are another vintage of an excluded cell and carry a reviewed ``US_FISCAL_TARGET_EXCLUSION_VINTAGE_BYPASSES`` entry @@ -1108,6 +1167,7 @@ def us_fiscal_target_exclusion_receipt( ) reviewed: set[str] = set() all_vintage: set[str] = set() + source_column: set[str] = set() m_chip: set[str] = set() for fact in materialized_facts: source_record_id = _source_record_id(fact) @@ -1117,6 +1177,8 @@ def us_fiscal_target_exclusion_receipt( all_vintage.add(source_record_id) elif source_record_id in US_FISCAL_TARGET_SUPPORT_EXCLUSIONS: reviewed.add(source_record_id) + elif _is_source_column_concept_exclusion(fact): + source_column.add(source_record_id) elif _source_name(fact) == "cms_medicaid": reference = _reference_from_ledger_fact(fact, target_period=target_period) if reference is not None and _is_m_chip_state_chip_enrollment( @@ -1144,6 +1206,22 @@ def us_fiscal_target_exclusion_receipt( "entries": sorted(US_FISCAL_TARGET_ALL_VINTAGE_SUPPORT_EXCLUSIONS), "source_record_ids": sorted(all_vintage), }, + "source_column_concept_exclusion": { + "action": "dropped", + "scope": "SOI facts read from a listed column under a listed " + "measure id, every vintage", + "entries": [ + { + "measure_id": measure_id, + "source_column_id": source_column_id, + "reason": reason, + } + for (measure_id, source_column_id), reason in sorted( + US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS.items() + ) + ], + "source_record_ids": sorted(source_column), + }, "m_chip_state_chip_enrollment": { "action": "dropped", "scope": "CMS CHIP enrollment rows for these states, every period", @@ -1490,7 +1568,8 @@ class _SoiTotalControl: #: TY2022. Its amount, net of losses, is 1.4% below col 38 in TY2022. #: #: A family absent from this register has no reviewed concept and is never a -#: capital-gains control. +#: capital-gains control. A registered family whose concept is not the +#: model's supplies shares only (``_soi_capital_gains_share_family``). _SOI_SCHEDULE_D_TAXABLE_NET_GAIN = "schedule_d_taxable_net_gain" _SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS = "form_1040_line_7_net_gain_or_loss" _SOI_CAPITAL_GAINS_FAMILY_CONCEPTS: dict[str, str] = { @@ -1519,44 +1598,54 @@ def _rebase_stale_soi_capital_gains_distributions( *, target_period: int | str, ) -> TargetRegistry: - """Use Historic Table 2 capital-gains rows as shares of a Table 1.4 level. - - HT2 rows count Form 1040 line 7 (gain or loss), a population - ``capital_gains_gross`` cannot measure, so they never ship as levels. Each - one ships as its share of the HT2 national total, scaled to the - concept-matched control of ``_soi_capital_gains_active_totals`` at or after - its own period, and declares the bridge in ``soi_source_concept`` / - ``soi_control_concept``. Without such a control, the row is dropped. The - national all-AGI HT2 rows retire because the control's own row owns the - national concept. + """Use line-7 capital-gains rows as shares of a Table 1.4 level. + + Historic Table 2 and congressional-district rows count Form 1040 line 7 + (gain or loss), a population ``capital_gains_gross`` cannot measure, so + they never ship as levels. Each one ships as its share of its own family's + national total at its own period (an HT2 row of the HT2 US row, a + congressional-district state or district row of that file's US row), + scaled to the concept-matched control of ``_soi_capital_gains_active_totals`` + at or after that period, and declares the bridge in ``soi_source_concept`` + / ``soi_control_concept``. Without such a control or family total, the row + is dropped. Each family's national all-AGI row retires because the + control's own row owns the national concept. + + One factor per family, quantity and period scales a family's state and + district rows alike, so the district-to-state hierarchy reconciliation that + runs next sees the same proportions. The congressional-district file's + states sum exactly to its US row, so on the full surface its states and + its districts each sum to the control. Before this, they shipped as + line-7 levels, 2.41x the Table 1.4 return count (microcosm#1038). """ controls = _soi_capital_gains_active_totals( facts, target_period=target_period, ) - stale_national_totals = _soi_capital_gains_stale_national_totals(facts) + share_totals = _soi_capital_gains_share_national_totals(facts) specs: list[TargetSpec] = [] for spec in registry.specs: kind = _soi_capital_gains_kind(spec) - if kind is None or not _is_stale_soi_historic_capital_gains_spec(spec): + family = _soi_capital_gains_share_family_from_spec(spec) if kind else None + if kind is None or family is None: specs.append(spec) continue key = _soi_capital_gains_control_key_from_spec(spec) control = controls.get(key) - source_total = stale_national_totals.get((*key, spec.metadata["source_period"])) + source_total = share_totals.get((family, *key, spec.metadata["source_period"])) if control is None or not _period_not_before( control.period_key, _period_key_from_value(spec.metadata["source_period"]) ): continue - if source_total in (None, 0): + if source_total is None or source_total.value == 0: continue if _is_national_all_agi_spec(spec): continue - factor = control.value / source_total + factor = control.value / source_total.value specs.append( replace( spec, @@ -1573,8 +1662,11 @@ def _rebase_stale_soi_capital_gains_distributions( "uprating_index_source_record_id": control.source_record_id, "uprating_factor": _format_float(factor), "stale_distribution_rebased_to_active_total": "true", - "soi_source_concept": _SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS, + "soi_source_concept": _SOI_CAPITAL_GAINS_FAMILY_CONCEPTS[family], "soi_control_concept": _SOI_CAPITAL_GAINS_MODEL_CONCEPT, + # The share's denominator: this family's own national + # row, from the same release as the row itself. + "soi_share_total_source_record_id": source_total.source_record_id, }, ) ) @@ -1668,15 +1760,76 @@ def _soi_capital_gains_concept(fact: object) -> str | None: ) -def _soi_capital_gains_stale_national_totals( +def _soi_capital_gains_share_family(record_set_id: str) -> str | None: + """The family of a capital-gains row that may ship only as a share. + + A registered family whose columns count something other than what + ``capital_gains_gross`` measures (Historic Table 2 and the + congressional-district file, both Form 1040 line 7) supplies shares, never + levels. The model's own family and unregistered families return ``None``. + """ + family = _soi_record_set_family(record_set_id) + concept = _SOI_CAPITAL_GAINS_FAMILY_CONCEPTS.get(family) + if concept is None or concept == _SOI_CAPITAL_GAINS_MODEL_CONCEPT: + return None + return family + + +def _soi_capital_gains_share_family_from_spec(spec: TargetSpec) -> str | None: + if spec.family != "irs_soi": + return None + return _soi_capital_gains_share_family( + spec.metadata.get("ledger_layout_record_set_id", "") + ) + + +class AmbiguousSoiCapitalGainsShareTotalError(ValueError): + """One share family carries two different national totals for one period.""" + + +@dataclass(frozen=True) +class _SoiShareTotal: + value: float + source_record_id: str + + +def _soi_capital_gains_share_national_totals( facts: tuple[object, ...], -) -> dict[tuple[str, str, str, str], float]: - totals: dict[tuple[str, str, str, str], float] = {} +) -> dict[tuple[str, str, str, str, str], _SoiShareTotal]: + """Each share family's national all-AGI total, the denominator of its shares. + + Keyed by (family, measure, filing status, universe, period), so an HT2 row + divides by the HT2 US row and a congressional-district row by that file's + US row, never by another family's. Two different values under one key + refuse the compile instead of letting feed order pick the denominator; + equal values under two record ids keep the smaller id, so the named + denominator does not depend on feed order either. + """ + totals: dict[tuple[str, str, str, str, str], _SoiShareTotal] = {} for fact in facts: key = _soi_capital_gains_control_key_from_fact(fact) - if key is None or not _is_stale_soi_historic_capital_gains_fact(fact): + if key is None: + continue + family = _soi_capital_gains_share_family( + _str_at(fact, "layout", "record_set_id") + ) + if family is None: continue - totals[(*key, str(_period_value(fact)))] = _numeric_value(fact) + source_record_id = _source_record_id(fact) + if not source_record_id: + continue + total_key = (family, *key, str(_period_value(fact))) + candidate = _SoiShareTotal(_numeric_value(fact), source_record_id) + current = totals.get(total_key) + if current is not None and current.value != candidate.value: + raise AmbiguousSoiCapitalGainsShareTotalError( + "SOI capital-gains share family has two national totals for " + f"{total_key}: {current.value!r} ({current.source_record_id}) " + f"and {candidate.value!r} ({candidate.source_record_id}). " + "Deduplicate them upstream." + ) + if current is None or candidate.source_record_id < current.source_record_id: + totals[total_key] = candidate return totals @@ -1694,22 +1847,6 @@ def _soi_capital_gains_kind_from_measure(measure_id: str) -> str | None: return None -def _is_stale_soi_historic_capital_gains_spec(spec: TargetSpec) -> bool: - if spec.family != "irs_soi": - return False - if _soi_capital_gains_kind(spec) is None: - return False - return ".historic_table_2." in spec.metadata.get("ledger_layout_record_set_id", "") - - -def _is_stale_soi_historic_capital_gains_fact(fact: object) -> bool: - if _source_name(fact) != "irs_soi": - return False - if _soi_capital_gains_kind_from_measure(_measure_id(fact)) is None: - return False - return ".historic_table_2." in _str_at(fact, "layout", "record_set_id") - - def _soi_capital_gains_control_key_from_fact( fact: object, ) -> tuple[str, str, str] | None: @@ -2622,6 +2759,16 @@ def _period_free_source_record_id(source_record_id: str) -> str: return _normalized_record_set_id(source_record_id) +def _is_source_column_concept_exclusion(fact: object) -> bool: + """Whether an SOI fact reads a column its measure id does not mean.""" + if _source_name(fact) != "irs_soi": + return False + return ( + _measure_id(fact), + _str_at(fact, "layout", "source_column_id"), + ) in US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS + + def _is_all_vintage_support_exclusion(source_record_id: str) -> bool: return _period_free_source_record_id(source_record_id) in ( _period_free_source_record_ids(US_FISCAL_TARGET_ALL_VINTAGE_SUPPORT_EXCLUSIONS) @@ -2656,18 +2803,6 @@ def _is_soi_congressional_district_record_set(fact: object) -> bool: return ".congressional_district_2022." in record_set_id -def _is_soi_cd_premium_tax_credit_amount_conflict( - fact: object, *, measure_id: str -) -> bool: - # The SOI CD A85530 amount is not the same gross annual PTC control used by - # assigned_aca_ptc. Keep Historic Table 2 national/state PTC amount controls - # until Ledger has an explicit CD reconciliation for this concept. - return ( - measure_id == "premium_tax_credit_amount" - and _is_soi_congressional_district_record_set(fact) - ) - - def _is_period_token(value: str) -> bool: normalized = value.lower().replace("-", "_") if normalized.startswith("month"): @@ -2734,6 +2869,8 @@ def _reference_from_ledger_fact( return None if _is_all_vintage_support_exclusion(source_record_id): return None + if _is_source_column_concept_exclusion(fact): + return None source_name = _source_name(fact) if source_name == "irs_soi": return _soi_reference_from_fact( @@ -2791,8 +2928,6 @@ def _soi_reference_from_fact( and geography_level != "congressional_district" ): return None - if _is_soi_cd_premium_tax_credit_amount_conflict(fact, measure_id=measure_id): - return None variable = SOI_AMOUNT_MEASURE_VARIABLES.get(measure_id) is_count = False if variable is None: diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_targets.py b/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_targets.py index cc7037cc8..64d27d418 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_targets.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_targets.py @@ -107,10 +107,14 @@ def test_soi_congressional_district_targets_are_always_compiled() -> None: groupby_value_id="hi_01", geography_id="5001700US1501", ), - _soi_congressional_district_fact( + # The pinned feed reads the CD PTC measures from A85530, additional + # Medicare tax, so they never compile (source-column exclusion). + _cd_column_fact( "premium_tax_credit_amount", - 90_000_000, + "A85530", + value=90_000_000, groupby_value_id="hi_01", + geography_level="congressional_district", geography_id="5001700US1501", ), _soi_congressional_district_fact( @@ -164,9 +168,10 @@ def test_soi_congressional_district_targets_are_always_compiled() -> None: geography_level="state", geography_id="0400000US15", ), - _soi_congressional_district_fact( + _cd_column_fact( "premium_tax_credit_amount", - 180_000_000, + "A85530", + value=180_000_000, groupby_value_id="hi_total", geography_level="state", geography_id="0400000US15", @@ -734,12 +739,14 @@ def test_exclusion_vintage_scope_registers_are_consistent() -> None: def test_exclusion_receipt_lists_the_ids_each_rule_acts_on(monkeypatch) -> None: - """One synthetic feed exercises all four rules. The M-CHIP and all-vintage + """One synthetic feed exercises all five rules. The M-CHIP and all-vintage rules list every vintage the feed carries (both CMS months even though a 2024 build selects only 2024-12), exact-key hits land under - reviewed_exclusion, and the allowlisted bypass is the fact selection - activates. The receipt is plain JSON.""" - synthetic, expected = _four_rule_synthetic_feed(monkeypatch) + reviewed_exclusion, the source-column rule lists the facts read from a + listed column and not the one read from the right column, and the + allowlisted bypass is the fact selection activates. The receipt is plain + JSON.""" + synthetic, expected = _exclusion_rule_synthetic_feed(monkeypatch) facts = [*packaged_reference_facts(), *synthetic] receipt = us_fiscal_target_exclusion_receipt(facts, target_period=2024) @@ -750,6 +757,7 @@ def test_exclusion_receipt_lists_the_ids_each_rule_acts_on(monkeypatch) -> None: assert {name: rule["action"] for name, rule in rules.items()} == { "reviewed_exclusion": "dropped", "all_vintage_reviewed_exclusion": "dropped", + "source_column_concept_exclusion": "dropped", "m_chip_state_chip_enrollment": "dropped", "allowlisted_vintage_bypass": "allowed", } @@ -761,11 +769,92 @@ def test_exclusion_receipt_lists_the_ids_each_rule_acts_on(monkeypatch) -> None: assert rules["m_chip_state_chip_enrollment"]["state_fips"] == sorted( _M_CHIP_STATE_FIPS ) + assert [ + (entry["measure_id"], entry["source_column_id"]) + for entry in rules["source_column_concept_exclusion"]["entries"] + ] == sorted(US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS) + + +def test_source_column_exclusion_drops_only_the_mislabeled_column() -> None: + """A congressional-district fact labelled limited_state_local_taxes but read + from N18425/A18425 (state and local income taxes, Schedule A line 5a), or + labelled premium_tax_credit but read from N85530 (additional Medicare + tax), never compiles, at any geography; the same measures read from + A18460 and N85770, the columns a corrected package would read, compile + (microcosm#1038).""" + mislabeled = [ + _cd_column_fact("limited_state_local_taxes_returns", "N18425"), + _cd_column_fact("limited_state_local_taxes_amount", "A18425"), + _cd_column_fact( + "limited_state_local_taxes_returns", + "N18425", + groupby_value_id="hi_total", + geography_level="state", + geography_id="0400000US15", + ), + ] + corrected = _cd_column_fact( + "limited_state_local_taxes_amount", + "A18460", + groupby_value_id="hi_total", + geography_level="state", + geography_id="0400000US15", + ) + medicare = _cd_column_fact("premium_tax_credit_returns", "N85530") + ptc = _cd_column_fact( + "premium_tax_credit_returns", + "N85770", + groupby_value_id="hi_total", + geography_level="state", + geography_id="0400000US15", + ) + mislabeled = [*mislabeled, medicare] + facts = [*packaged_reference_facts(), *mislabeled, corrected, ptc] + + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + + compiled = {spec.metadata["ledger_source_record_id"] for spec in registry.specs} + assert not compiled & {fact["lineage"]["source_record_id"] for fact in mislabeled} + assert corrected["lineage"]["source_record_id"] in compiled + assert ptc["lineage"]["source_record_id"] in compiled + rule = us_fiscal_target_exclusion_receipt(facts)["rules"][ + "source_column_concept_exclusion" + ] + assert rule["source_record_ids"] == sorted( + fact["lineage"]["source_record_id"] for fact in mislabeled + ) + + +def test_source_column_exclusions_name_the_columns_their_guides_define() -> None: + """Each entry's reason names its own column and the column its measure id + means, as 22incddocguide.docx defines them: N18425/A18425 (state and local + income taxes, Schedule A line 5a) stand in for the limited deduction + N18460/A18460 (line 5e), and N85530/A85530 (additional Medicare tax) for + the premium tax credit N85770/A85770.""" + right_columns = { + ("limited_state_local_taxes_returns", "N18425"): "N18460", + ("limited_state_local_taxes_amount", "A18425"): "A18460", + ("premium_tax_credit_returns", "N85530"): "N85770", + ("premium_tax_credit_amount", "A85530"): "A85770", + } + assert set(US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS) == set(right_columns) + for key, reason in US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS.items(): + measure_id, column = key + assert column in reason + assert right_columns[key] in reason + # Counts read N columns and amounts read A columns. + assert ( + column[0] + == right_columns[key][0] + == ("N" if measure_id.endswith("_returns") else "A") + ) def test_exclusion_receipt_is_complete_for_any_subfeed(monkeypatch) -> None: """Invariant (receipt completeness): each rule lists exactly the facts it - drops or allows. For any subset of the four-rule synthetic feed, each rule + drops or allows. For any subset of the five-rule synthetic feed, each rule lists exactly the full feed's ids for that rule that the subset carries. Hypothesis is a workspace dependency; the wheels job installs no test extras, so it skips there.""" @@ -773,7 +862,7 @@ def test_exclusion_receipt_is_complete_for_any_subfeed(monkeypatch) -> None: from hypothesis import given, settings from hypothesis import strategies as st - synthetic, expected = _four_rule_synthetic_feed(monkeypatch) + synthetic, expected = _exclusion_rule_synthetic_feed(monkeypatch) ids = [fiscal_targets._source_record_id(fact) for fact in synthetic] reference = packaged_reference_facts() @@ -859,9 +948,11 @@ def check(years: list[int], reviewed: bool) -> None: def test_pinned_feed_exclusion_receipt(pinned_feed_national_state_surface) -> None: """On the pinned feed the receipt names the 40 M-CHIP CHIP ids (20 states - x 2 CMS months), the 16 other-income ids (ty2020-ty2023) and the ty2020 - and ty2023 tips return counts, and allows no vintage bypass (decision - d179).""" + x 2 CMS months), the 16 other-income ids (ty2020-ty2023), the ty2020 and + ty2023 tips return counts, and the 1,952 congressional-district SALT and + premium-tax-credit facts read from the wrong IRS column (488 per measure + after the district crosswalk: the US row, 51 states, 436 districts), and + allows no vintage bypass (decision d179, microcosm#1038).""" _, _, receipt = pinned_feed_national_state_surface rules = receipt["rules"] m_chip_ids = rules["m_chip_state_chip_enrollment"]["source_record_ids"] @@ -886,23 +977,44 @@ def test_pinned_feed_exclusion_receipt(pinned_feed_national_state_surface) -> No ] ) assert rules["allowlisted_vintage_bypass"]["source_record_ids"] == [] + column_ids = rules["source_column_concept_exclusion"]["source_record_ids"] + assert all(".congressional_district_2022." in i for i in column_ids) + assert { + measure: sum( + f".{measure}" in source_record_id for source_record_id in column_ids + ) + for measure, _ in US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS + } == {measure: 488 for measure, _ in US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS} + assert len(column_ids) == 4 * 488 def test_pinned_feed_national_state_surface_restores_the_fences( pinned_feed_national_state_surface, ) -> None: - """The release surface on the pinned feed: 32,842 compiled targets after - Medicaid substitution, 5,694 national_state targets at registry - 65e4dde11c83, 32 CHIP rows none of them for an M-CHIP state, no - other-income row and no tips return count. Before microcosm#956 it was - 32,867 / 5,719 at d5f9d854fe11, before decision d179 dropped the ty2020 - tips return count it was 32,843 / 5,695 at 386fac439e77, and before - microcosm#1035 moved the capital-gains control back to Table 1.4 it was - d315c75804ef at the same counts (docs/us-chronicle-feed-repin.md).""" + """The release surface on the pinned feed: 31,376 compiled targets after + Medicaid substitution (the full surface, at 059dc56d78db), 5,694 + national_state targets at registry 47dce807b412, 32 CHIP rows none of them + for an M-CHIP state, no other-income row and no tips return count. Before + microcosm#956 it was 32,867 / 5,719 at d5f9d854fe11, before decision d179 + dropped the ty2020 tips return count it was 32,843 / 5,695 at + 386fac439e77, before microcosm#1035 moved the capital-gains control back + to Table 1.4 it was d315c75804ef at the same counts, and before + microcosm#1038 it was 32,842 compiled at 65e4dde11c83 (full e02123644d42): + that change drops 1,466 congressional-district specs (the two + capital-gains US rows and the SALT and premium-tax-credit rows read from + the wrong column) and moves national_state only by one provenance key, + soi_share_total_source_record_id, on the 102 Historic Table 2 + capital-gains rows (docs/us-chronicle-feed-repin.md, + docs/us-soi-capital-gains-concepts.md).""" + from microcosm.calibrate import TargetRegistry + registry, surface, _ = pinned_feed_national_state_surface - assert len(registry.specs) == 32_842 + assert len(registry.specs) == 31_376 + builder = _load_repo_tool("build_us_fiscal_refresh_release") + full, _ = builder._select_target_surface(registry.specs, "full") + assert TargetRegistry(full, country="us").version == "059dc56d78db" assert len(surface.specs) == 5_694 - assert surface.version == "65e4dde11c83" + assert surface.version == "47dce807b412" chip = [ spec for spec in surface.specs @@ -962,6 +1074,9 @@ def test_pinned_feed_capital_gains_rows_take_the_table_1_4_level( national = next(spec for spec in surface.specs if spec.name == control_id) for spec in rows: assert spec.metadata["uprating_index_source_record_id"] == control_id + assert spec.metadata["soi_share_total_source_record_id"] == ( + f"irs_soi.ty2022.historic_table_2.us.all.{measure}" + ) assert spec.metadata["uprating_factor"] == factor assert spec.metadata["soi_source_concept"] == ( "form_1040_line_7_net_gain_or_loss" @@ -986,25 +1101,7 @@ def test_pinned_feed_no_soi_state_family_sums_past_its_national_target( both a stale HT2 US row and a newer Table 1.x row, a few percent apart, so the bound is the larger of them.""" _, surface, _ = pinned_feed_national_state_surface - - def quantity(spec) -> tuple[object, ...]: - metadata = spec.metadata - return ( - metadata.get("variable"), - metadata.get("measure_mode"), - metadata.get("agi_lower_bound"), - metadata.get("agi_upper_bound"), - metadata.get("filing_status"), - metadata.get("soi_return_universe", "all_returns"), - metadata.get("itemized_only"), - tuple( - sorted( - (key, value) - for key, value in metadata.items() - if key.startswith("ledger_filter_") - ) - ), - ) + quantity = _soi_quantity national: dict[tuple[object, ...], list[float]] = {} states: dict[tuple[object, ...], float] = {} @@ -1024,6 +1121,133 @@ def quantity(spec) -> tuple[object, ...]: } == {} +def _soi_quantity(spec) -> tuple[object, ...]: + """The model quantity and filters an SOI target constrains.""" + metadata = spec.metadata + return ( + metadata.get("variable"), + metadata.get("measure_mode"), + metadata.get("agi_lower_bound"), + metadata.get("agi_upper_bound"), + metadata.get("filing_status"), + metadata.get("soi_return_universe", "all_returns"), + metadata.get("itemized_only"), + tuple( + sorted( + (key, value) + for key, value in metadata.items() + if key.startswith("ledger_filter_") + ) + ), + ) + + +def _is_congressional_district_file_spec(spec) -> bool: + return ( + fiscal_targets._soi_record_set_family( + spec.metadata.get("ledger_layout_record_set_id", "") + ) + == "congressional_district" + ) + + +def test_pinned_feed_congressional_district_capital_gains_take_the_table_1_4_level( + pinned_feed_national_state_surface, +) -> None: + """On the full surface the congressional-district file's 51 state and 436 + district capital-gains rows per measure are shares of that file's US row + times the Table 1.4 ty2023 level, and the US row retires + (microcosm#1038). The states, and the districts, each sum to the national + target: 12,392,020 returns, where the line-7 levels summed to 29,845,710 + (2.41x), and the aged Table 1.4 amount, where they summed to $1,522.2B + (1.198x).""" + registry, _, _ = pinned_feed_national_state_surface + specs = {spec.name: spec for spec in registry.specs} + for measure, control_id, cd_us_id, total in ( + ( + "net_capital_gains_returns", + "irs_soi.ty2023.table_1_4.all.net_capital_gains_returns", + f"{_CG_CD_RECORD_SET}.us.net_capital_gains_returns", + 12_392_020.0, + ), + ( + "net_capital_gains_amount", + "irs_soi.ty2023.table_1_4.all.net_capital_gains_amount", + f"{_CG_CD_RECORD_SET}.us.net_capital_gains_amount", + None, + ), + ): + rows = [ + spec + for spec in registry.specs + if spec.metadata.get("source_measure_id") == measure + and _is_congressional_district_file_spec(spec) + ] + levels: dict[str, list] = {} + for spec in rows: + levels.setdefault(spec.metadata["ledger_geography_level"], []).append(spec) + assert {level: len(group) for level, group in levels.items()} == { + "state": 51, + "congressional_district": 436, + } + assert cd_us_id not in specs + national = specs[control_id] + factors = {spec.metadata["uprating_factor"] for spec in rows} + assert len(factors) == 1 + for spec in rows: + assert spec.metadata["uprating_index_source_record_id"] == control_id + assert spec.metadata["soi_share_total_source_record_id"] == cd_us_id + assert spec.metadata["soi_source_concept"] == ( + "form_1040_line_7_net_gain_or_loss" + ) + assert spec.metadata["soi_control_concept"] == ( + "schedule_d_taxable_net_gain" + ) + for level in ("state", "congressional_district"): + level_total = sum(spec.value for spec in levels[level]) + assert math.isclose(level_total, national.value, rel_tol=1e-12) + if total is not None: + assert math.isclose(level_total, total, rel_tol=1e-12) + + +def test_pinned_feed_no_congressional_district_family_sums_past_its_national_target( + pinned_feed_national_state_surface, +) -> None: + """Invariant: the congressional-district file covers a subset of returns + (no Puerto Rico or other areas, ZIP-unmatched returns dropped), so on the + full surface none of its families (its US row, its states, or its + districts) may sum past every national non-district target of the same + model quantity and filters. Before microcosm#1038 three of 38 compared + families did: capital-gains returns at 2.41x, the SALT amount at 1.98x + (read from the uncapped income-tax column A18425) and the capital-gains + amount at 1.198x. 35 of today's 52 families have a national comparator; + the rest (EITC by children, charitable, interest paid, QBI, taxable + interest, filer individuals) have none on this surface.""" + registry, _, _ = pinned_feed_national_state_surface + builder = _load_repo_tool("build_us_fiscal_refresh_release") + full, _ = builder._select_target_surface(registry.specs, "full") + + national: dict[tuple[object, ...], list[float]] = {} + district_file: dict[tuple[object, ...], dict[str, float]] = {} + for spec in full: + if spec.family != "irs_soi": + continue + level = spec.metadata.get("ledger_geography_level") + if _is_congressional_district_file_spec(spec): + sums = district_file.setdefault(_soi_quantity(spec), {}) + sums[level] = sums.get(level, 0.0) + spec.value + elif level == "country": + national.setdefault(_soi_quantity(spec), []).append(spec.value) + compared = {key: sums for key, sums in district_file.items() if key in national} + assert (len(district_file), len(compared)) == (52, 35) + assert { + (key[:2], level): total / max(national[key]) + for key, sums in compared.items() + for level, total in sums.items() + if total > max(national[key]) * (1 + 1e-9) + } == {} + + def test_reviewed_zero_support_facts_are_not_active_targets() -> None: excluded_source_record_id = ( "hhs_acf_tanf.fy2024.cash_assistance.ar." @@ -4556,6 +4780,440 @@ def check(state_values, other_areas, control_value) -> None: check() +def _capital_gains_cd_package_facts( + *, + us: tuple[float, float] = (98.0, 925.0), + hawaii: tuple[float, float] = (4.0, 40.0), + districts: tuple[tuple[float, float], ...] = ((1.5, 10.0), (2.5, 30.0)), +) -> list[dict[str, object]]: + """A congressional-district package's line-7 rows, stamped ty2023 like the + pinned feed: the US row, Hawaii, and Hawaii's two districts, each as + (returns, amount).""" + facts = [] + for index, measure in enumerate( + ("net_capital_gains_returns", "net_capital_gains_amount") + ): + facts.append( + _soi_congressional_district_fact( + measure, + us[index], + groupby_value_id="us", + geography_level="country", + geography_id="0100000US", + ) + ) + facts.append( + _soi_congressional_district_fact( + measure, + hawaii[index], + groupby_value_id="hi_total", + geography_level="state", + geography_id="0400000US15", + ) + ) + for number, values in enumerate(districts, start=1): + facts.append( + _soi_congressional_district_fact( + measure, + values[index], + groupby_value_id=f"hi_0{number}", + geography_id=f"5001700US150{number}", + ) + ) + return facts + + +def _congressional_district_capital_gains_specs(registry) -> dict[str, object]: + return { + spec.name: spec + for spec in registry.specs + if spec.metadata.get("source_measure_id") + in {"net_capital_gains_returns", "net_capital_gains_amount"} + and ".congressional_district_" in spec.metadata["ledger_layout_record_set_id"] + } + + +def test_congressional_district_capital_gains_rows_are_shares_of_table_1_4() -> None: + """The congressional-district file's N01000/A01000 count Form 1040 line 7, + like Historic Table 2, so its state and district rows ship as shares of + that file's own US row times the Table 1.4 level; the US row retires + (microcosm#1038). On the pinned feed they shipped as line-7 levels, and + the US returns row was 2.41x the Table 1.4 target of the same model + quantity on the full surface.""" + facts = [ + *packaged_reference_facts(), + *_capital_gains_ht2_facts(), + *_capital_gains_cd_package_facts(), + *_capital_gains_t14_facts(), + ] + + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + + cd = _congressional_district_capital_gains_specs(registry) + assert not [ + name + for name, spec in cd.items() + if spec.metadata["ledger_geography_level"] == "country" + ] + for measure, control_id, control, cd_us, hawaii, districts in ( + ("net_capital_gains_returns", _CG_T14_RETURNS, 40.0, 98.0, 4.0, (1.5, 2.5)), + ("net_capital_gains_amount", _CG_T14_AMOUNT, 800.0, 925.0, 40.0, (10.0, 30.0)), + ): + factor = control / cd_us + state = cd[f"{_CG_CD_RECORD_SET}.hi_total.{measure}"] + children = [cd[f"{_CG_CD_RECORD_SET}.hi_0{n}.{measure}"] for n in (1, 2)] + assert math.isclose(state.value, hawaii * factor, rel_tol=1e-12) + for child, raw in zip(children, districts, strict=True): + assert math.isclose(child.value, raw * factor, rel_tol=1e-12) + # One factor scales the state and its districts, so the hierarchy + # reconciliation that follows leaves the proportions alone. + assert child.metadata["hierarchy_reconciliation_factor"] == "1" + for spec in (state, *children): + assert spec.metadata["uprating_factor"] == fiscal_targets._format_float( + factor + ) + assert spec.metadata["uprating_index_source_record_id"] == control_id + assert spec.metadata["soi_source_concept"] == ( + "form_1040_line_7_net_gain_or_loss" + ) + assert spec.metadata["soi_control_concept"] == ( + "schedule_d_taxable_net_gain" + ) + + # Historic Table 2 rows keep dividing by the HT2 US row (100 / 1,000), not + # by the congressional-district US row. + specs = {spec.metadata["ledger_source_record_id"]: spec for spec in registry.specs} + assert specs[_CG_HT2_CA_RETURNS].value == 10.0 + assert specs[_CG_HT2_CA_AMOUNT].value == 200.0 + + +#: The pinned feed's 22incd.csv: TY2022 data that Chronicle stamps ty2023 +#: (PolicyEngine/chronicle#117). A source-vintage correction keys on these. +_CD_2022_SOURCE_SHA256 = ( + "137522878af78d624cfddc0e17cd36ae76a21b7bed166c7bb96ad3243a18a668" +) + + +def _capital_gains_cd_package_at(stamp: int) -> list[dict[str, object]]: + """The congressional-district capital-gains rows of + ``_capital_gains_cd_package_facts``, stamped ``stamp``, carrying the + pinned 22incd.csv's digest and raw key.""" + record_set_id = f"irs_soi.ty{stamp}.congressional_district_2022.all_returns" + facts = [] + for fact in _capital_gains_cd_package_facts(): + row = fact["layout"]["groupby_value_id"] + measure = fact["layout"]["measure_id"] + restamped = _dynamic_ledger_fact( + source_record_id=f"{record_set_id}.{row}.{measure}", + source_name="irs_soi", + measure_id=measure, + value=fact["value"], + period_value=stamp, + geography_level=fact["geography"]["level"], + geography_id=fact["geography"]["id"], + dimensions={"income_range": "all", "filing_status": "all"}, + layout_record_set_id=record_set_id, + groupby_dimension="irs_soi.congressional_district", + groupby_value_id=row, + ) + restamped["source"].update( + { + "source_file": "22incd.csv", + "source_sha256": _CD_2022_SOURCE_SHA256, + "raw_r2_key": ( + "raw/irs_soi/soi-congressional-district-2022/2022/" + f"{_CD_2022_SOURCE_SHA256}/22incd.csv" + ), + "vintage": f"tax_year_{stamp}", + } + ) + facts.append(restamped) + return facts + + +def test_congressional_district_capital_gains_ignore_the_package_stamp() -> None: + """Differential: a congressional-district capital-gains row is its share of + the same package's US row times the Table 1.4 level, so the package's + period stamp cancels. Compiled and aged to 2024, the package stamped + ty2023, as on the pinned feed, gives exactly the values of the same package + stamped truthfully at 2022, and its amounts age from the control's 2023 + period on the CBO net-capital-gain ratio, not from 2022. The facts carry + the pinned file's digest and raw key, so a source-vintage correction that + recognises the restamp sees them: it must leave these values alone + (microcosm#1030, microcosm#1038).""" + cbo = [ + _cbo_income_source_projection_fact(2023, "net_capital_gain", value=981.4e9), + _cbo_income_source_projection_fact(2024, "net_capital_gain", value=1290.9e9), + ] + compiled = {} + for stamp in (2023, 2022): + registry = compile_us_fiscal_target_registry( + [ + *packaged_reference_facts(), + *_capital_gains_cd_package_at(stamp), + *_capital_gains_t14_facts(), + *cbo, + ], + target_period=2024, + age_targets=True, + allow_unaged_dollar_targets=True, + ) + compiled[stamp] = { + ( + spec.metadata["source_measure_id"], + spec.metadata["ledger_geography_id"], + ): (spec) + for spec in _congressional_district_capital_gains_specs(registry).values() + } + + assert set(compiled[2023]) == set(compiled[2022]) + assert len(compiled[2023]) == 6 + ratio = 1290.9e9 / 981.4e9 + for key, spec in compiled[2023].items(): + truthful = compiled[2022][key] + assert math.isclose(spec.value, truthful.value, rel_tol=1e-12), key + assert spec.metadata["uprating_to_period"] == "2023" + if key[0] == "net_capital_gains_amount": + for aged in (spec, truthful): + assert math.isclose( + float(aged.metadata["aging_factor"]), ratio, rel_tol=1e-12 + ) + share = {"0400000US15": 40.0, "5001700US1501": 10.0}.get(key[1], 30.0) + assert math.isclose( + spec.value, share / 925.0 * 800.0 * ratio, rel_tol=1e-12 + ) + + +def test_congressional_district_capital_gains_rows_without_a_control_are_dropped() -> ( + None +): + """Without a Schedule D control no congressional-district capital-gains row + compiles: the file's line-7 counts are never a capital_gains_gross level.""" + facts = [*packaged_reference_facts(), *_capital_gains_cd_package_facts()] + + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + + assert _congressional_district_capital_gains_specs(registry) == {} + + +def test_capital_gains_share_family_refuses_two_national_totals() -> None: + """Two different national totals for one share family, quantity and period + refuse the compile: feed order must not pick a share denominator.""" + rival = _soi_congressional_district_fact( + "net_capital_gains_returns", + 99.0, + groupby_value_id="us_rival", + geography_level="country", + geography_id="0100000US", + ) + facts = [ + *packaged_reference_facts(), + *_capital_gains_cd_package_facts(), + *_capital_gains_t14_facts(), + rival, + ] + + with pytest.raises(fiscal_targets.AmbiguousSoiCapitalGainsShareTotalError): + fiscal_targets._soi_capital_gains_share_national_totals(tuple(facts)) + with pytest.raises(fiscal_targets.AmbiguousSoiCapitalGainsShareTotalError): + compile_us_fiscal_target_registry(facts, allow_unaged_dollar_targets=True) + + +@pytest.mark.parametrize("twin_first", [True, False], ids=["twin-first", "us-first"]) +def test_capital_gains_share_denominator_is_order_free(twin_first: bool) -> None: + """Two records carrying the same national total for one share family are + not ambiguous, and the denominator a share row names is the smaller + record id whichever comes first in the feed.""" + twin = _soi_congressional_district_fact( + "net_capital_gains_returns", + 98.0, + groupby_value_id="us_twin", + geography_level="country", + geography_id="0100000US", + ) + package = _capital_gains_cd_package_facts() + facts = [ + *packaged_reference_facts(), + *((twin, *package) if twin_first else (*package, twin)), + *_capital_gains_t14_facts(), + ] + + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + + hawaii = _congressional_district_capital_gains_specs(registry)[ + f"{_CG_CD_RECORD_SET}.hi_total.net_capital_gains_returns" + ] + assert hawaii.metadata["soi_share_total_source_record_id"] == min( + f"{_CG_CD_RECORD_SET}.us.net_capital_gains_returns", + f"{_CG_CD_RECORD_SET}.us_twin.net_capital_gains_returns", + ) + assert math.isclose(hawaii.value, 4.0 * 40.0 / 98.0, rel_tol=1e-12) + + +@pytest.mark.parametrize( + ("record_set_id", "family"), + [ + ("irs_soi.ty2022.historic_table_2.state_broad.ca", "historic_table_2"), + ( + "irs_soi.ty2023.congressional_district_2022.all_returns", + "congressional_district", + ), + ( + "irs_soi.ty2024.congressional_district_2024.all_returns", + "congressional_district", + ), + ("irs_soi.ty2023.table_1_4", None), + ("irs_soi.ty2023.itemized_all_returns", None), + ("bea.cy2023.regional", None), + ], +) +def test_capital_gains_share_families_are_the_line_7_families( + record_set_id: str, family: str | None +) -> None: + """Only a registered family whose concept is not the model's supplies + shares: Table 1.4 is the model's own concept and an unregistered family + has no reviewed concept.""" + assert fiscal_targets._soi_capital_gains_share_family(record_set_id) == family + + +def test_congressional_district_capital_gains_shares_conserve_the_control() -> None: + """Property: whatever the congressional-district package, the Table 1.4 + level and the feed order, no congressional-district capital-gains row + ships at the country level or without the Schedule D bridge; each state + row is its share of the file's US row times the control; each state's + districts sum to that state row; and the states, and the districts, sum + to the control times the states' share of the US row, so never past the + control. The order is part of the draw, so the result is order-free.""" + pytest.importorskip("hypothesis") + from hypothesis import given, settings + from hypothesis import strategies as st + + # Two-district states and one single-district state, whose source row is + # the state total only (22incd.csv carries no district row for them). + two_district_states = (("hi", "15"), ("id", "16"), ("me", "23")) + at_large = ("vt", "50") + measure = "net_capital_gains_returns" + values = st.integers(min_value=1, max_value=10_000_000) + + @settings(max_examples=25, deadline=None) + @given( + districts=st.lists( + st.tuples(values, values), + min_size=len(two_district_states), + max_size=len(two_district_states), + ), + state_excess=st.lists( + st.integers(min_value=0, max_value=1_000_000), + min_size=len(two_district_states), + max_size=len(two_district_states), + ), + at_large_value=values, + other_areas=st.integers(min_value=0, max_value=50_000_000), + control=values, + order=st.randoms(use_true_random=False), + ) + def check(districts, state_excess, at_large_value, other_areas, control, order): + facts = [] + state_values = {} + for (postal, fips), pair, excess in zip( + two_district_states, districts, state_excess, strict=True + ): + state_values[fips] = float(sum(pair) + excess) + facts.append( + _soi_congressional_district_fact( + measure, + state_values[fips], + groupby_value_id=f"{postal}_total", + geography_level="state", + geography_id=f"0400000US{fips}", + ) + ) + for number, value in enumerate(pair, start=1): + facts.append( + _soi_congressional_district_fact( + measure, + float(value), + groupby_value_id=f"{postal}_0{number}", + geography_id=f"5001700US{fips}0{number}", + ) + ) + state_values[at_large[1]] = float(at_large_value) + facts.append( + _soi_congressional_district_fact( + measure, + float(at_large_value), + groupby_value_id=f"{at_large[0]}_total", + geography_level="state", + geography_id=f"0400000US{at_large[1]}", + ) + ) + cd_us = sum(state_values.values()) + other_areas + facts.append( + _soi_congressional_district_fact( + measure, + cd_us, + groupby_value_id="us", + geography_level="country", + geography_id="0100000US", + ) + ) + facts.append( + _soi_capital_gains_fact( + 2023, + source_record_id=_CG_T14_RETURNS, + measure_id=measure, + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=float(control), + ) + ) + order.shuffle(facts) + registry = compile_us_fiscal_target_registry( + [*packaged_reference_facts(), *facts], allow_unaged_dollar_targets=True + ) + + cd = _congressional_district_capital_gains_specs(registry).values() + by_level: dict[str, list] = {} + for spec in cd: + by_level.setdefault(spec.metadata["ledger_geography_level"], []).append( + spec + ) + assert spec.metadata["soi_control_concept"] == ( + "schedule_d_taxable_net_gain" + ) + assert spec.metadata["uprating_index_source_record_id"] == _CG_T14_RETURNS + assert set(by_level) == {"state", "congressional_district"} + assert len(by_level["state"]) == len(state_values) + assert len(by_level["congressional_district"]) == 2 * len(two_district_states) + + factor = control / cd_us + for spec in by_level["state"]: + raw = state_values[spec.metadata["state_fips"]] + assert math.isclose(spec.value, raw * factor, rel_tol=1e-12) + state_by_fips = { + spec.metadata["state_fips"]: spec.value for spec in by_level["state"] + } + district_totals: dict[str, float] = {} + for spec in by_level["congressional_district"]: + fips = spec.metadata["state_fips"] + district_totals[fips] = district_totals.get(fips, 0.0) + spec.value + for fips, total in district_totals.items(): + assert math.isclose(total, state_by_fips[fips], rel_tol=1e-12) + + expected = control * sum(state_values.values()) / cd_us + states_total = sum(state_by_fips.values()) + assert math.isclose(states_total, expected, rel_tol=1e-12) + assert states_total <= control * (1 + 1e-12) + assert sum(district_totals.values()) <= control * (1 + 1e-12) + + check() + + def test_cross_period_soi_taxable_interest_agi_slice_without_total_is_dropped() -> None: source_record_id = ( "irs_soi.ty2022.historic_table_2.us.200k_to_500k.taxable_interest_amount" diff --git a/test_support/microcosm_build/us_fiscal_targets.py b/test_support/microcosm_build/us_fiscal_targets.py index 7f83264f1..4cb6fb7a4 100644 --- a/test_support/microcosm_build/us_fiscal_targets.py +++ b/test_support/microcosm_build/us_fiscal_targets.py @@ -18,6 +18,7 @@ US_FISCAL_TARGET_COVERAGE_REQUIREMENTS, US_FISCAL_TARGET_EXCLUSION_VINTAGE_BYPASSES, US_FISCAL_TARGET_REFERENCES, + US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS, US_FISCAL_TARGET_SUPPORT_EXCLUSIONS, US_JCT_TAX_EXPENDITURE_REFORMS, US_NONNEGATIVE_SOURCE_OUTPUTS, @@ -160,12 +161,37 @@ def _w2_tips_return_count_fact(tax_year: int) -> dict[str, object]: ) -def _four_rule_synthetic_feed(monkeypatch): +def _cd_column_fact( + measure_id: str, + source_column_id: str, + *, + groupby_value_id: str = "us", + geography_level: str = "country", + geography_id: str = "0100000US", + value: float = 1_000_000.0, +) -> dict[str, object]: + """A congressional-district fact read from IRS column ``source_column_id``, + as every pinned-feed fact records in ``layout.source_column_id``.""" + fact = _soi_congressional_district_fact( + measure_id, + value, + groupby_value_id=groupby_value_id, + geography_level=geography_level, + geography_id=geography_id, + ) + fact["layout"]["source_column_id"] = source_column_id + return fact + + +def _exclusion_rule_synthetic_feed(monkeypatch): """Synthetic facts that exercise every receipt rule, and each rule's ids. The register holds no bypass since d179, so this reviews a synthetic one: a taxable-interest cell excluded at ty2023 with its ty2020 vintage - allowlisted. + allowlisted. The congressional-district SALT facts read the income-tax + column N18425/A18425 that the pinned feed labels as the limited deduction, + and one reads N18460, the column a corrected package would read, which + must compile. """ tanf = ( "hhs_acf_tanf.fy2024.cash_assistance.ar." @@ -217,8 +243,30 @@ def _four_rule_synthetic_feed(monkeypatch): ) ), ] + salt_excluded = [ + _cd_column_fact("limited_state_local_taxes_returns", "N18425"), + _cd_column_fact("limited_state_local_taxes_amount", "A18425"), + _cd_column_fact( + "limited_state_local_taxes_amount", + "A18425", + groupby_value_id="hi_total", + geography_level="state", + geography_id="0400000US15", + ), + ] + salt_kept = _cd_column_fact( + "limited_state_local_taxes_returns", + "N18460", + groupby_value_id="hi_total", + geography_level="state", + geography_id="0400000US15", + ) + facts = [*facts, *salt_excluded, salt_kept] expected = { "reviewed_exclusion": sorted([tanf, interest_excluded]), + "source_column_concept_exclusion": sorted( + fact["lineage"]["source_record_id"] for fact in salt_excluded + ), "all_vintage_reviewed_exclusion": sorted( [ *( diff --git a/tools/build_us_fiscal_refresh_release.py b/tools/build_us_fiscal_refresh_release.py index e421049cf..7cc40f261 100644 --- a/tools/build_us_fiscal_refresh_release.py +++ b/tools/build_us_fiscal_refresh_release.py @@ -86,6 +86,7 @@ SIPP_2023_VOLUNTARY_FILING_DONOR_SIZE_BYTES, SOI_VARIABLE_MAP, US_FISCAL_TARGET_COVERAGE_REQUIREMENTS, + US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS, US_FISCAL_TARGET_SUPPORT_EXCLUSIONS, US_JCT_TAX_EXPENDITURE_REFORMS, US_MEDICAID_ENROLLMENT_TARGET_TABLE, @@ -14687,6 +14688,16 @@ def _main(argv: Sequence[str] | None = None) -> None: US_FISCAL_TARGET_SUPPORT_EXCLUSIONS.items() ) ] + coverage["fiscal_target_source_column_exclusions"] = [ + { + "measure_id": measure_id, + "source_column_id": source_column_id, + "reason": reason, + } + for (measure_id, source_column_id), reason in sorted( + US_FISCAL_TARGET_SOURCE_COLUMN_EXCLUSIONS.items() + ) + ] coverage[US_FISCAL_TARGET_EXCLUSION_RECEIPT_KEY] = fiscal_target_exclusion_receipt write_us_source_coverage_diagnostics( coverage, release_dir / "us_source_coverage.json"