From 0f343249fd9a53816f336b9d34b811f38b650727 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 27 Sep 2026 04:55:28 -0400 Subject: [PATCH 1/2] Choose the capital-gains rebase control by concept and period, not feed order Historic Table 2 and congressional-district N01000 count Form 1040 line 7 returns (gain, loss-limited loss or distributions only); Table 1.4 col 37, the population capital_gains_gross measures, counts Schedule D gain returns only. The rebase control is now the latest Table 1.4 fact not after the build period, CD and HT2 rows never qualify, a tie between matched records refuses the compile, and a line-7 row without a usable control is dropped instead of shipping as a level. Rebased rows declare the bridge in soi_source_concept / soi_control_concept. On the pinned feed the 51 national_state HT2 state return counts go from 29,599,604 (2.39x the national 12,392,020 target) back to Build P's 12,289,836; amounts keep their values. national_state registry d315c75804ef -> 65e4dde11c83 (microcosm#1035). Co-Authored-By: Claude Opus 5.5 --- ...035-capital-gains-control-concept.fixed.md | 1 + docs/us-chronicle-feed-repin.md | 10 + docs/us-soi-capital-gains-concepts.md | 120 +++++ .../build/us_runtime/fiscal_targets.py | 139 ++++- .../tests/test_us_fiscal_targets.py | 484 +++++++++++++++++- 5 files changed, 728 insertions(+), 26 deletions(-) create mode 100644 changelog.d/1035-capital-gains-control-concept.fixed.md create mode 100644 docs/us-soi-capital-gains-concepts.md diff --git a/changelog.d/1035-capital-gains-control-concept.fixed.md b/changelog.d/1035-capital-gains-control-concept.fixed.md new file mode 100644 index 000000000..0e190d99d --- /dev/null +++ b/changelog.d/1035-capital-gains-control-concept.fixed.md @@ -0,0 +1 @@ +Rebase Historic Table 2 capital-gains rows only onto the Table 1.4 Schedule D taxable-net-gain control, the concept `capital_gains_gross` measures. Feed order no longer picks the control: congressional-district and Historic Table 2 rows (Form 1040 line 7, gain or loss) never qualify, a tie between matched records refuses the compile, and a line-7 row with no usable control is dropped instead of shipping as a level. On the pinned feed the 51 `national_state` state return-count targets move from 29,599,604 (2.39x the national target) to 12,289,836 (microcosm#1035). diff --git a/docs/us-chronicle-feed-repin.md b/docs/us-chronicle-feed-repin.md index 14aee6d60..8bf63b09c 100644 --- a/docs/us-chronicle-feed-repin.md +++ b/docs/us-chronicle-feed-repin.md @@ -171,6 +171,16 @@ compiled register goes from 32,867 to 32,843 targets, and the count, the only change: 32,842 compiled targets and 5,694 in `national_state` (`d315c75804ef`). +**Erratum (27 September 2026, microcosm#1035).** Row order changed too. The +new feed is sorted by the re-derived `aggregate_fact_key`s, and the +capital-gains rebase kept the first of two equal-period controls. The +congressional-district US row (TY2022 data stamped ty2023) now sorted ahead of +Table 1.4 ty2023. It took over the returns control and raised the 51 Historic +Table 2 state return counts from 12,289,836 to 29,599,604. The control is now +chosen by concept and period alone +([us-soi-capital-gains-concepts.md](us-soi-capital-gains-concepts.md)). The +counts are unchanged and `national_state` is `65e4dde11c83`. + The labelled feed exposed one latent defect in microcosm, fixed in this change: `apply_us_medicaid_enrollment_substitutions` built Rhode Island's substituted spec by cloning a neighbouring state's spec and kept that state's diff --git a/docs/us-soi-capital-gains-concepts.md b/docs/us-soi-capital-gains-concepts.md new file mode 100644 index 000000000..841d6fdf8 --- /dev/null +++ b/docs/us-soi-capital-gains-concepts.md @@ -0,0 +1,120 @@ +# SOI capital-gains concepts and the rebase control + +`capital_gains_gross` targets come from three IRS SOI products that count +different populations under the same measure ids (`net_capital_gains_returns`, +`net_capital_gains_amount`). This page records what each one counts, which +one the model measures, and how `fiscal_targets.py` combines them +(microcosm#1035). + +## What each table counts + +| Source | Columns | What it counts | +|---|---|---| +| Table 1.4 (`23in14ar.xls`) | col 37 / 38, "Sales of capital assets reported on Form 1040, Schedule D: Taxable net gain" | Schedule D returns that net to a gain, and that gain | +| Table 1.4 | col 39 / 40, "Taxable net loss" | Schedule D returns that net to a loss, and the loss after the $3,000 limit | +| Table 1.4 | col 35 / 36, "Capital gain distributions reported on Form 1040" | Returns that report only distributions, without Schedule D | +| Historic Table 2 (`22in55cmcsv.csv`), congressional-district file (`22incd.csv`) | N01000 / A01000 | "Number of returns with net capital gain (less loss)" and its amount, at Form 1040 line 7 (both documentation guides) | + +Line 7 carries a Schedule D gain, a loss-limited Schedule D loss, or +distributions reported without Schedule D (2022 Form 1040 instructions, line +7, Exception 1). Columns 35, 37 and 39 are disjoint: Table 1.3 row 18, "Sales +of capital assets net gain", equals col 35 + col 37 exactly in TY2022 and +TY2023. + +N01000 is the union of the three Table 1.4 populations: + +| Tax year | HT2 US N01000 | Table 1.4 col 35 + 37 + 39 | Difference | HT2 A01000 vs col 36 + 38 − 40 | +|---|---:|---:|---:|---:| +| 2020 | 29,008,620 | 29,003,885 | +0.016% | −0.63% | +| 2021 | 32,996,180 | 33,076,998 | −0.244% | −0.29% | +| 2022 | 30,465,850 | 30,461,045 | +0.016% | −0.17% | +| 2023 | 29,481,840 | 29,434,188 | +0.162% | −0.28% | + +In TY2022, col 37 (12,915,122) is 42.4% of N01000. The amounts are close +(A01000 is 1.4% below col 38), because distributions and capped losses are +small in dollars. The counts differ by 2.36x. + +The congressional-district file covers a subset of HT2: it leaves out +territories, APO/FPO and foreign addresses and returns without a matched ZIP +code. Its US N01000 is 98.0% of HT2's. + +## What the model measures + +`capital_gains_gross` maps to PE-US `capital_gains`, which is short-term plus +long-term gains, i.e. Schedule D before the loss limit. Distributions reported +without Schedule D are the separate `non_sch_d_capital_gains`, with its own +Table 1.4 col 35 target. The SOI materializer counts a tax unit when the +unit's summed `capital_gains` is positive and sums its positive part, so the +count is col 37 and the sum is col 38. The Build P release met col 37 at +12,391,446 against 12,392,020. + +No released record has negative capital gains, so the model has no Schedule D +loss returns. An N01000 count therefore has no model counterpart at any +level. + +## How the targets combine + +- **Controls.** Only a family registered with the model's concept + (`_SOI_CAPITAL_GAINS_FAMILY_CONCEPTS`, today Table 1.4 alone) can supply the + national level, and the latest national all-AGI fact not after the build + period wins. HT2 and congressional-district rows never do, whatever their + period stamp. Two different records at the winning period raise + `AmbiguousSoiCapitalGainsControlError`, so feed order never decides. +- **Historic Table 2 rows** enter only as shares. Each state row is its share + of the HT2 US total, scaled to the control, landed at the control's period, + and tagged `soi_source_concept = form_1040_line_7_net_gain_or_loss` and + `soi_control_concept = schedule_d_taxable_net_gain`. A row with no control + at or after its own period is dropped, because a line-7 level would be a + 2.36x mismatch. The HT2 national all-AGI rows retire because the control's + own row owns the national concept. +- **Congressional-district rows** are not controls. Their own targets, which + are on the `full` surface only, are not reconciled here. + +The share step assumes that each state's share of line-7 returns equals its +share of Schedule D gain returns. No IRS state product publishes a gain-only +count, so this cannot be checked directly. AGI composition is one measurable +source of difference: weighting each state's HT2 N01000 by AGI class with +Table 1.4's gain share per class would move TY2022 state targets by −2.9% (WV) +to +4.3% (DC), with a median of 1.6% and no state beyond 5%. + +## How the returns control drifted + +The rebase used to keep the first fact among equal-period candidates. The +July pin listed Table 1.4 ty2023 first, so Build P rebased the 51 state +return counts onto 12,392,020 (factor 0.406751165649407; states summing to +12,289,836). The September re-pin (`consumer_facts_us_c5e5bf8`) sorts rows by +`aggregate_fact_key`, a content hash. There, the congressional-district US row +sorts first; it carries TY2022 data stamped ty2023 (PolicyEngine/chronicle#117). +It became the control (factor 0.97964), and the states summed to 29,599,604, +2.39x the national target of the same model quantity on the same surface. The +amount control happened to stay on Table 1.4, but reversing the feed would +have moved it to the congressional-district row (factor 0.92455 instead of +0.77190). + +## Effect on the pinned feed + +Compiled at 2024 with target aging and the packaged congressional-district +crosswalk, then narrowed to `national_state`: + +| Targets | Before | After | +|---|---:|---:| +| 51 HT2 state return counts, sum | 29,599,604 | 12,289,836 | +| California | 3,774,052 | 1,566,997 | +| Texas | 2,108,705 | 875,540 | +| Florida | 2,093,099 | 869,060 | +| New York | 1,974,435 | 819,791 | +| National Table 1.4 return count (unchanged) | 12,392,020 | 12,392,020 | +| `national_state` registry | `d315c75804ef` | `65e4dde11c83` | + +- All 60 HT2 return-count specs move by the same factor, 0.415202720927. That + is the 51 state rows plus 9 congressional-district proxies, which are on + `full` only. +- The 60 matching amount specs keep their values and gain only the two + concept keys. +- No spec enters or leaves either surface, which keeps 32,842 compiled and + 5,694 on `national_state`. +- The state return counts are back at the Build P values; California's + 1,566,996.66 equals the July scorecard's. + +`test_pinned_feed_no_soi_state_family_sums_past_its_national_target` holds +the invariant across all 63 SOI state families on the surface. diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py index 1cf1d0ec5..4b5334f83 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py @@ -45,6 +45,7 @@ from microcosm.calibrate.geography_constants import US_STATE_FIPS_TO_POSTAL __all__ = [ + "AmbiguousSoiCapitalGainsControlError", "US_FISCAL_MACRO_REALISM_BANDS", "US_FISCAL_TARGET_REGISTRY", "US_FISCAL_TARGET_SPECS", @@ -1471,13 +1472,62 @@ class _SoiTotalControl: period_key: tuple[int, int, str] +#: What an SOI record-set family's ``net_capital_gains_*`` columns count, read +#: from the published IRS workbooks and documentation guides, TY2020-TY2023 +#: (microcosm#1035): +#: +#: - Table 1.4 cols 37/38, "Sales of capital assets reported on Form 1040, +#: Schedule D: Taxable net gain", cover Schedule D returns that net to a +#: gain. Returns reporting only capital gain distributions on Form 1040 +#: (col 35) and Schedule D loss returns (col 39) are outside them: Table 1.3 +#: row 18 "Sales of capital assets net gain" equals col 35 + col 37 exactly. +#: - Historic Table 2 and the congressional-district file carry N01000/A01000, +#: which both documentation guides define as "Number of returns with net +#: capital gain (less loss)" / "Net capital gain (less loss) amount" at Form +#: 1040 line 7: a Schedule D gain, a loss-limited Schedule D loss, or +#: distributions only. HT2 US N01000 equals Table 1.4 col 35 + col 37 + +#: col 39 within 0.25% in each of TY2020-TY2023, and is 2.36x col 37 in +#: TY2022. Its amount, net of losses, is 1.4% below col 38 in TY2022. +#: +#: A family absent from this register has no reviewed concept and is never a +#: capital-gains control. +_SOI_SCHEDULE_D_TAXABLE_NET_GAIN = "schedule_d_taxable_net_gain" +_SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS = "form_1040_line_7_net_gain_or_loss" +_SOI_CAPITAL_GAINS_FAMILY_CONCEPTS: dict[str, str] = { + "table_1_4": _SOI_SCHEDULE_D_TAXABLE_NET_GAIN, + "historic_table_2": _SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS, + "congressional_district": _SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS, +} +#: The concept ``capital_gains_gross`` measures. It is the positive part of +#: PE-US ``capital_gains`` (short-term plus long-term, i.e. Schedule D, before +#: the loss limit); distributions reported without Schedule D are the separate +#: ``non_sch_d_capital_gains``. Its indicator therefore counts Table 1.4 col 37 +#: returns and its sum is col 38. The model carries no Schedule D loss returns, +#: so a line-7 count has no model counterpart at any level. +_SOI_CAPITAL_GAINS_MODEL_CONCEPT = _SOI_SCHEDULE_D_TAXABLE_NET_GAIN + + +class AmbiguousSoiCapitalGainsControlError(ValueError): + """Two different concept-matched records tie for one capital-gains control.""" + + def _rebase_stale_soi_capital_gains_distributions( registry: TargetRegistry, facts: tuple[object, ...], *, target_period: int | str, ) -> TargetRegistry: - """Use stale SOI capital-gains rows as shares, not hard old-year totals.""" + """Use Historic Table 2 capital-gains rows as shares of a Table 1.4 level. + + HT2 rows count Form 1040 line 7 (gain or loss), a population + ``capital_gains_gross`` cannot measure, so they never ship as levels. Each + one ships as its share of the HT2 national total, scaled to the + concept-matched control of ``_soi_capital_gains_active_totals`` at or after + its own period, and declares the bridge in ``soi_source_concept`` / + ``soi_control_concept``. Without such a control, the row is dropped. The + national all-AGI HT2 rows retire because the control's own row owns the + national concept. + """ controls = _soi_capital_gains_active_totals( facts, @@ -1494,15 +1544,11 @@ def _rebase_stale_soi_capital_gains_distributions( key = _soi_capital_gains_control_key_from_spec(spec) control = controls.get(key) source_total = stale_national_totals.get((*key, spec.metadata["source_period"])) - if control is None: - specs.append(spec) - continue - if source_total in (None, 0): - continue - if not _period_not_before( + if control is None or not _period_not_before( control.period_key, _period_key_from_value(spec.metadata["source_period"]) ): - specs.append(spec) + continue + if source_total in (None, 0): continue if _is_national_all_agi_spec(spec): @@ -1525,6 +1571,8 @@ def _rebase_stale_soi_capital_gains_distributions( "uprating_index_source_record_id": control.source_record_id, "uprating_factor": _format_float(factor), "stale_distribution_rebased_to_active_total": "true", + "soi_source_concept": _SOI_FORM_1040_LINE_7_NET_GAIN_OR_LOSS, + "soi_control_concept": _SOI_CAPITAL_GAINS_MODEL_CONCEPT, }, ) ) @@ -1536,13 +1584,25 @@ def _soi_capital_gains_active_totals( *, target_period: int | str, ) -> dict[tuple[str, str, str], _SoiTotalControl]: + """The capital-gains control per (measure, filing status, universe). + + Only a national full-AGI fact whose family counts what + ``capital_gains_gross`` measures qualifies, so HT2 and + congressional-district rows never do, whatever their period stamp. The + latest qualifying period not after the build period wins, and the choice + depends on the fact set alone: two different records at the winning + period refuse the compile. Feed order used to break that tie, and the + September re-pin, which re-sorted the feed by content hash, handed the + returns control to the congressional-district US row (microcosm#1035). + """ controls: dict[tuple[str, str, str], _SoiTotalControl] = {} + tied_record_ids: dict[tuple[str, str, str], set[str]] = {} target_period_key = _period_key_from_value(target_period) for fact in facts: key = _soi_capital_gains_control_key_from_fact(fact) if key is None: continue - if _is_stale_soi_historic_capital_gains_fact(fact): + if _soi_capital_gains_concept(fact) != _SOI_CAPITAL_GAINS_MODEL_CONCEPT: continue period_key = _period_key(fact) if not _not_after_target_period(period_key, target_period_key): @@ -1550,22 +1610,59 @@ def _soi_capital_gains_active_totals( source_record_id = _source_record_id(fact) if not source_record_id: continue - candidate = _SoiTotalControl( - value=_numeric_value(fact), - source_period=str(_period_value(fact)), - source_record_id=source_record_id, - period_key=period_key, - ) current = controls.get(key) - if current is None or _prefer_candidate( - candidate.period_key, - current.period_key, - target_period_key=target_period_key, - ): - controls[key] = candidate + if current is not None and period_key[:2] == current.period_key[:2]: + tied_record_ids[key].add(source_record_id) + continue + if current is None or period_key[:2] > current.period_key[:2]: + controls[key] = _SoiTotalControl( + value=_numeric_value(fact), + source_period=str(_period_value(fact)), + source_record_id=source_record_id, + period_key=period_key, + ) + tied_record_ids[key] = {source_record_id} + ambiguous = { + key: sorted(record_ids) + for key, record_ids in tied_record_ids.items() + if len(record_ids) > 1 + } + if ambiguous: + raise AmbiguousSoiCapitalGainsControlError( + "SOI capital-gains control is ambiguous: different records tie at " + f"the latest period for {ambiguous}. Deduplicate them upstream; " + "feed order must not pick a rebase control (microcosm#1035)." + ) return controls +def _soi_record_set_family(record_set_id: str) -> str: + """An SOI record set's table family, without period or data vintage. + + ``irs_soi.ty2023.table_1_4`` is ``table_1_4``, + ``irs_soi.ty2022.historic_table_2.state_broad`` is ``historic_table_2`` and + ``irs_soi.ty2023.congressional_district_2022.all_returns`` is + ``congressional_district``. + """ + parts = record_set_id.split(".") + if parts[0] != "irs_soi": + return "" + rest = parts[2:] if len(parts) > 1 and _is_period_token(parts[1]) else parts[1:] + if not rest: + return "" + family = rest[0] + if family.startswith("congressional_district"): + return "congressional_district" + return family + + +def _soi_capital_gains_concept(fact: object) -> str | None: + """The reviewed IRS concept of a capital-gains fact's family, if any.""" + return _SOI_CAPITAL_GAINS_FAMILY_CONCEPTS.get( + _soi_record_set_family(_str_at(fact, "layout", "record_set_id")) + ) + + def _soi_capital_gains_stale_national_totals( facts: tuple[object, ...], ) -> dict[tuple[str, str, str, str], float]: diff --git a/packages/microcosm-build/tests/test_us_fiscal_targets.py b/packages/microcosm-build/tests/test_us_fiscal_targets.py index 3559eb845..97b1d992a 100644 --- a/packages/microcosm-build/tests/test_us_fiscal_targets.py +++ b/packages/microcosm-build/tests/test_us_fiscal_targets.py @@ -1179,15 +1179,16 @@ def test_pinned_feed_national_state_surface_restores_the_fences( ) -> None: """The release surface on the pinned feed: 32,842 compiled targets after Medicaid substitution, 5,694 national_state targets at registry - d315c75804ef, 32 CHIP rows none of them for an M-CHIP state, no + 65e4dde11c83, 32 CHIP rows none of them for an M-CHIP state, no other-income row and no tips return count. Before microcosm#956 it was - 32,867 / 5,719 at d5f9d854fe11, and before decision d179 dropped the - ty2020 tips return count it was 32,843 / 5,695 at 386fac439e77 - (docs/us-chronicle-feed-repin.md).""" + 32,867 / 5,719 at d5f9d854fe11, before decision d179 dropped the ty2020 + tips return count it was 32,843 / 5,695 at 386fac439e77, and before + microcosm#1035 moved the capital-gains control back to Table 1.4 it was + d315c75804ef at the same counts (docs/us-chronicle-feed-repin.md).""" registry, surface, _ = pinned_feed_national_state_surface assert len(registry.specs) == 32_842 assert len(surface.specs) == 5_694 - assert surface.version == "d315c75804ef" + assert surface.version == "65e4dde11c83" chip = [ spec for spec in surface.specs @@ -1214,6 +1215,101 @@ def test_pinned_feed_national_state_surface_restores_the_fences( ) +def test_pinned_feed_capital_gains_rows_take_the_table_1_4_level( + pinned_feed_national_state_surface, +) -> None: + """The 51 Historic Table 2 state capital-gains rows take their level from + Table 1.4 ty2023, the concept capital_gains_gross measures, and not from + the congressional-district US row that shares its period stamp and sorts + first in the feed (microcosm#1035). The return counts sum to 12,289,836, + 99.2% of the national 12,392,020 target, as in the Build P release; with + the CD row as control they summed to 29,599,604.""" + _, surface, _ = pinned_feed_national_state_surface + expected = { + "net_capital_gains_returns": ( + "irs_soi.ty2023.table_1_4.all.net_capital_gains_returns", + "0.406751165649407", + 12_289_835.97216556, + ), + "net_capital_gains_amount": ( + "irs_soi.ty2023.table_1_4.all.net_capital_gains_amount", + "0.771900044145164", + None, + ), + } + for measure, (control_id, factor, total) in expected.items(): + rows = [ + spec + for spec in surface.specs + if ".historic_table_2.state_broad." in spec.name + and spec.metadata.get("source_measure_id") == measure + ] + assert len(rows) == 51 + national = next(spec for spec in surface.specs if spec.name == control_id) + for spec in rows: + assert spec.metadata["uprating_index_source_record_id"] == control_id + assert spec.metadata["uprating_factor"] == factor + assert spec.metadata["soi_source_concept"] == ( + "form_1040_line_7_net_gain_or_loss" + ) + assert spec.metadata["soi_control_concept"] == ( + "schedule_d_taxable_net_gain" + ) + state_total = sum(spec.value for spec in rows) + assert state_total < national.value + if total is not None: + assert math.isclose(state_total, total, rel_tol=1e-12) + + +def test_pinned_feed_no_soi_state_family_sums_past_its_national_target( + pinned_feed_national_state_surface, +) -> None: + """Invariant: the states partition the nation (HT2 state rows omit only + other areas and Puerto Rico), so on national_state no SOI state family may + sum past every national target of the same model quantity and filters. + Before microcosm#1035 the capital_gains_gross return counts, one of the 63 + compared families, summed to 2.39x theirs. Some quantities still carry + both a stale HT2 US row and a newer Table 1.x row, a few percent apart, so + the bound is the larger of them.""" + _, surface, _ = pinned_feed_national_state_surface + + def quantity(spec) -> tuple[object, ...]: + metadata = spec.metadata + return ( + metadata.get("variable"), + metadata.get("measure_mode"), + metadata.get("agi_lower_bound"), + metadata.get("agi_upper_bound"), + metadata.get("filing_status"), + metadata.get("soi_return_universe", "all_returns"), + metadata.get("itemized_only"), + tuple( + sorted( + (key, value) + for key, value in metadata.items() + if key.startswith("ledger_filter_") + ) + ), + ) + + national: dict[tuple[object, ...], list[float]] = {} + states: dict[tuple[object, ...], float] = {} + for spec in surface.specs: + if spec.family != "irs_soi": + continue + if spec.metadata.get("state_fips"): + states[quantity(spec)] = states.get(quantity(spec), 0.0) + spec.value + else: + national.setdefault(quantity(spec), []).append(spec.value) + compared = {key: total for key, total in states.items() if key in national} + assert len(compared) == 63 + assert { + key[:2]: total / max(national[key]) + for key, total in compared.items() + if total > max(national[key]) + } == {} + + def test_reviewed_zero_support_facts_are_not_active_targets() -> None: excluded_source_record_id = ( "hhs_acf_tanf.fy2024.cash_assistance.ar." @@ -4506,6 +4602,384 @@ def test_stale_soi_capital_gains_without_source_total_is_dropped() -> None: assert state_record_id not in source_record_ids +_CG_T14_RETURNS = "irs_soi.ty2023.table_1_4.all.net_capital_gains_returns" +_CG_T14_AMOUNT = "irs_soi.ty2023.table_1_4.all.net_capital_gains_amount" +_CG_CD_RECORD_SET = "irs_soi.ty2023.congressional_district_2022.all_returns" +_CG_HT2_CA_RETURNS = ( + "irs_soi.ty2022.historic_table_2.state_broad.ca.all.net_capital_gains_returns" +) +_CG_HT2_CA_AMOUNT = ( + "irs_soi.ty2022.historic_table_2.state_broad.ca.all.net_capital_gains_amount" +) + + +def _capital_gains_ht2_facts() -> list[dict[str, object]]: + """TY2022 Historic Table 2 line-7 rows: US 100 / 1,000 and CA 25 / 250.""" + return [ + _soi_capital_gains_fact( + 2022, + source_record_id=( + "irs_soi.ty2022.historic_table_2.us.all.net_capital_gains_returns" + ), + measure_id="net_capital_gains_returns", + value=100.0, + ), + _soi_capital_gains_fact( + 2022, + source_record_id=( + "irs_soi.ty2022.historic_table_2.us.all.net_capital_gains_amount" + ), + value=1_000.0, + ), + _soi_capital_gains_fact( + 2022, + source_record_id=_CG_HT2_CA_RETURNS, + measure_id="net_capital_gains_returns", + geography_level="state", + geography_id="0400000US06", + value=25.0, + ), + _soi_capital_gains_fact( + 2022, + source_record_id=_CG_HT2_CA_AMOUNT, + geography_level="state", + geography_id="0400000US06", + value=250.0, + ), + ] + + +def _capital_gains_cd_us_facts() -> list[dict[str, object]]: + """The congressional-district file's US row: TY2022 line-7 data stamped + ty2023, as on the pinned feed (PolicyEngine/chronicle#117).""" + return [ + _soi_capital_gains_fact( + 2023, + source_record_id=f"{_CG_CD_RECORD_SET}.us.net_capital_gains_returns", + measure_id="net_capital_gains_returns", + layout_record_set_id=_CG_CD_RECORD_SET, + value=98.0, + ), + _soi_capital_gains_fact( + 2023, + source_record_id=f"{_CG_CD_RECORD_SET}.us.net_capital_gains_amount", + layout_record_set_id=_CG_CD_RECORD_SET, + value=925.0, + ), + ] + + +def _capital_gains_t14_facts() -> list[dict[str, object]]: + """Table 1.4 ty2023 Schedule D taxable net gain: 40 returns / 800.""" + return [ + _soi_capital_gains_fact( + 2023, + source_record_id=_CG_T14_RETURNS, + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=40.0, + ), + _soi_capital_gains_fact( + 2023, + source_record_id=_CG_T14_AMOUNT, + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=800.0, + ), + ] + + +@pytest.mark.parametrize("cd_first", [True, False], ids=["cd-first", "t14-first"]) +def test_capital_gains_controls_are_table_1_4_in_either_feed_order( + cd_first: bool, +) -> None: + """A congressional-district US row that shares Table 1.4's period never + becomes the control, whichever comes first in the feed (microcosm#1035). + On the pinned feed it sorted first and set the HT2 state returns rows at + 2.39x the national Table 1.4 target of the same model quantity.""" + cd, t14 = _capital_gains_cd_us_facts(), _capital_gains_t14_facts() + facts = [ + *packaged_reference_facts(), + *_capital_gains_ht2_facts(), + *(cd + t14 if cd_first else t14 + cd), + ] + + controls = fiscal_targets._soi_capital_gains_active_totals( + tuple(facts), target_period=2024 + ) + assert {key[0]: control.source_record_id for key, control in controls.items()} == { + "net_capital_gains_returns": _CG_T14_RETURNS, + "net_capital_gains_amount": _CG_T14_AMOUNT, + } + + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + specs = {spec.metadata["ledger_source_record_id"]: spec for spec in registry.specs} + ca_returns = specs[_CG_HT2_CA_RETURNS] + ca_amount = specs[_CG_HT2_CA_AMOUNT] + assert ca_returns.value == 10.0 + assert ca_returns.metadata["uprating_factor"] == "0.4" + assert ca_returns.metadata["uprating_index_source_record_id"] == _CG_T14_RETURNS + assert ca_amount.value == 200.0 + assert ca_amount.metadata["uprating_index_source_record_id"] == _CG_T14_AMOUNT + for spec in (ca_returns, ca_amount): + # The bridge is declared: the row supplies a line-7 share, the level + # comes from the Schedule D concept capital_gains_gross measures. + assert spec.metadata["soi_source_concept"] == ( + "form_1040_line_7_net_gain_or_loss" + ) + assert spec.metadata["soi_control_concept"] == "schedule_d_taxable_net_gain" + + +def test_capital_gains_rows_without_a_concept_matched_control_are_dropped() -> None: + """With only a congressional-district row stamped later, no control + exists, and a line-7 HT2 row cannot stand as a capital_gains_gross level.""" + facts = [ + *packaged_reference_facts(), + *_capital_gains_ht2_facts(), + *_capital_gains_cd_us_facts(), + ] + + assert ( + fiscal_targets._soi_capital_gains_active_totals( + tuple(facts), target_period=2024 + ) + == {} + ) + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + source_record_ids = { + spec.metadata["ledger_source_record_id"] for spec in registry.specs + } + assert _CG_HT2_CA_RETURNS not in source_record_ids + assert _CG_HT2_CA_AMOUNT not in source_record_ids + + +def test_capital_gains_rows_whose_control_predates_them_are_dropped() -> None: + """A control older than the HT2 row would reverse its period, so the row + drops rather than ship its line-7 level.""" + facts = [ + *packaged_reference_facts(), + *_capital_gains_ht2_facts(), + _soi_capital_gains_fact( + 2021, + source_record_id="irs_soi.ty2021.table_1_4.all.net_capital_gains_returns", + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2021.table_1_4", + value=41.0, + ), + ] + + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + source_record_ids = { + spec.metadata["ledger_source_record_id"] for spec in registry.specs + } + assert _CG_HT2_CA_RETURNS not in source_record_ids + + +def test_capital_gains_control_refuses_two_matched_records_at_one_period() -> None: + duplicate = _soi_capital_gains_fact( + 2023, + source_record_id="irs_soi.ty2023.table_1_4.all_returns.net_capital_gains_returns", + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=41.0, + ) + facts = (*_capital_gains_t14_facts(), duplicate) + + for ordered in (facts, tuple(reversed(facts))): + with pytest.raises( + fiscal_targets.AmbiguousSoiCapitalGainsControlError, + match="all_returns.net_capital_gains_returns", + ): + fiscal_targets._soi_capital_gains_active_totals(ordered, target_period=2024) + # The same record twice is one candidate, not a tie. + same = (*_capital_gains_t14_facts(), *_capital_gains_t14_facts()) + assert ( + len(fiscal_targets._soi_capital_gains_active_totals(same, target_period=2024)) + == 2 + ) + + +@pytest.mark.parametrize( + ("record_set_id", "family"), + [ + ("irs_soi.ty2023.table_1_4", "table_1_4"), + ("irs_soi.table_1_4", "table_1_4"), + ("irs_soi.ty2022.historic_table_2.us", "historic_table_2"), + ("irs_soi.ty2022.historic_table_2.state_broad", "historic_table_2"), + (_CG_CD_RECORD_SET, "congressional_district"), + ( + "irs_soi.ty2024.congressional_district_2024.all_returns", + "congressional_district", + ), + ("irs_soi.congressional_district_2022.all_returns", "congressional_district"), + ("irs_soi.ty2023.table_4_3.all_returns_excluding_dependents", "table_4_3"), + ("irs_soi.ty2023", ""), + ("cbo.revenue_projection.ty2023", ""), + ("", ""), + ], +) +def test_soi_record_set_family_drops_period_and_data_vintage( + record_set_id: str, family: str +) -> None: + assert fiscal_targets._soi_record_set_family(record_set_id) == family + + +def test_capital_gains_control_is_order_free_latest_and_concept_matched() -> None: + """Property (microcosm#1035): for any set of national capital-gains facts + and any build period, the control per measure is the latest Table 1.4 + fact not after the build period, whatever the feed order, and no + Historic Table 2, congressional-district or unreviewed-family fact is ever + chosen. With no such Table 1.4 fact there is no control.""" + pytest.importorskip("hypothesis") + from hypothesis import given, settings + from hypothesis import strategies as st + + record_sets = { + "table_1_4": "irs_soi.ty{period}.table_1_4", + "historic_table_2": "irs_soi.ty{period}.historic_table_2.us", + "congressional_district": ( + "irs_soi.ty{period}.congressional_district_2022.all_returns" + ), + "unreviewed": "irs_soi.ty{period}.table_4_3", + } + measures = ("net_capital_gains_returns", "net_capital_gains_amount") + candidate = st.tuples( + st.sampled_from(sorted(record_sets)), + st.integers(min_value=2018, max_value=2027), + st.sampled_from(measures), + st.floats(min_value=1.0, max_value=1e12, allow_nan=False), + ) + + @settings(max_examples=300, deadline=None) + @given( + st.lists(candidate, max_size=14, unique_by=lambda item: item[:3]), + st.integers(min_value=2019, max_value=2026), + st.randoms(use_true_random=False), + ) + def check(candidates, target_period, rng) -> None: + facts = [ + _soi_capital_gains_fact( + period, + source_record_id=( + f"{record_sets[family].format(period=period)}.us.{measure}" + ), + measure_id=measure, + layout_record_set_id=record_sets[family].format(period=period), + value=value, + ) + for family, period, measure, value in candidates + ] + shuffled = list(facts) + rng.shuffle(shuffled) + + controls = fiscal_targets._soi_capital_gains_active_totals( + tuple(facts), target_period=target_period + ) + assert controls == fiscal_targets._soi_capital_gains_active_totals( + tuple(shuffled), target_period=target_period + ) + chosen = {key[0]: control for key, control in controls.items()} + for measure in measures: + eligible = [ + (period, value) + for family, period, candidate_measure, value in candidates + if family == "table_1_4" + and candidate_measure == measure + and period <= target_period + ] + if not eligible: + assert measure not in chosen + continue + period, value = max(eligible) + control = chosen[measure] + assert control.source_period == str(period) + assert control.value == value + assert control.source_record_id == ( + f"irs_soi.ty{period}.table_1_4.us.{measure}" + ) + + check() + + +def test_rebased_capital_gains_states_never_outgrow_their_control() -> None: + """Property: whatever the HT2 state rows and national total, rebased + state rows keep their HT2 shares and so sum to the control times the + states' share of the HT2 nation; when the HT2 states sum to at most the + HT2 nation, they sum to at most the control.""" + pytest.importorskip("hypothesis") + from hypothesis import given, settings + from hypothesis import strategies as st + + states = (("ca", "06"), ("ny", "36"), ("tx", "48")) + + @settings(max_examples=25, deadline=None) + @given( + st.lists( + st.integers(min_value=1, max_value=10_000_000), + min_size=len(states), + max_size=len(states), + ), + st.integers(min_value=0, max_value=5_000_000), + st.integers(min_value=1, max_value=50_000_000), + ) + def check(state_values, other_areas, control_value) -> None: + national = sum(state_values) + other_areas + facts = [ + *packaged_reference_facts(), + _soi_capital_gains_fact( + 2022, + source_record_id=( + "irs_soi.ty2022.historic_table_2.us.all.net_capital_gains_returns" + ), + measure_id="net_capital_gains_returns", + value=float(national), + ), + *( + _soi_capital_gains_fact( + 2022, + source_record_id=( + "irs_soi.ty2022.historic_table_2.state_broad." + f"{postal}.all.net_capital_gains_returns" + ), + measure_id="net_capital_gains_returns", + geography_level="state", + geography_id=f"0400000US{fips}", + value=float(value), + ) + for (postal, fips), value in zip(states, state_values, strict=True) + ), + _soi_capital_gains_fact( + 2023, + source_record_id=_CG_T14_RETURNS, + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=float(control_value), + ), + ] + registry = compile_us_fiscal_target_registry( + facts, allow_unaged_dollar_targets=True + ) + rebased = [ + spec + for spec in registry.specs + if ".historic_table_2.state_broad." in spec.name + and spec.metadata.get("source_measure_id") == "net_capital_gains_returns" + ] + assert len(rebased) == len(states) + total = sum(spec.value for spec in rebased) + assert math.isclose( + total, control_value * sum(state_values) / national, rel_tol=1e-12 + ) + assert total <= control_value * (1 + 1e-12) + + check() + + def test_cross_period_soi_taxable_interest_agi_slice_without_total_is_dropped() -> None: source_record_id = ( "irs_soi.ty2022.historic_table_2.us.200k_to_500k.taxable_interest_amount" From 6ebfcf7f4f90bc30fbea63cddb40eff67ed97f2d Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 27 Sep 2026 05:22:06 -0400 Subject: [PATCH 2/2] Refuse any ambiguous capital-gains control; qualify two claims (#1036 review) The chooser now treats one record id carrying two values or period labels at the winning period as ambiguous too, so no feed order can pick the value. Tests cover that, an equivalent period label, and a tie at a superseded period. The model-concept comment scopes "no loss returns" to the Build P release data, and the doc's AGI-mix median is the median absolute change. Co-Authored-By: Claude Opus 5.5 --- docs/us-soi-capital-gains-concepts.md | 2 +- .../build/us_runtime/fiscal_targets.py | 39 +++++++------- .../tests/test_us_fiscal_targets.py | 51 +++++++++++++++++++ 3 files changed, 74 insertions(+), 18 deletions(-) diff --git a/docs/us-soi-capital-gains-concepts.md b/docs/us-soi-capital-gains-concepts.md index 841d6fdf8..4475329c5 100644 --- a/docs/us-soi-capital-gains-concepts.md +++ b/docs/us-soi-capital-gains-concepts.md @@ -75,7 +75,7 @@ share of Schedule D gain returns. No IRS state product publishes a gain-only count, so this cannot be checked directly. AGI composition is one measurable source of difference: weighting each state's HT2 N01000 by AGI class with Table 1.4's gain share per class would move TY2022 state targets by −2.9% (WV) -to +4.3% (DC), with a median of 1.6% and no state beyond 5%. +to +4.3% (DC), with a median absolute change of 1.6% and no state beyond 5%. ## How the returns control drifted diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py index 4b5334f83..acd4751dd 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/fiscal_targets.py @@ -1502,8 +1502,10 @@ class _SoiTotalControl: #: PE-US ``capital_gains`` (short-term plus long-term, i.e. Schedule D, before #: the loss limit); distributions reported without Schedule D are the separate #: ``non_sch_d_capital_gains``. Its indicator therefore counts Table 1.4 col 37 -#: returns and its sum is col 38. The model carries no Schedule D loss returns, -#: so a line-7 count has no model counterpart at any level. +#: returns and its sum is col 38; a line-7 count, which also counts losses and +#: distributions-only returns, is a different population. (PE-US allows +#: negative gains, but the Build P release data carries no tax unit with a +#: net loss, so it could not reach a line-7 count at any level.) _SOI_CAPITAL_GAINS_MODEL_CONCEPT = _SOI_SCHEDULE_D_TAXABLE_NET_GAIN @@ -1590,13 +1592,14 @@ def _soi_capital_gains_active_totals( ``capital_gains_gross`` measures qualifies, so HT2 and congressional-district rows never do, whatever their period stamp. The latest qualifying period not after the build period wins, and the choice - depends on the fact set alone: two different records at the winning - period refuse the compile. Feed order used to break that tie, and the + depends on the fact set alone: two different candidates at the winning + period (different records, or one record with different values or period + labels) refuse the compile. Feed order used to break that tie, and the September re-pin, which re-sorted the feed by content hash, handed the returns control to the congressional-district US row (microcosm#1035). """ controls: dict[tuple[str, str, str], _SoiTotalControl] = {} - tied_record_ids: dict[tuple[str, str, str], set[str]] = {} + tied: dict[tuple[str, str, str], set[tuple[str, float, str]]] = {} target_period_key = _period_key_from_value(target_period) for fact in facts: key = _soi_capital_gains_control_key_from_fact(fact) @@ -1610,26 +1613,28 @@ def _soi_capital_gains_active_totals( source_record_id = _source_record_id(fact) if not source_record_id: continue + candidate = _SoiTotalControl( + value=_numeric_value(fact), + source_period=str(_period_value(fact)), + source_record_id=source_record_id, + period_key=period_key, + ) + identity = (source_record_id, candidate.value, candidate.source_period) current = controls.get(key) if current is not None and period_key[:2] == current.period_key[:2]: - tied_record_ids[key].add(source_record_id) + tied[key].add(identity) continue if current is None or period_key[:2] > current.period_key[:2]: - controls[key] = _SoiTotalControl( - value=_numeric_value(fact), - source_period=str(_period_value(fact)), - source_record_id=source_record_id, - period_key=period_key, - ) - tied_record_ids[key] = {source_record_id} + controls[key] = candidate + tied[key] = {identity} ambiguous = { - key: sorted(record_ids) - for key, record_ids in tied_record_ids.items() - if len(record_ids) > 1 + key: sorted(candidates) + for key, candidates in tied.items() + if len(candidates) > 1 } if ambiguous: raise AmbiguousSoiCapitalGainsControlError( - "SOI capital-gains control is ambiguous: different records tie at " + "SOI capital-gains control is ambiguous: different candidates tie at " f"the latest period for {ambiguous}. Deduplicate them upstream; " "feed order must not pick a rebase control (microcosm#1035)." ) diff --git a/packages/microcosm-build/tests/test_us_fiscal_targets.py b/packages/microcosm-build/tests/test_us_fiscal_targets.py index 97b1d992a..d6508f484 100644 --- a/packages/microcosm-build/tests/test_us_fiscal_targets.py +++ b/packages/microcosm-build/tests/test_us_fiscal_targets.py @@ -4804,6 +4804,57 @@ def test_capital_gains_control_refuses_two_matched_records_at_one_period() -> No ) +def _t14_returns_fact(period: object, *, record_id: str, value: float) -> dict: + fact = _soi_capital_gains_fact( + 2023, + source_record_id=record_id, + measure_id="net_capital_gains_returns", + layout_record_set_id="irs_soi.ty2023.table_1_4", + value=value, + ) + fact["period"] = {**fact["period"], "value": period} + return fact + + +@pytest.mark.parametrize( + "rival", + [ + # One record id carrying two values: order would pick the value. + {"period": 2023, "record_id": _CG_T14_RETURNS, "value": 41.0}, + # Another record at the same period under an equivalent label. + { + "period": "tax_year_2023", + "record_id": "irs_soi.ty2023.table_1_4.v2.net_capital_gains_returns", + "value": 40.0, + }, + ], + ids=["same-id-other-value", "equivalent-period-label"], +) +def test_capital_gains_control_refuses_any_ambiguous_candidate(rival) -> None: + control = _t14_returns_fact(2023, record_id=_CG_T14_RETURNS, value=40.0) + other = _t14_returns_fact( + rival["period"], record_id=rival["record_id"], value=rival["value"] + ) + for ordered in ((control, other), (other, control)): + with pytest.raises(fiscal_targets.AmbiguousSoiCapitalGainsControlError): + fiscal_targets._soi_capital_gains_active_totals(ordered, target_period=2024) + + +def test_capital_gains_tie_at_a_superseded_period_is_not_ambiguous() -> None: + """Only a tie at the winning period matters: a later control supersedes + an older tie, in any order.""" + older = ( + _t14_returns_fact(2022, record_id="irs_soi.ty2022.table_1_4.a", value=1.0), + _t14_returns_fact(2022, record_id="irs_soi.ty2022.table_1_4.b", value=2.0), + ) + latest = _t14_returns_fact(2023, record_id=_CG_T14_RETURNS, value=40.0) + for ordered in ((*older, latest), (latest, *older), (older[0], latest, older[1])): + (control,) = fiscal_targets._soi_capital_gains_active_totals( + ordered, target_period=2024 + ).values() + assert (control.source_record_id, control.value) == (_CG_T14_RETURNS, 40.0) + + @pytest.mark.parametrize( ("record_set_id", "family"), [