Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
14 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/us-acs-local-state-cd-sparse.added.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Calibrate the US ACS local-area release (`tools/build_us_acs_local_release.py`) on a sparse target matrix and add the `--soi-mode state_cd` surface. The materialize stage now writes a (targets x households) float32 CSR matrix (`target_matrix.npz`, row-aligned with `target_registry.json`) and a row-aligned `target_roles.json` beside a structure-only lean H5; no dense households x targets matrix exists at any point, and calibration builds each training target from its registry spec with a CSR-row measure, so the unchanged calibrate kernel compiles it. District SOI rows are materialized as one geography-free carrier per concept restricted to each district's households, which equals the direct materialization (one row per carrier checked on the first chunk; every chunk rebuilds each `state_cd` state parent from its stored district rows, and in other modes checks each district row that has a same-concept state row against it). `state_cd` binds the TY2022 SOI congressional-district file's 21,743 district rows (427 districts; at-large states, the district file's mislabeled SALT and PTC columns, and four (state, concept) blocks outside a 1.25x factor band excluded) on top of the `state` surface, keeping one vintage per state concept: Historic Table 2 sets each state level, district rows keep only their within-state shares, and the six district-file-only measures are lifted onto the Historic Table 2 basis by a sibling concept's two state levels; a rebase or bridge ratio more than 1.25x from its measure's median across states is dropped and recorded, and a sign conflict or a district file whose state row disagrees with its own districts is refused. A hash-assigned 10% of (state x concept family) district blocks is held out of calibration and scored against a population pro-rata baseline; `calibration_summary.json` also records Kish ESS nationally, per state, per district and over distinct households, the top-1% weight share, and household-weight share by spine, and it, `weights_latest.npz`, `consumer_export.json` and `spine_qa.json` carry the run-identity digest that resume, finalize and package require, with the solver settings and a calibrated-weights digest binding the H5 to the summary that describes it; the run identity also binds the lean H5. `--sample-fraction` draws f001/f004/f010/f025 development rungs, which package refuses. `microcosm.build.holdout` gains `hash_holdout_unit`. The ladder population targets' district hierarchy ids now carry the 119th-plan prefix (`5001900US`) their ladder uses.
430 changes: 411 additions & 19 deletions docs/us-acs-local-soi-target-surface.md

Large diffs are not rendered by default.

135 changes: 135 additions & 0 deletions experiments/us-acs-local-state-cd-20260927/concentration_report.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,135 @@
"""Weight concentration of the evaluation's weight vectors, arm by arm.

Kish ESS nationally, per state and per congressional district, ESS over
distinct households, and the top-1% weight share, for each named weight
vector over one lean checkpoint's households. Uses the tool's own
``weight_origin_summary``.

uv run python experiments/us-acs-local-state-cd-20260927/concentration_report.py \\
--lean <checkpoint>/target_frame_lean.h5 \\
--arm design=<weights.npz>:design --arm state=<weights.npz>:weights ... \\
--out concentration.json
"""

from __future__ import annotations

import argparse
import importlib.util
import json
from pathlib import Path

import numpy as np
import pandas as pd

REPO = Path(__file__).resolve().parents[2]


def load_tool():
path = REPO / "tools" / "build_us_acs_local_release.py"
spec = importlib.util.spec_from_file_location("build_us_acs_local_release", path)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module


def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--lean", type=Path, required=True)
parser.add_argument(
"--arm",
action="append",
required=True,
help="label=path.npz:key (weights aligned to the lean household table)",
)
parser.add_argument("--focus-state", default="25")
parser.add_argument("--out", type=Path, required=True)
args = parser.parse_args()
tool = load_tool()
with pd.HDFStore(args.lean, mode="r") as store:
households = (
store.select(
"household",
columns=[
"household_id",
"state_fips",
"congressional_district_geoid",
"household_spine",
"household_source_id",
],
)
if store.get_storer("household").is_table
else store["household"]
)
origin = {
"spine": households["household_spine"].to_numpy(),
"source_id": households["household_source_id"].to_numpy(),
"state": pd.to_numeric(households["state_fips"])
.astype(int)
.map("{:02d}".format)
.to_numpy(),
"district": pd.to_numeric(households["congressional_district_geoid"])
.astype(int)
.map("{:04d}".format)
.to_numpy(),
}
report: dict[str, dict] = {}
for arm in args.arm:
label, rest = arm.split("=", 1)
path, key = rest.rsplit(":", 1)
weights = np.load(path)[key]
if len(weights) != len(households):
raise SystemExit(f"{label}: {len(weights)} weights, {len(households)} rows")
summary = tool.cd_surface.weight_origin_summary(weights, **origin)
focus_districts = {
code: value
for code, value in summary["effective_sample_size_by_district"].items()
if code.startswith(args.focus_state)
}
report[label] = {
"national_ess_rows": summary["effective_sample_size_rows"],
"national_ess_distinct_households": summary.get(
"effective_sample_size_distinct_households"
),
"top_1pct_weight_share": summary["top_1pct_weight_share"],
"weight_share_by_spine": summary.get("weight_share_by_spine"),
"state_ess": summary["effective_sample_size_by_state_distribution"],
"district_ess": summary["effective_sample_size_by_district_distribution"],
f"state_{args.focus_state}_ess": summary[
"effective_sample_size_by_state"
].get(args.focus_state),
f"state_{args.focus_state}_district_ess": focus_districts,
"by_state": summary["effective_sample_size_by_state"],
"by_district": summary["effective_sample_size_by_district"],
}
labels = list(report)
if len(labels) >= 2:
base, *others = labels
for other in others:
for level in ("by_state", "by_district"):
a, b = report[base][level], report[other][level]
ratios = np.asarray([b[k] / a[k] for k in a if a[k] > 0 and k in b])
report[other][f"{level}_ess_ratio_to_{base}"] = {
"min": float(ratios.min()),
"median": float(np.median(ratios)),
"max": float(ratios.max()),
"share_lower": float((ratios < 1).mean()),
}
args.out.write_text(json.dumps(report, indent=1))
for label, entry in report.items():
print(
label,
json.dumps(
{
key: value
for key, value in entry.items()
if key not in ("by_state", "by_district", "weight_share_by_spine")
},
indent=None,
default=str,
)[:1500],
)
return 0


if __name__ == "__main__":
raise SystemExit(main())
Loading
Loading