Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions docs/calibration-l2-basis.md
Original file line number Diff line number Diff line change
Expand Up @@ -177,6 +177,15 @@ recalibrated from its own checkpoint (`experiments/us-acs-local-l2-basis-2026092
`l2_lambda`, so the gain is optimistic.
- **What limits ESS is the starting weights.** The experiment's README has the
frontier, the holdout, candidate ESS floors and the recommendation.
- **On the weighted loss.** With the national release's target weighting
(microcosm#1104), `l2_lambda` is in units of a different loss, so the
README's "On the weighted loss" section re-picks it. Held-out weighted error
is lowest at projection `l2_lambda = 0.1` on the default weights. But the
weighting gives each state's district populations one shared budget, and at
0.1 it leaves 135 of 436 trained district populations more than 10% off.
`l2_lambda = 0.03` with `--target-family-loss-multiplier
census_population=8` misfits 3 and has slightly lower held-out error (by
0.6%). The multiplier was chosen after the first pass of runs.

## Using it

Expand Down
287 changes: 283 additions & 4 deletions experiments/us-acs-local-l2-basis-20260928/README.md

Large diffs are not rendered by default.

639 changes: 634 additions & 5 deletions experiments/us-acs-local-l2-basis-20260928/analyze.py

Large diffs are not rendered by default.

147 changes: 147 additions & 0 deletions experiments/us-acs-local-l2-basis-20260928/check_weights.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,147 @@
"""Check that the current shared weighting reproduces every weighted run's weights.

The ``w_`` runs used microcosm#1104's ``target_loss_weights.py`` before it
merged. They stay valid for main as long as main's module gives each run
the same weights. The module's bytes can change without that: review fixes
that touch only other surfaces leave these weights alone. So this checks the
weights, not the file.

For every ``results/runs/w_*.json``, it rebuilds the run's training weights
the way ``sweep.py`` did (the run's training rows and family multipliers) and
its full-surface yardstick, with whichever module ``sweep.py`` loads now:
the package import, or the file named by ``MICROCOSM_TARGET_LOSS_WEIGHTS_FILE``.
It compares both loss-vector digests with the run's receipt. It also checks
each receipt's own consistency block: the solver's epoch-0 loss within 1e-5
of the weighted loss of the starting weights recomputed outside it, and its
final loss equal to the recomputed weighted loss of its returned weights.
The launch gate checked those only on the first pass's first run. It writes
``results/weights_check.json`` and exits non-zero on any failure.

Run: ``uv run python experiments/us-acs-local-l2-basis-20260928/check_weights.py``
"""

from __future__ import annotations

import json
import math
import os
import sys
from pathlib import Path

HERE = Path(__file__).resolve().parent
sys.path.insert(0, str(HERE))

from sweep import ( # noqa: E402
DEFAULT_CHECKPOINT,
SHARED_WEIGHTS_FILE_ENV,
RunSpec,
load_inputs,
loss_vector_sha256,
loss_weights_for,
row_names,
sha256,
shared_weight_module,
training_rows,
)

RUNS = HERE / "results" / "runs"
OUT = HERE / "results" / "weights_check.json"
#: float32 solver against float64 recomputation; observed at most 9.3e-6.
EPOCH0_RTOL = 1e-5


def _finite(value) -> bool:
# bool is an int subclass, and a JSON false must not pass as 0.0.
return (
isinstance(value, (int, float))
and not isinstance(value, bool)
and math.isfinite(value)
)


def main() -> int:
inputs = load_inputs(DEFAULT_CHECKPOINT, verify=True, need_registry=True)
module = shared_weight_module()
all_rows = list(range(inputs.matrix.shape[0]))
full_names = row_names(inputs, all_rows)
cache: dict[tuple, tuple[str, str]] = {}
checked, mismatches = [], []
for path in sorted(RUNS.glob("w_*.json")):
payload = json.loads(path.read_text())
spec = RunSpec.from_mapping(payload["spec"])
recorded = payload["target_loss_weights"]
key = (spec.holdout_fold, spec.family_loss_multipliers)
if key not in cache:
train = training_rows(inputs, spec.holdout_fold)
weights = loss_weights_for(spec, inputs, train)
cache[key] = (
loss_vector_sha256(row_names(inputs, train), weights.train),
loss_vector_sha256(full_names, weights.full),
)
train_digest, full_digest = cache[key]
consistency = payload.get("consistency") or {}
epoch0 = consistency.get("recomputed_design_loss_relative_to_epoch0")
final = consistency.get("recomputed_minus_result_final_loss")
row = {
"epoch0_relative": epoch0,
"final_difference": final,
"consistent": _finite(epoch0)
and abs(epoch0) < EPOCH0_RTOL
and _finite(final)
and final == 0.0,
"run_id": spec.run_id,
"train_matches": train_digest == recorded["train"]["loss_vector_sha256"],
"full_matches": full_digest
== recorded["full_surface"]["loss_vector_sha256"],
"recorded_module_sha256": recorded["module_sha256"],
}
checked.append(row)
if not (row["train_matches"] and row["full_matches"] and row["consistent"]):
mismatches.append(
{
**row,
"train_digest_now": train_digest,
"train_digest_recorded": recorded["train"]["loss_vector_sha256"],
}
)
result = {
"ok": not mismatches and bool(checked),
"module_file": module.__file__,
"module_sha256": sha256(Path(module.__file__)),
"module_loaded_from_file": bool(os.environ.get(SHARED_WEIGHTS_FILE_ENV))
and Path(module.__file__).resolve()
== Path(os.environ[SHARED_WEIGHTS_FILE_ENV]).resolve(),
"recorded_module_sha256": sorted(
{row["recorded_module_sha256"] for row in checked}
),
"registry_sha256": inputs.registry_receipt["verified_registry_sha256"],
"n_runs": len(checked),
"n_distinct_weightings": len(cache),
# Over the receipts that record finite values; the others are in
# mismatches already.
"max_abs_epoch0_relative": max(
(
abs(r["epoch0_relative"])
for r in checked
if _finite(r["epoch0_relative"])
),
default=None,
),
"max_abs_final_difference": max(
(
abs(r["final_difference"])
for r in checked
if _finite(r["final_difference"])
),
default=None,
),
"n_mismatches": len(mismatches),
"mismatches": mismatches,
}
OUT.write_text(json.dumps(result, indent=1) + "\n")
print(json.dumps({k: v for k, v in result.items() if k != "mismatches"}, indent=1))
return 0 if result["ok"] else 1


if __name__ == "__main__":
raise SystemExit(main())
153 changes: 153 additions & 0 deletions experiments/us-acs-local-l2-basis-20260928/cross_score.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,153 @@
"""Score the equal-weight solves on the weighted loss.

The weighted grid's solves record both losses; the equal-weight grid's record
only the unweighted one. This scores every equal-weight run's final weights
(``weights.npz`` beside the checkpoint), the published release's weights and
the design weights on the shared module's target-loss weights, the same way
``sweep.py`` scores a weighted run:

* training fit over the run's training rows, weighted as a build calibrating
on those rows would weight them (the full surface, or one fold's
complement);
* held-out fit over the run's held-out fold, with each target's full-surface
weight.

It writes ``results/cross_scores.json``: per run, the weighted capped error
and weighted share within 10% on training and held-out targets, the run's
receipt sha256 and the weights' sha256, plus the full-surface weight vector's
content hash, so analyze.py can set the two grids side by side on one loss.

Run: ``uv run python experiments/us-acs-local-l2-basis-20260928/cross_score.py``
"""

from __future__ import annotations

import hashlib
import json
import sys
from pathlib import Path

import numpy as np

HERE = Path(__file__).resolve().parent
sys.path.insert(0, str(HERE))

from gradient_scale import RELEASE_WEIGHTS # noqa: E402
from sweep import ( # noqa: E402
DEFAULT_CHECKPOINT,
DEFAULT_OUT_ROOT,
SHARED_TARGET_LOSS_WEIGHTS,
RunSpec,
estimates_for,
fit_block,
load_inputs,
loss_vector_sha256,
loss_weights_for,
row_names,
sha256,
training_rows,
)

RUNS = HERE / "results" / "runs"
OUT = HERE / "results" / "cross_scores.json"


def _measures(block: dict) -> dict:
overall = block["overall"]
return {
"weighted_mean_capped_scaled_error": overall[
"weighted_mean_capped_scaled_error"
],
"weighted_fraction_within_10pct": overall["weighted_fraction_within_10pct"],
"mean_capped_scaled_error": overall["mean_capped_scaled_error"],
"fraction_within_10pct": overall["fraction_within_10pct"],
}


def main() -> None:
inputs = load_inputs(DEFAULT_CHECKPOINT, verify=True, need_registry=True)
all_rows = np.arange(inputs.matrix.shape[0], dtype=np.int64)
full = loss_weights_for(
RunSpec(run_id="full", target_weighting="shared"), inputs, all_rows
)
fold_weights = {
fold: loss_weights_for(
RunSpec(run_id=f"fold{fold}", target_weighting="shared"),
inputs,
training_rows(inputs, fold),
)
for fold in inputs.folds
}

def score(weights: np.ndarray, fold: int | None) -> dict:
estimates = estimates_for(inputs, weights)
if fold is None:
return {
"train": _measures(
fit_block(estimates, inputs.meta, all_rows, loss_weights=full.train)
)
}
train = training_rows(inputs, fold)
held = inputs.folds[fold]
return {
"train": _measures(
fit_block(
estimates,
inputs.meta,
train,
loss_weights=fold_weights[fold].train,
)
),
"holdout": _measures(
fit_block(estimates, inputs.meta, held, loss_weights=full.full[held])
),
}

runs: dict[str, dict] = {}
for receipt in sorted(RUNS.glob("*.json")):
payload = json.loads(receipt.read_text())
spec = payload.get("spec") or {}
if payload.get("kind") != "calibrate":
continue
if spec.get("target_weighting", "equal") != "equal":
continue
path = DEFAULT_OUT_ROOT / spec["run_id"] / "weights.npz"
if not path.exists():
runs[spec["run_id"]] = {"status": "weights not found", "path": str(path)}
continue
weights = np.asarray(np.load(path)["weights"], dtype=np.float64)
fold = spec.get("holdout_fold")
runs[spec["run_id"]] = {
"holdout_fold": fold,
"receipt_sha256": sha256(receipt),
"weights_sha256": sha256(path),
**score(weights, fold),
}
published = np.asarray(np.load(RELEASE_WEIGHTS)["weights"], dtype=np.float64)
references = {
"published_release": (published, str(RELEASE_WEIGHTS), sha256(RELEASE_WEIGHTS)),
"design": (inputs.design, "checkpoint households.parquet design_weight", None),
}
for label, (weights, source, digest) in references.items():
runs[label] = {
"source": source,
"weights_sha256": digest,
"full_surface": score(weights, None),
**{f"fold_{fold}": score(weights, fold) for fold in (0, 1)},
}
payload = {
"function": SHARED_TARGET_LOSS_WEIGHTS,
"registry_sha256": inputs.registry_receipt["verified_registry_sha256"],
"full_surface_loss_vector_sha256": loss_vector_sha256(
row_names(inputs, all_rows), full.full
),
"runs": runs,
}
OUT.write_text(json.dumps(payload, indent=1) + "\n")
print(f"wrote {OUT}: {len(runs)} entries")
digest = hashlib.sha256(OUT.read_bytes()).hexdigest()
print(f"sha256 {digest}")


if __name__ == "__main__":
main()
Loading
Loading