diff --git a/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_9b2c5dd94d3d.py b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_9b2c5dd94d3d.py new file mode 100644 index 00000000..299ca88a --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_9b2c5dd94d3d.py @@ -0,0 +1,1880 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddQuantileFill` (odd years, primary): a two-part conditional + draw. The probability of a zero year and the conditional quantiles of a + positive share, relative to the neighbours' level, by sex, age, the + shares at ``t-1`` and ``t+1`` and the context at ``t-3``, ``t+3`` and + further; a Gaussian AR(1) copula correlates a person's draws across + masked years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "OddQuantileFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _invert(table: np.ndarray, rows: np.ndarray, value: np.ndarray): + """The level at which each row's quantile function reaches ``value``. + + Where the function is flat at ``value`` (a run of equal quantiles), the + middle of the run's levels. + """ + + points = table.shape[1] + levels = np.linspace(0.0, 1.0, points) + out = np.empty(len(rows)) + for start in range(0, len(rows), 200_000): + block = slice(start, start + 200_000) + curve = table[rows[block]].astype(np.float64) + target = np.asarray(value[block], dtype=np.float64)[:, None] + below = (curve < target).sum(axis=1) + above = (curve <= target).sum(axis=1) + flat = below < above + result = np.empty(len(curve)) + result[above == 0] = 0.0 + result[below >= points] = 1.0 + middle = flat & (above > 0) & (below < points) + result[middle] = 0.5 * ( + levels[below[middle]] + levels[above[middle] - 1] + ) + between = ~flat & (below > 0) & (below < points) + index = np.flatnonzero(between) + left = curve[index, below[index] - 1] + right = curve[index, below[index]] + share = np.where( + right > left, + (target[index, 0] - left) + / np.where(right > left, right - left, 1), + 0.5, + ) + result[index] = levels[below[index] - 1] + share * ( + levels[below[index]] - levels[below[index] - 1] + ) + out[block] = result + return out + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + buffer = io.BytesIO() + np.savez_compressed(buffer, **arrays) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: the two-part conditional draw +# -------------------------------------------------------------------------- +_N_SHARE_BINS = 20 + + +def _share_bin(value: np.ndarray, edges: np.ndarray) -> np.ndarray: + """0 zero, 1..20 quantile bins of a positive share below the cap, 21 cap.""" + + bins = np.searchsorted(edges, value, side="right") + 1 + bins = np.where(value <= 0, 0, bins) + return np.where(value >= 1.0, _N_SHARE_BINS + 1, bins) + + +def _coarse_age(band: np.ndarray) -> np.ndarray: + """Age bands grouped: under 30, 30-44, 45-59, 60 and over.""" + + return np.digitize(band, [4, 7, 10]) + + +@dataclass(frozen=True) +class OddQuantileFill: + """Two-part conditional draw for masked odd years, with an AR(1) copula. + + A unit's reference level ``m`` is the geometric mean of its positive + neighbours' shares (the one positive neighbour's share if only one is, + 1 if neither is). Its cell is the finest of seven nested keys with at + least ``MIN_CELL`` TRAIN units, built from sex, five-year age band, the + bins of the shares at ``t-1`` and ``t+1`` (zero, 20 quantile bins of a + positive share below the cap, at the cap), the context at ``t-3`` and + ``t+3`` (missing, zero, below or above the median positive share), and + whether any share at offsets 5, 7 or 9 is positive. In the cell: ``p0`` + the share of zero years, and 65 quantiles of ``log(x_t / m)`` among + positive years. A uniform ``u`` maps to zero if ``u < p0``, else to + ``min(m * exp(Q((u - p0) / (1 - p0))), 1)``. The uniforms of a person's + consecutive masked years (two years apart) are joined by a Gaussian + AR(1) copula with correlation ``rho`` by sex and age band, learned on + TRAIN from the probability integral transforms of consecutive units. + """ + + share_edges: np.ndarray + context_median: float + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + rho: np.ndarray + stream: str = "epuf_fill.odd_quantile.v1" + name: str = "odd_quantile" + + # -- keys --------------------------------------------------------------- + @staticmethod + def _parts(context: OddContext, share_edges, median): + left = np.nan_to_num(context.left, nan=-1.0) + right = np.nan_to_num(context.right, nan=-1.0) + # A missing neighbour takes the other's value (the PSID fallback). + left = np.where(left < 0, right, left) + right = np.where(right < 0, left, right) + positive_left = np.where(left > 0, left, 1.0) + positive_right = np.where(right > 0, right, 1.0) + level = np.where( + (left > 0) & (right > 0), + np.sqrt(positive_left * positive_right), + np.where(left > 0, positive_left, positive_right), + ) + + def context_code(value): + return np.where( + np.isnan(value), + 0, + np.where(value <= 0, 1, np.where(value < median, 2, 3)), + ) + + return { + "sex": context.sex, + "age": _age_band(context.age), + "left_bin": _share_bin(left, share_edges), + "right_bin": _share_bin(right, share_edges), + "context3": 4 * context_code(context.left3) + + context_code(context.right3), + "wide": context.wide, + "level": level, + "valid": ~(np.isnan(context.left) & np.isnan(context.right)), + } + + @staticmethod + def _keys(parts) -> list[np.ndarray]: + sex = parts["sex"] + age = parts["age"] + coarse = _coarse_age(age) + left = parts["left_bin"] + right = parts["right_bin"] + context3 = parts["context3"] + wide = parts["wide"] + + # Nested keys from finest to coarsest; a dropped component is held + # at a sentinel (age 16-20 marks the coarse bands, 21 none). + def key(s, a, lb, rb, c3, w): + return ((((s * 22 + a) * 23 + lb) * 23 + rb) * 17 + c3) * 3 + w + + return [ + key(sex, age, left, right, context3, wide), + key(sex, age, left, right, context3, 2), + key(sex, age, left, right, 16, 2), + key(sex, 16 + coarse, left, right, 16, 2), + key(sex, 21, left, right, 16, 2), + key(0 * sex, 21, left, right, 16, 2), + key(0 * sex, 21, np.minimum(left, 1), np.minimum(right, 1), 16, 2), + ] + + def _cells(self, parts) -> tuple[np.ndarray, np.ndarray]: + """(level, row) of each unit's finest populated cell.""" + + keys = self._keys(parts) + level = np.full(len(keys[0]), -1, dtype=np.int64) + row = np.full(len(keys[0]), -1, dtype=np.int64) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + if (level < 0).any(): + raise ValueError("a unit has no populated cell at any level") + return level, row + + def _quantile(self, parts, u: np.ndarray) -> np.ndarray: + level, row = self._cells(parts) + out = np.zeros(len(u)) + for index in np.unique(level): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate(self.level_quantiles[index], row[take], v) + share = np.minimum(parts["level"][take] * np.exp(residual), 1.0) + out[take] = np.where(positive, share, 0.0) + return out + + # -- fitting -------------------------------------------------------------- + @classmethod + def fit( + cls, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + unit_years: tuple[int, ...], + rho_seed: int = 0, + ) -> tuple[OddQuantileFill, dict[str, object]]: + """Fit on complete TRAIN shares; every year of ``unit_years`` a unit. + + Every person-year of ``unit_years`` whose two neighbours are inside + the matrix is a training unit (all years are recorded on TRAIN). + """ + + shares = np.asarray(shares, dtype=np.float64) + known = np.isfinite(shares) + n = len(shares) + rows_list, years_list = [], [] + for year in unit_years: + rows_list.append(np.arange(n)) + years_list.append(np.full(n, year)) + rows = np.concatenate(rows_list) + unit_year = np.concatenate(years_list) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + target = _take(shares, rows, _column_of(years, unit_year)) + neighbours = np.concatenate([context.left, context.right]) + inside = neighbours[(neighbours > 0) & (neighbours < 1.0)] + share_edges = np.quantile( + inside, np.linspace(0, 1, _N_SHARE_BINS + 1)[1:-1] + ) + median = float(np.median(inside)) + parts = cls._parts(context, share_edges, median) + keys = cls._keys(parts) + positive = target > 0 + residual = np.log(np.where(positive, target, 1.0)) - np.log( + parts["level"] + ) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zeros_unique, zeros = np.unique( + key[in_cells & ~positive], return_counts=True + ) + totals = count[count >= MIN_CELL] + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zeros_unique)] = zeros + p0 = p0 / totals + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + # A populated cell with no positive unit draws only zeros. + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0.astype(np.float64)) + level_quantiles.append(full) + provisional = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=np.zeros((4, 16)), + ) + rho, rho_diagnostics = provisional._fit_rho( + parts, target, rows, unit_year, rho_seed + ) + fill = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=rho, + ) + cells = [len(k) for k in level_keys] + return fill, { + "n_units": int(len(target)), + "cells_per_level": cells, + **rho_diagnostics, + } + + def _pit(self, parts, target, rows, unit_year, seed) -> np.ndarray: + """Randomised probability integral transforms of true shares.""" + + level, row = self._cells(parts) + jitter = hash_uniform( + "epuf_fill.odd_quantile.pit", seed, rows, unit_year + ) + out = np.empty(len(target)) + for index in np.unique(level): + take = np.flatnonzero(level == index) + p0 = self.level_p0[index][row[take]] + zero = target[take] <= 0 + out[take[zero]] = jitter[take[zero]] * p0[zero] + positive = take[~zero] + residual = np.log(target[positive]) - np.log( + parts["level"][positive] + ) + # At the cap the residual is censored: spread it over the mass + # the quantile function puts at or above the cap. + at_cap = target[positive] >= 1.0 + v = _invert(self.level_quantiles[index], row[positive], residual) + cap_v = v.copy() + cap_v[at_cap] = v[at_cap] + jitter[positive][at_cap] * ( + 1.0 - v[at_cap] + ) + p0_positive = p0[~zero] + out[positive] = p0_positive + (1.0 - p0_positive) * cap_v + return np.clip(out, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, parts, target, rows, unit_year, seed): + """AR(1) correlation of consecutive units' normal scores (t, t+2). + + On a 5 percent sample of persons (by seed): every unit's + probability integral transform under the fitted cells, its normal + score, and the correlation of the scores of ``t`` and ``t+2`` for + the same person, by sex and age band at ``t``. + """ + + persons = np.unique(rows) + rng = np.random.default_rng(seed) + chosen = persons[rng.random(len(persons)) < 0.05] + index = np.flatnonzero(np.isin(rows, chosen)) + sub = {k: v[index] for k, v in parts.items()} + z = ndtri( + self._pit(sub, target[index], rows[index], unit_year[index], seed) + ) + first_year = int(unit_year.min()) + n_years = int(unit_year.max()) - first_year + 1 + position = np.searchsorted(chosen, rows[index]) + grid = np.full((len(chosen), n_years), np.nan) + grid[position, unit_year[index] - first_year] = z + sex_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + sex_grid[position, unit_year[index] - first_year] = sub["sex"] + age_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + age_grid[position, unit_year[index] - first_year] = sub["age"] + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex = sex_grid[:, :-2].ravel() + age = age_grid[:, :-2].ravel() + both = np.isfinite(now) & np.isfinite(later) + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = both & (sex == s) & (age == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + overall = float(np.corrcoef(now[both], later[both])[0, 1]) + four_now = grid[:, :-4].ravel() + four_later = grid[:, 4:].ravel() + four = np.isfinite(four_now) & np.isfinite(four_later) + lag4 = float(np.corrcoef(four_now[four], four_later[four])[0, 1]) + return rho, { + "rho_persons": int(len(chosen)), + "rho_pairs": int(both.sum()), + "rho_overall": overall, + "lag4_normal_score_correlation": lag4, + "lag4_ar1_prediction": overall**2, + } + + # -- filling -------------------------------------------------------------- + def fill( + self, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + person_key: np.ndarray, + fill_mask: np.ndarray, + seed: int, + ) -> np.ndarray: + """Fill the masked cells; masked cells with no known neighbour stay NaN.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + parts = self._parts(context, self.share_edges, self.context_median) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + rho = self.rho[np.clip(parts["sex"], 0, 3), parts["age"]] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + valid = parts["valid"] + drawn = np.full(len(rows), np.nan) + if valid.any(): + sub = {k: v[valid] for k, v in parts.items()} + drawn[valid] = self._quantile(sub, ndtr(z[valid])) + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + # -- persistence ---------------------------------------------------------- + def to_bytes(self) -> bytes: + arrays = { + "kind": np.array(self.name), + "share_edges": self.share_edges, + "context_median": np.array(self.context_median), + "rho": self.rho, + } + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> OddQuantileFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + share_edges=arrays["share_edges"], + context_median=float(arrays["context_median"]), + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + rho=arrays["rho"], + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: a quantile regression forest (QRF) draw +# -------------------------------------------------------------------------- + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + ] + ).astype(np.float32) + + +#: The reference level of a unit with no positive share around it. +_DEFAULT_LEVEL = 0.3 + + +def reference_level(context: OddContext) -> np.ndarray: + """The level a unit's share is drawn relative to. + + The geometric mean of the positive shares at ``t-1`` and ``t+1``; else + of those at ``t-3`` and ``t+3``; else the mean positive share at + offsets 5-9; else 0.3. + """ + + def geometric(a, b): + a = np.nan_to_num(a, nan=0.0) + b = np.nan_to_num(b, nan=0.0) + both = (a > 0) & (b > 0) + one = np.where(a > 0, a, b) + value = np.where(both, np.sqrt(np.where(both, a * b, 1.0)), one) + return np.where((a > 0) | (b > 0), value, np.nan) + + level = geometric(context.left, context.right) + level = np.where( + np.isnan(level), geometric(context.left3, context.right3), level + ) + level = np.where( + np.isnan(level) & (context.wide_mean > 0), context.wide_mean, level + ) + return np.where(np.isnan(level), _DEFAULT_LEVEL, level) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + A random forest (scikit-learn; split target ``log(share + 0.01)``) + partitions TRAIN units by :func:`odd_features`. Every TRAIN unit used + in the fit is passed down every tree, and each leaf keeps the sorted + true shares of the units that reach it (zeros and the cap included, + stored as shares times 65,535). The forest's conditional law of a unit + is the average over trees of its leaves' empirical laws: zero, the cap + and every positive share in between keep their own mass, so the draw + is two-part by construction. A draw picks a tree by one seeded + uniform and takes the leaf's value at the quantile of a second, the + copula uniform. A person's consecutive masked years (two apart) are + joined by a Gaussian AR(1) copula with correlation ``rho`` by sex and + age band, learned on TRAIN from the probability integral transforms of + consecutive units under the forest's law. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + stream: str = "epuf_fill.odd_forest.v2" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + *, + n_units=600_000, + n_trees=8, + min_leaf=25, + max_features=0.7, + seed=0, + n_jobs=8, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + known = np.isfinite(shares) + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x = odd_features(context) + y = target[chosen] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y + 0.01)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 16)), + ) + rho, diagnostics = provisional._fit_rho( + shares, years, birth_year, sex, unit_years, seed + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + **diagnostics, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _draw(self, x, tree_u, u): + """One draw per unit: a tree by ``tree_u``, a leaf value by ``u``.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + out = np.empty(len(u)) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if not len(rows): + continue + leaf = self._leaves(x[rows], t) + start = self.leaf_offsets[leaf] + count = self.leaf_offsets[leaf + 1] - start + pick = start + np.minimum( + (u[rows] * count).astype(np.int64), count - 1 + ) + out[rows] = self.leaf_values[pick] / _SHARE_SCALE + return out + + def _cdf(self, x, share, jitter): + """The forest's randomised CDF at each unit's true share.""" + + stored = np.round(np.clip(share, 0.0, 1.0) * _SHARE_SCALE) + total = np.zeros(len(share)) + for t in range(self.n_trees): + leaf = self._leaves(x, t) + start = self.leaf_offsets[leaf] + stop = self.leaf_offsets[leaf + 1] + below = np.empty(len(share)) + equal = np.empty(len(share)) + for i in range(len(share)): + values = self.leaf_values[start[i] : stop[i]] + low = np.searchsorted(values, stored[i], side="left") + high = np.searchsorted(values, stored[i], side="right") + below[i] = low + equal[i] = high - low + count = stop - start + total += (below + jitter * equal) / count + return np.clip(total / self.n_trees, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, shares, years, birth_year, sex, unit_years, seed): + """AR(1) correlation of normal scores of consecutive units (t, t+2). + + On 8,000 TRAIN persons chosen by seed, for every unit year. + """ + + rng = np.random.default_rng([seed, 1]) + persons = np.sort( + rng.choice( + len(shares), size=min(8_000, len(shares)), replace=False + ) + ) + known = np.isfinite(shares) + n_years = len(unit_years) + grid = np.full((len(persons), n_years), np.nan) + sex_grid = np.zeros((len(persons), n_years), dtype=np.int64) + age_grid = np.zeros((len(persons), n_years), dtype=np.int64) + for j, year in enumerate(unit_years): + unit_year = np.full(len(persons), year) + context = odd_context( + shares, years, persons, unit_year, birth_year, sex, known + ) + share = _take(shares, persons, _column_of(years, unit_year)) + jitter = hash_uniform( + self.stream + ".pit", seed, persons, unit_year + ) + grid[:, j] = ndtri(self._cdf(odd_features(context), share, jitter)) + sex_grid[:, j] = context.sex + age_grid[:, j] = _age_band(context.age) + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex_now = sex_grid[:, :-2].ravel() + age_now = age_grid[:, :-2].ravel() + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = (sex_now == s) & (age_now == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + four = np.corrcoef(grid[:, :-4].ravel(), grid[:, 4:].ravel())[0, 1] + overall = float(np.corrcoef(now, later)[0, 1]) + return rho, { + "rho_overall": overall, + "lag4_normal_score_correlation": float(four), + "lag4_ar1_prediction": overall**2, + "pit_mean": float(np.mean(ndtr(grid))), + } + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + rho = self.rho[np.clip(context.sex, 0, 3), _age_band(context.age)] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + drawn = np.zeros(len(rows)) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + drawn[valid] = self._draw( + features, tree_u[valid], ndtr(z[valid]) + ) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +_FOREST_ARRAYS = ( + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + known = np.isfinite(shares) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum): + members = np.flatnonzero(stratum == value) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=context.left[keep].astype(np.float32), + bank_right=context.right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:_DONOR_BANK] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + -1 for rows with no masked cell or no bank donor of their sex and + birth year. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + out = np.full(len(shares), -1, dtype=np.int64) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + recipients = targets[ + (sex[targets] == s) & (birth_year[targets] == b) + ] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + column = np.sort( + donor_match[:, j][np.isfinite(donor_match[:, j])] + ) + ranks_donor[:, j] = np.where( + np.isfinite(donor_match[:, j]), + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + nearest = np.argpartition(distance, k - 1, axis=1)[:, :k] + nearest_distance = np.take_along_axis(distance, nearest, 1) + order = np.lexsort((nearest, nearest_distance), axis=1) + nearest = np.take_along_axis(nearest, order, 1) + pick = np.minimum( + (u[recipients[block]] * k).astype(np.int64), k - 1 + ) + no_match = count.max(axis=1) == 0 + random_donor = np.minimum( + (u[recipients[block]] * len(donors)).astype(np.int64), + len(donors) - 1, + ) + out[recipients[block]] = donors[ + np.where( + no_match, + random_donor, + nearest[np.arange(len(nearest)), pick], + ) + ] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_quantile": OddQuantileFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/docs/amendments/gate_epuf_fill_dev_scores_after_amendment_1.json b/docs/amendments/gate_epuf_fill_dev_scores_after_amendment_1.json new file mode 100644 index 00000000..3ba2bdfc --- /dev/null +++ b/docs/amendments/gate_epuf_fill_dev_scores_after_amendment_1.json @@ -0,0 +1,33 @@ +{ + "schema": "populace_dynamics.epuf_fill_dev_scores_after_amendment.v1", + "registration_id": "2026-10-03-epuf-career-fill", + "purpose": "Every DEV score of a candidate or oracle-style fill produced after the amendment-1 rules commit 7061200d (06:04:18 UTC) and before the round-2 fixes, disclosed per referee round 2 finding 3. DEV only; no TEST person was read. Every event ran after 06:04 UTC: the scripts and fitted files carry modification times from 06:12 UTC on (quick_gaps.py was written at 06:18:42 UTC and first run after it). Scores were produced through the registered path epuf_fill_gate.score_candidate (union mask, validity checks). Events 1-6 passed no tolerances (gaps only); events 7-8 scored against the v2 tolerances. Gaps are filled minus true for correlations and log ratios otherwise, over two draw seeds (7100, 7101). Values are transcribed from the console output of each run.", + "fitting": "Every fill was fitted on TRAIN only (unit years 1991-2005).", + "events": [ + {"order": 1, "approx_utc": "06:19", "candidate": "odd_forest v1 (Vincentized QRF)", "description": "random forest, split target log(share + 0.01), 12 trees, min leaf 150, 2,000,000 TRAIN units; leaves store p0, pcap and 33 quantiles of log share; quantile functions averaged over trees; AR(1) copula (fitted rho about -0.007)", + "pooled_a22_74_gaps": {"men": {"r1": -0.0074, "r2": -0.0248, "r3": -0.0174, "r4": -0.0305, "zint": 0.0011, "zexit": -0.0583, "wint": 0.0451, "atcap": -0.0152, "level": -0.0026}, + "women": {"r1": -0.0201, "r2": -0.0417, "r3": -0.0249, "r4": -0.0434, "zint": 0.1473, "zexit": -0.0195, "wint": 0.0879, "atcap": 0.0319, "level": 0.0018}}}, + {"order": 2, "approx_utc": "06:28", "candidate": "odd_forest_rel (neighbour-relative leaves), with a draw bug", "description": "as event 1 with leaves holding quantiles of log(share / reference level); the draw omitted the reference level", "outcome": "bug: level gap +0.73 (men) and r1 -0.77; not a model result"}, + {"order": 3, "approx_utc": "06:32", "candidate": "odd_forest_rel, bug fixed", "description": "event 2 with the reference level applied in the draw", + "pooled_a22_74_gaps": {"men": {"r1": -0.0125, "r2": -0.0328, "r3": -0.0207, "r4": -0.0328, "zint": -0.0867, "zexit": -0.0167, "wint": -0.1138, "atcap": -0.0115, "level": 0.0042, "q10": -0.0914, "q50": 0.0065, "q90": 0.0}, + "women": {"r1": -0.0193, "r2": -0.0410, "r3": -0.0250, "r4": -0.0411, "zint": 0.0387, "zexit": 0.0056, "wint": 0.0545, "atcap": 0.0631, "level": 0.0094, "q10": -0.0590, "q50": 0.0095, "q90": 0.0174}}}, + {"order": 4, "approx_utc": "06:35", "candidate": "O1-from-TRAIN (diagnostic, oracle-style fill)", "description": "a TRAIN true value drawn at random from the same O1 stratum (sex, odd-universe membership, five-year age band, bins of t-1 and t+1); used to check the scoring path", + "gaps": {"odd.men.a22_74": {"atcap": 0.0010, "level": 0.0001, "r1": -0.0017, "r2": -0.0206, "r3": -0.0170, "r4": -0.0349, "wint": 0.0365, "zexit": -0.0054, "zint": 0.0511}, + "odd.men.a30_44": {"atcap": -0.0004, "level": 0.0011, "r1": -0.0014, "r2": -0.0215, "r3": -0.0186, "r4": -0.0375, "wint": 0.0036, "zexit": -0.0112, "zint": 0.0074}, + "odd.women.a22_74": {"atcap": 0.0040, "level": -0.0014, "r1": -0.0023, "r2": -0.0149, "r3": -0.0152, "r4": -0.0303, "wint": 0.0178, "zexit": -0.0011, "zint": 0.0740}, + "odd.women.a30_44": {"atcap": 0.0075, "level": 0.0003, "r1": -0.0014, "r2": -0.0177, "r3": -0.0170, "r4": -0.0339, "wint": 0.0392, "zexit": 0.0057, "zint": 0.0134}}}, + {"order": 5, "approx_utc": "06:38", "candidate": "odd_qrf v1 (Meinshausen QRF draw)", "description": "random forest as event 1 but 8 trees, min leaf 25, 600,000 units; leaves keep the true shares of the TRAIN units in them; a draw picks a tree and takes the leaf value at the copula uniform", + "gaps": {"odd.men.a22_74": {"atcap": -0.0113, "level": 0.0003, "q10": -0.0316, "q50": -0.0036, "q90": 0.0, "r1": -0.0008, "r2": -0.0101, "r3": -0.0099, "r4": -0.0167, "wint": 0.1388, "zexit": -0.0497, "zint": 0.0524}, + "odd.men.a30_44": {"atcap": -0.0268, "level": -0.0018, "q10": -0.0883, "q50": -0.0073, "q90": 0.0, "r1": 0.0009, "r2": -0.0099, "r3": -0.0089, "r4": -0.0204, "wint": 0.0853, "zexit": -0.0721, "zint": -0.1524}, + "odd.women.a22_74": {"atcap": 0.0261, "level": 0.0042, "q10": -0.0037, "q50": 0.0056, "q90": 0.0052, "r1": -0.0094, "r2": -0.0186, "r3": -0.0137, "r4": -0.0229, "wint": 0.1673, "zexit": -0.0169, "zint": 0.1631}, + "odd.women.a30_44": {"atcap": -0.0019, "level": 0.0021, "q10": -0.0246, "q50": -0.0019, "q90": 0.0004, "r1": -0.0069, "r2": -0.0212, "r3": -0.0125, "r4": -0.0289, "wint": 0.1836, "zexit": -0.0264, "zint": 0.0855}}}, + {"order": 6, "approx_utc": "06:45", "candidate": "odd_qrf_b", "description": "event 5 with 2,000,000 units, 10 trees, min leaf 20, max features 0.8", + "gaps": {"odd.men.a22_74": {"atcap": -0.0057, "q10": -0.0349, "r1": 0.0014, "r2": -0.0062, "r3": -0.0070, "r4": -0.0120, "wint": 0.0397, "zexit": -0.0420, "zint": -0.0117}, + "odd.women.a22_74": {"atcap": 0.0240, "q10": 0.0011, "r1": -0.0076, "r2": -0.0143, "r3": -0.0112, "r4": -0.0177, "wint": 0.0379, "zexit": -0.0075, "zint": 0.1294}}}, + {"order": 7, "approx_utc": "06:54", "candidate": "odd_qrf_sex (one forest per sex)", "description": "event 6's forest fitted separately for men and women, 1,500,000 units each", "n_failing": 39, "n_gating": 183, + "failing_cells_gap_tolerance": [["odd.men.a22_74.r2", -0.0132, 0.0046], ["odd.men.a60_74.zint", 0.5070, 0.1776], ["odd.men.a22_74.r4", -0.0170, 0.0070], ["odd.men.a30_44.r4", -0.0203, 0.0092], ["odd.men.a22_74.r3", -0.0100, 0.0047], ["odd.men.a30_44.r2", -0.0131, 0.0065], ["odd.men.a60_74.r2", -0.0403, 0.0208], ["odd.women.a22_29.r2", 0.0243, 0.0131], ["odd.women.a60_74.r2", -0.0411, 0.0222], ["odd.men.a22_74.wint", 0.1358, 0.0734], ["odd.women.a30_44.r4", -0.0184, 0.0102], ["odd.men.a45_59.r2", -0.0126, 0.0074], ["odd.women.a22_74.r4", -0.0130, 0.0077], ["odd.women.a22_74.r3", -0.0087, 0.0052], ["odd.women.a45_59.r2", -0.0123, 0.0074], ["odd.men.a22_29.r3", -0.0202, 0.0133], ["odd.men.a30_44.r3", -0.0091, 0.0061], ["odd.men.a22_74.zint", 0.0789, 0.0548], ["odd.men.a22_29.r1", -0.0105, 0.0073], ["odd.women.a60_74.r1", -0.0139, 0.0098], ["odd.women.a60_74.r4", -0.0519, 0.0363], ["odd.women.a45_59.r4", -0.0177, 0.0125], ["odd.men.a22_74.zexit", -0.0254, 0.0192], ["odd.women.a30_44.r2", -0.0093, 0.0071], ["odd.women.a22_74.r2", -0.0060, 0.0048], ["odd.men.a60_74.r1", -0.0122, 0.0098], ["odd.men.a45_59.zexit", -0.0441, 0.0363], ["odd.men.a30_44.zexit", -0.0388, 0.0320], ["odd.men.a60_74.r4", -0.0449, 0.0382], ["odd.men.a45_59.wint", 0.1515, 0.1334], ["odd.women.a60_74.r3", -0.0205, 0.0185], ["odd.men.a45_59.r4", -0.0134, 0.0121], ["odd.men.a45_59.zint", 0.1019, 0.0929], ["odd.women.a22_29.q50", 0.0238, 0.0218], ["odd.men.a30_44.wint", 0.1301, 0.1200], ["odd.men.a45_59.r3", -0.0073, 0.0072], ["odd.women.b1966_1980.paime_p25", -0.0337, 0.0332], ["odd.women.a22_74.zexit", -0.0163, 0.0160], ["odd.men.a22_74.r1", -0.0026, 0.0026]]}, + {"order": 8, "approx_utc": "07:05", "candidate": "odd_qrf_sex2 (one forest per sex, level features)", "description": "event 7 with 3,000,000 units per sex, min leaf 15, and features adding the mean, geometric mean and number positive of the shares at t-3, t-1, t+1, t+3", "n_failing": 24, "n_gating": 183, + "failing_cells_gap_tolerance": [["odd.women.a22_29.r2", 0.0281, 0.0131], ["odd.men.a60_74.zint", 0.3148, 0.1776], ["odd.men.a22_74.r2", -0.0074, 0.0046], ["odd.men.a22_29.r1", -0.0103, 0.0073], ["odd.women.a60_74.r2", -0.0311, 0.0222], ["odd.women.a30_44.r4", -0.0142, 0.0102], ["odd.men.a30_44.r4", -0.0121, 0.0092], ["odd.men.a60_74.r2", -0.0259, 0.0208], ["odd.men.a30_44.r2", -0.0080, 0.0065], ["odd.men.a22_74.zexit", -0.0228, 0.0192], ["odd.men.a22_74.r4", -0.0083, 0.0070], ["odd.women.a22_29.q50", 0.0254, 0.0218], ["odd.women.a45_59.r2", -0.0084, 0.0074], ["odd.men.a30_44.zexit", -0.0360, 0.0320], ["odd.women.a22_29.level", 0.0203, 0.0185], ["odd.women.a60_74.r1", -0.0107, 0.0098], ["odd.men.a45_59.r2", -0.0081, 0.0074], ["odd.men.a30_44.zint", -0.0968, 0.0899], ["odd.men.a22_29.r3", -0.0137, 0.0133], ["odd.men.a22_29.q50", 0.0237, 0.0230], ["odd.men.a45_59.zexit", -0.0374, 0.0363], ["odd.women.b1966_1980.paime_p25", -0.0337, 0.0332], ["odd.women.a30_44.r2", -0.0071, 0.0071], ["odd.women.a22_29.r4", 0.0200, 0.0198]]} + ], + "pre_family": "No pre-career candidate was scored after 7061200d." +} diff --git a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl new file mode 100644 index 00000000..13e1e11d --- /dev/null +++ b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl @@ -0,0 +1,5 @@ +{"utc": "2026-10-04T07:22:15+00:00", "candidate": "pre_donor", "family": "pre", "artifact_sha256": "74aa1601dcd4230b1daddfa3998d0299a92af7c174912404e8293bd476e17337", "code_sha256": "09ffcfa06ee064d14e3403cb3101b05a20fee1aaec7c7b97abb77567d1cc4cc5", "seeds": [7100, 7101, 7102, 7103], "n_failing": 4, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1930_1945.aime_p10", 0.33330266185993374, 0.16621909484652816], ["pre.women.b1940_1945.pzero", -0.055262963678008536, 0.04701699812231318], ["pre.women.b1940_1945.aime_p10", 0.29046229495810305, 0.26939110963753704], ["pre.women.b1935_1939.aime_p10", 0.36695525334628876, 0.3512121935176347], ["pre.women.b1946_1980.yr_cross", 0.015630127488157397, 0.015684351176846988], ["pre.men.b1946_1980.ylevel", -0.01508791014685329, 0.01520534060461241], ["pre.women.b1956_1965.yr_cross", 0.02934637723770933, 0.03047360671843986], ["pre.men.b1940_1945.pr_in", 0.05951680046203045, 0.06500235072966114], ["pre.women.b1946_1980.paime_p10", 0.04740223889458406, 0.05217708504283977], ["pre.men.b1956_1965.ylevel", -0.02466055259076949, 0.027201666962537053], ["pre.men.b1946_1980.yr_cross", 0.012922680711020706, 0.014632893176957621], ["pre.women.b1930_1934.aime_p10", 0.33702794655493395, 0.3914251373153828]]} +{"utc": "2026-10-04T07:30:46+00:00", "candidate": "pre_donor2", "family": "pre", "artifact_sha256": "eb954ac3a9ceb25e1d559ac555e6ec2527b310f951dd7cfdf3b34edc322d3165", "code_sha256": "ff279cd69d1b94ec1447d9d14db0d893546142cd34b54fa8b23e22b3e76a0584", "seeds": [7100, 7101, 7102, 7103], "n_failing": 6, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1946_1980.yr_cross", 0.03024586359102016, 0.015684351176846988], ["pre.women.b1956_1965.yr_cross", 0.0550962861890853, 0.03047360671843986], ["pre.women.b1946_1980.yzero", 0.011763944031377704, 0.008556191699814752], ["pre.men.b1946_1980.yr_cross", 0.016499643760206573, 0.014632893176957621], ["pre.women.b1946_1955.yr_cross", 0.04061298355664006, 0.03669253843323441], ["pre.women.b1966_1980.yr_cross", 0.022903326696107174, 0.02158530789673962], ["pre.men.b1946_1980.yzero", 0.011116994769767019, 0.011384493587046594], ["pre.men.b1946_1980.ylevel", -0.013914974175574635, 0.01520534060461241], ["pre.women.b1966_1980.yzero", 0.012869325136944165, 0.014833297517636257], ["pre.men.b1956_1965.ylevel", -0.022079629566834402, 0.027201666962537053]]} +{"utc": "2026-10-04T07:31:20+00:00", "candidate": "pre_chain2", "family": "pre", "artifact_sha256": "37c2d72e6a7700cf7ac9dfe705434152f71e708ed92adf0fbad78611328b5761", "code_sha256": "ff279cd69d1b94ec1447d9d14db0d893546142cd34b54fa8b23e22b3e76a0584", "seeds": [7100, 7101, 7102, 7103], "n_failing": 55, "n_gating": 136, "tier": "not_adopted", "worst": [["pre.women.b1966_1980.ylevel", 0.45917079627409274, 0.017769920520911725], ["pre.men.b1966_1980.ylevel", 0.4811824306056769, 0.021382204828984532], ["pre.women.b1966_1980.yzero", 0.2583487735491933, 0.014833297517636257], ["pre.women.b1946_1980.yzero", 0.1489263401252473, 0.008556191699814752], ["pre.women.b1946_1980.ylevel", 0.20714311009839204, 0.015450091379218515], ["pre.men.b1946_1980.yzero", 0.15148656243554004, 0.011384493587046594], ["pre.men.b1966_1980.yzero", 0.19620307014155636, 0.01675184764087755], ["pre.men.b1946_1980.ylevel", 0.1430531564860713, 0.01520534060461241], ["pre.women.b1956_1965.yzero", 0.15643826543566397, 0.017102901274216892], ["pre.men.b1956_1965.yzero", 0.16956100424184672, 0.022693736955041323]]} +{"utc": "2026-10-04T07:42:02+00:00", "candidate": "odd_qrf_sex2", "family": "odd", "artifact_sha256": "be03f0f3c702e391008fb72eec16d799385be93fbd5de01ce78a79e1fe45330b", "code_sha256": "9b2c5dd94d3d51e5f53994b7aa707ce7955b99eeb1e6e8bc481e753d1260761a", "seeds": [7100, 7101, 7102, 7103], "n_failing": 23, "n_gating": 183, "tier": "not_adopted", "worst": [["odd.women.a22_29.r2", 0.02714856253605591, 0.013111971249341917], ["odd.men.a60_74.zint", 0.33247368988113246, 0.17758159695027936], ["odd.women.a60_74.r2", -0.03422723027556407, 0.022174026993056747], ["odd.men.a22_74.r2", -0.007062508310260118, 0.004576681103770273], ["odd.men.a22_29.r1", -0.010609081809470733, 0.007273490161268873], ["odd.women.a30_44.r4", -0.014271254170137415, 0.010187042430903364], ["odd.men.a30_44.zexit", -0.04086629032766431, 0.031953074094816486], ["odd.men.a60_74.r2", -0.02629022691682592, 0.020757725353477027], ["odd.men.a30_44.r4", -0.011478929773406477, 0.009218001294372115], ["odd.women.a45_59.r2", -0.009149018075044535, 0.0074184353060767995], ["odd.men.a22_74.zexit", -0.022985129792897574, 0.01916127382099423], ["odd.women.a60_74.r1", -0.011531037538756617, 0.009751654703991025], ["odd.men.a30_44.r2", -0.0075100613637079094, 0.006500650570758907], ["odd.women.a22_29.level", 0.020540551604554036, 0.018517427407013797], ["odd.men.a22_74.r4", -0.007762715111083396, 0.007000982741779954], ["odd.women.a60_74.r4", -0.03868148727219123, 0.036314307016843086], ["odd.women.a22_29.q50", 0.0232742152048111, 0.021849929026002995], ["odd.men.a45_59.r2", -0.007692352520425327, 0.0074262368998509335], ["odd.women.a45_59.r4", -0.012890705852779516, 0.012512985825002505], ["odd.women.a22_74.r4", -0.007885437242168614, 0.007666262733486471], ["odd.men.a22_29.r3", -0.01343644400734656, 0.013280307041936976], ["odd.men.a22_29.q50", 0.023239013149086052, 0.022980444394274636], ["odd.women.a30_44.r2", -0.007087772863819786, 0.007078708407960211], ["odd.men.a30_44.q10", -0.04751745643811356, 0.047669064141889934], ["odd.women.a22_74.r3", -0.005150669696892374, 0.005171487109756781], ["odd.men.a30_44.zint", -0.08884255086813031, 0.08989768311292742], ["odd.women.a22_29.r4", 0.01925238657360928, 0.01982762523559322], ["odd.men.a60_74.r1", -0.009265651157384092, 0.009750532797445479], ["odd.men.a22_29.level", 0.020253468381523643, 0.02171146931539837], ["odd.men.a22_74.r3", -0.004148209781093093, 0.004722018684826323]]} +{"utc": "2026-10-04T07:44Z (approximate)", "candidate": "pre_chain3 (diagnostic check, not scored against tolerances)", "family": "pre", "description": "the chained alternative refitted with single-year ages 15-24 and unit years 1951-2005; DEV persons born 1966-1980 (men, 30,000) filled with seed 7100 and compared with the truth by age", "printed": {"15": {"truth_mean": 0.0028, "truth_zero": 0.857, "fill_mean": 0.0081, "fill_zero": 0.863}, "16": {"truth_mean": 0.0094, "truth_zero": 0.661, "fill_mean": 0.0189, "fill_zero": 0.677}, "17": {"truth_mean": 0.0222, "truth_zero": 0.488, "fill_mean": 0.0406, "fill_zero": 0.508}, "18": {"truth_mean": 0.0409, "truth_zero": 0.369, "fill_mean": 0.0737, "fill_zero": 0.387}, "19": {"truth_mean": 0.0697, "truth_zero": 0.304, "fill_mean": 0.1113, "fill_zero": 0.331}, "20": {"truth_mean": 0.0934, "truth_zero": 0.287, "fill_mean": 0.1285, "fill_zero": 0.311}, "21": {"truth_mean": 0.1116, "truth_zero": 0.276, "fill_mean": 0.134, "fill_zero": 0.294}}, "earlier_check": "the same comparison with pre_chain2 (five-year age bands) printed fill means 0.0236-0.1416 against truth 0.0028-0.1116 at ages 15-21", "code_sha256": "9b2c5dd94d3d51e5f53994b7aa707ce7955b99eeb1e6e8bc481e753d1260761a"} diff --git a/docs/amendments/gate_epuf_fill_dev_scores_before_amendment_1.json b/docs/amendments/gate_epuf_fill_dev_scores_before_amendment_1.json new file mode 100644 index 00000000..2c071fb3 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_dev_scores_before_amendment_1.json @@ -0,0 +1,1089 @@ +{ + "schema": "populace_dynamics.epuf_fill_dev_scores_before_amendment.v1", + "registration_id": "2026-10-03-epuf-career-fill", + "purpose": "Every DEV score of a candidate fill produced before amendment 1, disclosed so the next referee can check that the amendments do not favour them (round-1 finding 7). DEV only; no TEST person was read. Scored against the v1 floors (runs/epuf_fill_gate_floors_v1.json) by the unregistered development scorer .diag/epuf-fill/dev_score.py, which replaces masked cells only and passes the full true matrix to the fill (the fill reads only unmasked cells).", + "fitting": "Every fill was fitted on a 30 percent hash sample of TRAIN persons (unit years 1991-2005 for odd fills, 1951-1990 for the chained fill).", + "events": [ + { + "order": 1, + "utc": "2026-10-04T05:25Z", + "candidate": "odd_quantile v0", + "description": "cells keyed on 10 level bins x 5 slope bins of the neighbours, context t-3/t+3 and offsets 5-9, residual relative to the neighbours' geometric mean; AR(1) copula (fitted rho about 0.004)", + "seeds": [ + 7100, + 7101, + 7102, + 7103 + ], + "n_failing": 29, + "n_gating": 70, + "worst_cells_transcribed_from_console": { + "odd.men.a30_44.r2": [ + -0.0517, + 0.0063 + ], + "odd.men.a30_44.r4": [ + -0.065, + 0.0087 + ], + "odd.men.a45_59.r2": [ + -0.0611, + 0.0083 + ], + "odd.women.a30_44.r2": [ + -0.0485, + 0.0067 + ], + "odd.women.a45_59.r2": [ + -0.0459, + 0.007 + ], + "odd.women.a30_44.r4": [ + -0.0587, + 0.0098 + ], + "odd.women.a30_44.r1": [ + -0.0198, + 0.0035 + ], + "odd.men.a30_44.r1": [ + -0.0191, + 0.0035 + ], + "odd.men.a45_59.r1": [ + -0.0216, + 0.0041 + ], + "odd.men.a45_59.r4": [ + -0.0688, + 0.0136 + ], + "odd.women.a45_59.r4": [ + -0.0519, + 0.0111 + ], + "odd.women.a45_59.r1": [ + -0.0166, + 0.0037 + ] + }, + "note": "[gap, tolerance] pairs; the full JSON of this run was not written (the run stopped at the next candidate)." + }, + { + "order": 2, + "utc": "2026-10-04T05:26Z", + "candidate": "odd_knn v0", + "description": "registered alternative: k=10 nearest TRAIN units in (t-1, t+1) by sex and five-year age band", + "seeds": [ + 7100, + 7101, + 7102, + 7103 + ], + "n_failing": 24, + "n_gating": 70, + "worst_cells_transcribed_from_console": { + "odd.men.a45_59.wint": [ + -1.1022, + 0.1541 + ], + "odd.women.a45_59.wint": [ + -0.9388, + 0.1333 + ], + "odd.men.a30_44.wint": [ + -0.7213, + 0.1142 + ], + "odd.women.a30_44.wint": [ + -0.6308, + 0.1005 + ], + "odd.men.a30_44.r4": [ + -0.0413, + 0.0087 + ], + "odd.women.a18_29.wint": [ + -0.6682, + 0.1488 + ], + "odd.men.a18_29.wint": [ + -0.6732, + 0.1638 + ], + "odd.men.a30_44.r2": [ + -0.0258, + 0.0063 + ], + "odd.women.a30_44.r4": [ + -0.0317, + 0.0098 + ], + "odd.men.a45_59.r2": [ + -0.0252, + 0.0083 + ], + "odd.women.a30_44.r2": [ + -0.0168, + 0.0067 + ], + "odd.men.a45_59.r4": [ + -0.0339, + 0.0136 + ] + } + }, + { + "order": 3, + "utc": "2026-10-04T05:27Z", + "candidate": "pre_donor v0", + "description": "rank-kNN donors, k=10, bank of 2,000 per sex and birth year", + "outcome": "stopped: masked cells of persons outside the bank (uncoded sex, births outside 1905-1985) were left NaN; no cell was scored" + }, + { + "order": 4, + "utc": "2026-10-04T05:35Z", + "candidate": "odd_quantile v1", + "description": "cells keyed on joint bins of t-1 and t+1 (zero, 20 quantile bins, cap), context, offsets 5-9; residual relative to the neighbours' geometric mean; AR(1) copula (fitted rho about 0.0005)", + "seeds": [ + 7100, + 7101 + ], + "n_gating": 70, + "n_failing": 26, + "passes": false, + "gating_cells": { + "odd.men.a18_29.level": { + "truth": 0.1856554260931258, + "filled": 0.18608404131499992, + "gap": 0.0023059989556522, + "tolerance": 0.02530824766097302, + "passes": true + }, + "odd.men.a18_29.r1": { + "truth": 0.8210934709127689, + "filled": 0.8128344044821749, + "gap": -0.008259066430593931, + "tolerance": 0.007487817283896471, + "passes": false + }, + "odd.men.a18_29.r2": { + "truth": 0.7156445156848613, + "filled": 0.6902242874262012, + "gap": -0.025420228258660083, + "tolerance": 0.012557715120508784, + "passes": false + }, + "odd.men.a18_29.r4": { + "truth": 0.5850452217259834, + "filled": 0.5531870927837401, + "gap": -0.03185812894224327, + "tolerance": 0.021292587125646432, + "passes": false + }, + "odd.men.a18_29.wint": { + "truth": 0.10126339049815054, + "filled": 0.08697218619397606, + "gap": -0.15213658045555745, + "tolerance": 0.16382575433989846, + "passes": true + }, + "odd.men.a18_29.zexit": { + "truth": 0.38930612041218593, + "filled": 0.403203672746326, + "gap": 0.0350758496339042, + "tolerance": 0.04962920531417247, + "passes": true + }, + "odd.men.a18_29.zint": { + "truth": 0.034823846728263656, + "filled": 0.03288175830054724, + "gap": -0.057384357863278446, + "tolerance": 0.09960815163206226, + "passes": true + }, + "odd.men.a30_44.atcap": { + "truth": 0.10574805846620577, + "filled": 0.09244180372779878, + "gap": -0.13448016046771327, + "tolerance": 0.04857889866017903, + "passes": false + }, + "odd.men.a30_44.level": { + "truth": 0.3982899917120792, + "filled": 0.39746936911881725, + "gap": -0.0020624900555529235, + "tolerance": 0.0139684504916683, + "passes": true + }, + "odd.men.a30_44.r1": { + "truth": 0.9000861304840623, + "filled": 0.8902038867802486, + "gap": -0.009882243703813631, + "tolerance": 0.0035183123602736976, + "passes": false + }, + "odd.men.a30_44.r2": { + "truth": 0.8479233224100629, + "filled": 0.8116338879418894, + "gap": -0.036289434468173454, + "tolerance": 0.006340195296747853, + "passes": false + }, + "odd.men.a30_44.r4": { + "truth": 0.7869108820616492, + "filled": 0.7337785101761347, + "gap": -0.05313237188551445, + "tolerance": 0.008697176421644505, + "passes": false + }, + "odd.men.a30_44.wint": { + "truth": 0.07239145006330527, + "filled": 0.07242868846354361, + "gap": 0.0005142710321623944, + "tolerance": 0.11420427683518232, + "passes": true + }, + "odd.men.a30_44.zexit": { + "truth": 0.4342024282437196, + "filled": 0.4336848878196162, + "gap": -0.0011926444271513903, + "tolerance": 0.032636532740128565, + "passes": true + }, + "odd.men.a30_44.zint": { + "truth": 0.017933882760031997, + "filled": 0.017939463843878484, + "gap": 0.00031115490579303184, + "tolerance": 0.08051204375260113, + "passes": true + }, + "odd.men.a45_59.atcap": { + "truth": 0.15826463477931516, + "filled": 0.14392299374265594, + "gap": -0.09499014474477208, + "tolerance": 0.04356937974768127, + "passes": false + }, + "odd.men.a45_59.level": { + "truth": 0.4279852808128974, + "filled": 0.42585689377654856, + "gap": -0.004985444636138814, + "tolerance": 0.014453976889178868, + "passes": true + }, + "odd.men.a45_59.r1": { + "truth": 0.9077661818197654, + "filled": 0.8978818265396777, + "gap": -0.009884355280087687, + "tolerance": 0.004111314219586486, + "passes": false + }, + "odd.men.a45_59.r2": { + "truth": 0.850421444941529, + "filled": 0.8131646648078616, + "gap": -0.0372567801336674, + "tolerance": 0.008347506877010717, + "passes": false + }, + "odd.men.a45_59.r4": { + "truth": 0.7598303885400162, + "filled": 0.7125722429702188, + "gap": -0.04725814556979735, + "tolerance": 0.01362625190129057, + "passes": false + }, + "odd.men.a45_59.wint": { + "truth": 0.05825168693530555, + "filled": 0.053124166063055166, + "gap": -0.09214112280446285, + "tolerance": 0.15412394675503588, + "passes": true + }, + "odd.men.a45_59.zexit": { + "truth": 0.45185706741770815, + "filled": 0.46197852490758673, + "gap": 0.022152499821214144, + "tolerance": 0.03920259283391604, + "passes": true + }, + "odd.men.a45_59.zint": { + "truth": 0.015835938476928848, + "filled": 0.016121045392022006, + "gap": 0.01784364137448069, + "tolerance": 0.10545535031261653, + "passes": true + }, + "odd.men.a60_74.atcap": { + "truth": 0.10024796315975912, + "filled": 0.08694022313543161, + "gap": -0.1424259562694261, + "tolerance": 0.11678068710169077, + "passes": false + }, + "odd.men.a60_74.level": { + "truth": 0.20453937198442498, + "filled": 0.20314848602180408, + "gap": -0.0068233151008298965, + "tolerance": 0.04536572508883381, + "passes": true + }, + "odd.men.a60_74.r1": { + "truth": 0.8712782553777247, + "filled": 0.8686099496263309, + "gap": -0.0026683057513938735, + "tolerance": 0.0093675927151269, + "passes": true + }, + "odd.men.a60_74.r2": { + "truth": 0.7888015745113253, + "filled": 0.76258825079463, + "gap": -0.02621332371669527, + "tolerance": 0.021939428752443885, + "passes": false + }, + "odd.men.a60_74.r4": { + "truth": 0.6781549627202678, + "filled": 0.6409690758565164, + "gap": -0.03718588686375135, + "tolerance": 0.0419843149619278, + "passes": true + }, + "odd.men.a60_74.zexit": { + "truth": 0.5047452182800409, + "filled": 0.5226310410278873, + "gap": 0.03482196491461498, + "tolerance": 0.041847126964410425, + "passes": true + }, + "odd.men.a60_74.zint": { + "truth": 0.027132152458672565, + "filled": 0.025664003660838562, + "gap": -0.055630087885179424, + "tolerance": 0.16089541808023508, + "passes": true + }, + "odd.men.b1936_1940.aime_p25": { + "truth": 1131.0, + "filled": 1131.0, + "gap": 0.0, + "tolerance": 0.10413587699123662, + "passes": true + }, + "odd.men.b1936_1940.aime_p50": { + "truth": 2430.0, + "filled": 2429.0, + "gap": -0.00041160733242140424, + "tolerance": 0.05101259969798402, + "passes": true + }, + "odd.men.b1936_1940.aime_p75": { + "truth": 3582.0, + "filled": 3578.0, + "gap": -0.0011173185519925966, + "tolerance": 0.03379657214494975, + "passes": true + }, + "odd.men.b1941_1945.aime_p25": { + "truth": 1174.25, + "filled": 1178.625, + "gap": 0.003718858878736242, + "tolerance": 0.08798542518866409, + "passes": true + }, + "odd.men.b1941_1945.aime_p50": { + "truth": 2821.5, + "filled": 2820.75, + "gap": -0.00026585139063950436, + "tolerance": 0.05081989806426865, + "passes": true + }, + "odd.men.b1941_1945.aime_p75": { + "truth": 4421.0, + "filled": 4419.375, + "gap": -0.00036763146773743927, + "tolerance": 0.03296414548783888, + "passes": true + }, + "odd.women.a18_29.level": { + "truth": 0.14595361593292944, + "filled": 0.14628307231176474, + "gap": 0.0022547238704886396, + "tolerance": 0.022701219029856848, + "passes": true + }, + "odd.women.a18_29.r1": { + "truth": 0.799002707000061, + "filled": 0.787612803921771, + "gap": -0.01138990307829002, + "tolerance": 0.007030001813072651, + "passes": false + }, + "odd.women.a18_29.r2": { + "truth": 0.6744743883343822, + "filled": 0.6492833848878561, + "gap": -0.02519100344652614, + "tolerance": 0.01247155021048016, + "passes": false + }, + "odd.women.a18_29.r4": { + "truth": 0.5116284868083779, + "filled": 0.4879751923831376, + "gap": -0.023653294425240334, + "tolerance": 0.01877841052722656, + "passes": false + }, + "odd.women.a18_29.wint": { + "truth": 0.10130434782608695, + "filled": 0.09183574879227054, + "gap": -0.09812768841349095, + "tolerance": 0.14880165994710184, + "passes": true + }, + "odd.women.a18_29.zexit": { + "truth": 0.3953034223502417, + "filled": 0.40632475711565685, + "gap": 0.02749910636266828, + "tolerance": 0.041086747761904435, + "passes": true + }, + "odd.women.a18_29.zint": { + "truth": 0.03753038230821729, + "filled": 0.03750344233048768, + "gap": -0.0007180755883609002, + "tolerance": 0.08105378527211228, + "passes": true + }, + "odd.women.a30_44.atcap": { + "truth": 0.034619662054697874, + "filled": 0.029689704606192323, + "gap": -0.1536214486808758, + "tolerance": 0.08282095025205272, + "passes": false + }, + "odd.women.a30_44.level": { + "truth": 0.2579924798566703, + "filled": 0.2581554675647999, + "gap": 0.0006315542447132838, + "tolerance": 0.01525041958748848, + "passes": true + }, + "odd.women.a30_44.r1": { + "truth": 0.888241419643356, + "filled": 0.8794515719183367, + "gap": -0.00878984772501934, + "tolerance": 0.0034720126766651514, + "passes": false + }, + "odd.women.a30_44.r2": { + "truth": 0.8242277470219269, + "filled": 0.795112182563535, + "gap": -0.029115564458391918, + "tolerance": 0.006661015175432743, + "passes": false + }, + "odd.women.a30_44.r4": { + "truth": 0.7490435327097834, + "filled": 0.7061801506397472, + "gap": -0.04286338207003626, + "tolerance": 0.009827512018128022, + "passes": false + }, + "odd.women.a30_44.wint": { + "truth": 0.06073485056210584, + "filled": 0.060172744721689056, + "gap": -0.009298173350947181, + "tolerance": 0.10054744200738645, + "passes": true + }, + "odd.women.a30_44.zexit": { + "truth": 0.45066799061202384, + "filled": 0.44885132695432384, + "gap": -0.004039193138246855, + "tolerance": 0.021292914338298448, + "passes": true + }, + "odd.women.a30_44.zint": { + "truth": 0.022222222222222223, + "filled": 0.020975545965738196, + "gap": -0.05773550784149162, + "tolerance": 0.07213256092968776, + "passes": true + }, + "odd.women.a45_59.atcap": { + "truth": 0.04053483605515179, + "filled": 0.035861452387741806, + "gap": -0.12249878493879729, + "tolerance": 0.08515639674437901, + "passes": false + }, + "odd.women.a45_59.level": { + "truth": 0.28104586423928846, + "filled": 0.28084758788434216, + "gap": -0.0007057436331625588, + "tolerance": 0.017215883497257563, + "passes": true + }, + "odd.women.a45_59.r1": { + "truth": 0.9101006329634369, + "filled": 0.9034002609287783, + "gap": -0.006700372034658564, + "tolerance": 0.0037474833362833664, + "passes": false + }, + "odd.women.a45_59.r2": { + "truth": 0.8498173267452248, + "filled": 0.8235239714806456, + "gap": -0.02629335526457921, + "tolerance": 0.006982462259567813, + "passes": false + }, + "odd.women.a45_59.r4": { + "truth": 0.7620167777938601, + "filled": 0.7252333728874433, + "gap": -0.03678340490641685, + "tolerance": 0.011136170330069556, + "passes": false + }, + "odd.women.a45_59.wint": { + "truth": 0.050115932427956277, + "filled": 0.04433587280556475, + "gap": -0.12254485082794853, + "tolerance": 0.13330824317006054, + "passes": true + }, + "odd.women.a45_59.zexit": { + "truth": 0.4693136110029843, + "filled": 0.47349811859348645, + "gap": 0.008876714058589474, + "tolerance": 0.028997424168220973, + "passes": true + }, + "odd.women.a45_59.zint": { + "truth": 0.01632776600720909, + "filled": 0.015444514762127506, + "gap": -0.055613185908567786, + "tolerance": 0.09280080971364012, + "passes": true + }, + "odd.women.a60_74.level": { + "truth": 0.12926943691615103, + "filled": 0.12853293522547724, + "gap": -0.005713707660912171, + "tolerance": 0.04440816434299283, + "passes": true + }, + "odd.women.a60_74.r1": { + "truth": 0.8610627385333514, + "filled": 0.8573540519984275, + "gap": -0.003708686534923844, + "tolerance": 0.009859716852600387, + "passes": true + }, + "odd.women.a60_74.r2": { + "truth": 0.7722738686836694, + "filled": 0.7448713764954531, + "gap": -0.027402492188216332, + "tolerance": 0.02069132660379932, + "passes": false + }, + "odd.women.a60_74.r4": { + "truth": 0.6501204938024134, + "filled": 0.6094375641494021, + "gap": -0.040682929653011346, + "tolerance": 0.03628392371863108, + "passes": false + }, + "odd.women.a60_74.zexit": { + "truth": 0.5155483759303063, + "filled": 0.5172791784457393, + "gap": 0.0033515839663482705, + "tolerance": 0.038573729006110224, + "passes": true + }, + "odd.women.b1936_1940.aime_p25": { + "truth": 287.75, + "filled": 287.0, + "gap": -0.0026098318423706246, + "tolerance": 0.13713107003771732, + "passes": true + }, + "odd.women.b1936_1940.aime_p50": { + "truth": 786.0, + "filled": 786.0, + "gap": 0.0, + "tolerance": 0.08788489934963226, + "passes": true + }, + "odd.women.b1936_1940.aime_p75": { + "truth": 1566.0, + "filled": 1567.25, + "gap": 0.0007978936033294914, + "tolerance": 0.06811366168257448, + "passes": true + }, + "odd.women.b1941_1945.aime_p25": { + "truth": 399.0, + "filled": 398.5, + "gap": -0.0012539186595939, + "tolerance": 0.10802967033363925, + "passes": true + }, + "odd.women.b1941_1945.aime_p50": { + "truth": 1077.0, + "filled": 1075.0, + "gap": -0.001858736594625654, + "tolerance": 0.07006551745181369, + "passes": true + }, + "odd.women.b1941_1945.aime_p75": { + "truth": 2116.0, + "filled": 2119.0, + "gap": 0.0014167652901093675, + "tolerance": 0.05896213826035607, + "passes": true + } + } + }, + { + "order": 5, + "utc": "2026-10-04T05:38Z", + "candidate": "pre_donor v1", + "description": "as v0, with masked cells of persons outside the bank set to zero", + "seeds": [ + 7100, + 7101 + ], + "n_gating": 60, + "n_failing": 1, + "passes": false, + "gating_cells": { + "pre.men.b1930_1934.aime_p25": { + "truth": 961.0, + "filled": 973.5, + "gap": 0.01292341584196155, + "tolerance": 0.16573412297046017, + "passes": true + }, + "pre.men.b1930_1934.aime_p50": { + "truth": 1871.0, + "filled": 1853.5, + "gap": -0.009397303683302383, + "tolerance": 0.08422855628807996, + "passes": true + }, + "pre.men.b1930_1934.aime_p75": { + "truth": 2604.0, + "filled": 2591.5, + "gap": -0.004811865698698625, + "tolerance": 0.056066692635054774, + "passes": true + }, + "pre.men.b1930_1934.plevel": { + "truth": 0.5292246059138269, + "filled": 0.524078256506457, + "gap": -0.009771909947440593, + "tolerance": 0.05371982702322317, + "passes": true + }, + "pre.men.b1930_1934.pr_cross": { + "truth": 0.6005766012618865, + "filled": 0.6018662640540315, + "gap": 0.001289662792145041, + "tolerance": 0.1038645931448928, + "passes": true + }, + "pre.men.b1930_1934.pr_in": { + "truth": 0.6408121492387024, + "filled": 0.5924012929730078, + "gap": -0.048410856265694635, + "tolerance": 0.08763102065690465, + "passes": true + }, + "pre.men.b1930_1934.pzero": { + "truth": 0.23849477516714046, + "filled": 0.227113951116651, + "gap": -0.048895524206467256, + "tolerance": 0.1240674733098974, + "passes": true + }, + "pre.men.b1935_1939.aime_p25": { + "truth": 1099.5, + "filled": 1099.5, + "gap": 0.0, + "tolerance": 0.18180060035524045, + "passes": true + }, + "pre.men.b1935_1939.aime_p50": { + "truth": 2295.0, + "filled": 2289.0, + "gap": -0.002617802542078884, + "tolerance": 0.08710844064665221, + "passes": true + }, + "pre.men.b1935_1939.aime_p75": { + "truth": 3360.0, + "filled": 3340.75, + "gap": -0.005745641296078574, + "tolerance": 0.05420999979205089, + "passes": true + }, + "pre.men.b1935_1939.plevel": { + "truth": 0.47885857925887765, + "filled": 0.4689466025280649, + "gap": -0.020916404108078157, + "tolerance": 0.05092078761084689, + "passes": true + }, + "pre.men.b1935_1939.pr_cross": { + "truth": 0.5480805317128244, + "filled": 0.5304552392698022, + "gap": -0.017625292443022245, + "tolerance": 0.08418455817243553, + "passes": true + }, + "pre.men.b1935_1939.pr_in": { + "truth": 0.5082293220171369, + "filled": 0.4786688255277492, + "gap": -0.02956049648938769, + "tolerance": 0.0802550261947586, + "passes": true + }, + "pre.men.b1935_1939.pzero": { + "truth": 0.21315539421109053, + "filled": 0.22099454244225708, + "gap": 0.036116556377628006, + "tolerance": 0.12714526639770404, + "passes": true + }, + "pre.men.b1940_1945.aime_p25": { + "truth": 1178.0, + "filled": 1183.5, + "gap": 0.004658064742506518, + "tolerance": 0.12860148242782113, + "passes": true + }, + "pre.men.b1940_1945.aime_p50": { + "truth": 2795.0, + "filled": 2788.0, + "gap": -0.0025076137087847172, + "tolerance": 0.06640528987407995, + "passes": true + }, + "pre.men.b1940_1945.aime_p75": { + "truth": 4361.5, + "filled": 4355.25, + "gap": -0.001434020953004378, + "tolerance": 0.03759842149421006, + "passes": true + }, + "pre.men.b1940_1945.plevel": { + "truth": 0.3760217560197671, + "filled": 0.3808042784646001, + "gap": 0.012638534847186023, + "tolerance": 0.04070319570318265, + "passes": true + }, + "pre.men.b1940_1945.pr_cross": { + "truth": 0.376062044153939, + "filled": 0.40160036283466166, + "gap": 0.02553831868072265, + "tolerance": 0.061643783252697155, + "passes": true + }, + "pre.men.b1940_1945.pr_in": { + "truth": 0.36011521796896784, + "filled": 0.4197226882430919, + "gap": 0.05960747027412405, + "tolerance": 0.06659553115593378, + "passes": true + }, + "pre.men.b1940_1945.pzero": { + "truth": 0.21918652571223796, + "filled": 0.21480021501523025, + "gap": -0.020214719257387825, + "tolerance": 0.09675892268985861, + "passes": true + }, + "pre.men.b1946_1955.ylevel": { + "truth": 0.1479729717947154, + "filled": 0.14626786769864775, + "gap": -0.011589983129838721, + "tolerance": 0.025169112844575684, + "passes": true + }, + "pre.men.b1946_1955.yr_cross": { + "truth": 0.4294012571477206, + "filled": 0.45275418562739156, + "gap": 0.023352928479670965, + "tolerance": 0.03138849366607008, + "passes": true + }, + "pre.men.b1946_1955.yzero": { + "truth": 0.41447106660067734, + "filled": 0.41350548778712626, + "gap": -0.0023323830757379094, + "tolerance": 0.02207662859954437, + "passes": true + }, + "pre.men.b1956_1965.ylevel": { + "truth": 0.09639046670103465, + "filled": 0.09391389692509604, + "gap": -0.026028931247350506, + "tolerance": 0.02924452650871014, + "passes": true + }, + "pre.men.b1956_1965.yr_cross": { + "truth": 0.40595517510382484, + "filled": 0.4235749781242754, + "gap": 0.017619803020450575, + "tolerance": 0.032753653850861354, + "passes": true + }, + "pre.men.b1956_1965.yzero": { + "truth": 0.4166095146652036, + "filled": 0.420961639576618, + "gap": 0.01039234471895889, + "tolerance": 0.0238928171053413, + "passes": true + }, + "pre.men.b1966_1980.ylevel": { + "truth": 0.055118361796504756, + "filled": 0.05439748860982647, + "gap": -0.013164918055038832, + "tolerance": 0.02003852127655176, + "passes": true + }, + "pre.men.b1966_1980.yr_cross": { + "truth": 0.4093896749871457, + "filled": 0.42927884599997357, + "gap": 0.01988917101282789, + "tolerance": 0.025237052352190405, + "passes": true + }, + "pre.men.b1966_1980.yzero": { + "truth": 0.41574858870643483, + "filled": 0.4170007742872307, + "gap": 0.0030073551063288795, + "tolerance": 0.01612744451279897, + "passes": true + }, + "pre.women.b1930_1934.aime_p25": { + "truth": 192.0, + "filled": 210.5, + "gap": 0.09199028109465424, + "tolerance": 0.16540132553588516, + "passes": true + }, + "pre.women.b1930_1934.aime_p50": { + "truth": 526.0, + "filled": 522.0, + "gap": -0.0076336248550710195, + "tolerance": 0.11219306491273605, + "passes": true + }, + "pre.women.b1930_1934.aime_p75": { + "truth": 1055.0, + "filled": 1037.75, + "gap": -0.016485858977018708, + "tolerance": 0.09277530966675321, + "passes": true + }, + "pre.women.b1930_1934.plevel": { + "truth": 0.17987838861322633, + "filled": 0.17713733383955912, + "gap": -0.015355674621867044, + "tolerance": 0.09009967952937194, + "passes": true + }, + "pre.women.b1930_1934.pr_cross": { + "truth": 0.5714979296198577, + "filled": 0.581362041200494, + "gap": 0.00986411158063627, + "tolerance": 0.11564361594419356, + "passes": true + }, + "pre.women.b1930_1934.pr_in": { + "truth": 0.5486860492872828, + "filled": 0.5187328325728836, + "gap": -0.02995321671439921, + "tolerance": 0.13190265796761472, + "passes": true + }, + "pre.women.b1930_1934.pzero": { + "truth": 0.5514059876498446, + "filled": 0.5461903619700004, + "gap": -0.009503794247826214, + "tolerance": 0.04460593142917569, + "passes": true + }, + "pre.women.b1935_1939.aime_p25": { + "truth": 267.0, + "filled": 296.5, + "gap": 0.10479856003753074, + "tolerance": 0.19235462769111364, + "passes": true + }, + "pre.women.b1935_1939.aime_p50": { + "truth": 726.0, + "filled": 720.75, + "gap": -0.007257678306284099, + "tolerance": 0.12530279413386425, + "passes": true + }, + "pre.women.b1935_1939.aime_p75": { + "truth": 1448.0, + "filled": 1426.875, + "gap": -0.014696555661600996, + "tolerance": 0.09904500913528351, + "passes": true + }, + "pre.women.b1935_1939.plevel": { + "truth": 0.18459645468504016, + "filled": 0.1819401276084356, + "gap": -0.014494452724367557, + "tolerance": 0.08785419867161383, + "passes": true + }, + "pre.women.b1935_1939.pr_cross": { + "truth": 0.5177687037806673, + "filled": 0.5398891769889543, + "gap": 0.022120473208287028, + "tolerance": 0.12812634996138048, + "passes": true + }, + "pre.women.b1935_1939.pr_in": { + "truth": 0.48588537175161917, + "filled": 0.5137841122259348, + "gap": 0.02789874047431562, + "tolerance": 0.14213673818819197, + "passes": true + }, + "pre.women.b1935_1939.pzero": { + "truth": 0.5152445437812253, + "filled": 0.5120496976071522, + "gap": -0.006219944287538581, + "tolerance": 0.05005640550118318, + "passes": true + }, + "pre.women.b1940_1945.aime_p25": { + "truth": 386.0, + "filled": 412.0, + "gap": 0.06518597988469566, + "tolerance": 0.14632730902529825, + "passes": true + }, + "pre.women.b1940_1945.aime_p50": { + "truth": 1041.0, + "filled": 1037.5, + "gap": -0.003367816510116306, + "tolerance": 0.09638615659276974, + "passes": true + }, + "pre.women.b1940_1945.aime_p75": { + "truth": 2058.0, + "filled": 2035.5, + "gap": -0.010993148450961776, + "tolerance": 0.07111523531116512, + "passes": true + }, + "pre.women.b1940_1945.plevel": { + "truth": 0.19012308454067162, + "filled": 0.1945813540101376, + "gap": 0.023178672360682828, + "tolerance": 0.06751533282014863, + "passes": true + }, + "pre.women.b1940_1945.pr_cross": { + "truth": 0.3145537439002571, + "filled": 0.34602394584525564, + "gap": 0.031470201944998555, + "tolerance": 0.10423586286866415, + "passes": true + }, + "pre.women.b1940_1945.pr_in": { + "truth": 0.19634726483554435, + "filled": 0.19350804350369938, + "gap": -0.002839221331844971, + "tolerance": 0.10786950445828188, + "passes": true + }, + "pre.women.b1940_1945.pzero": { + "truth": 0.45477910425062734, + "filled": 0.42867842749600793, + "gap": -0.05910476431003997, + "tolerance": 0.05146845462920649, + "passes": false + }, + "pre.women.b1946_1955.ylevel": { + "truth": 0.09059452101103667, + "filled": 0.09094511842211886, + "gap": 0.0038624935915136938, + "tolerance": 0.02894095947229668, + "passes": true + }, + "pre.women.b1946_1955.yr_cross": { + "truth": 0.34158903474392094, + "filled": 0.36404211137367803, + "gap": 0.022453076629757096, + "tolerance": 0.03797384039389136, + "passes": true + }, + "pre.women.b1946_1955.yzero": { + "truth": 0.5347643649882455, + "filled": 0.5349335055491772, + "gap": 0.00031623987856743696, + "tolerance": 0.014748010374272846, + "passes": true + }, + "pre.women.b1956_1965.ylevel": { + "truth": 0.06347343380872424, + "filled": 0.0631225842636031, + "gap": -0.005542835370753618, + "tolerance": 0.02985282517131118, + "passes": true + }, + "pre.women.b1956_1965.yr_cross": { + "truth": 0.3538451043271927, + "filled": 0.38134198162535016, + "gap": 0.027496877298157474, + "tolerance": 0.030904887590406976, + "passes": true + }, + "pre.women.b1956_1965.yzero": { + "truth": 0.47511142887567126, + "filled": 0.47705249707166436, + "gap": 0.004077177956489653, + "tolerance": 0.01743364522031428, + "passes": true + }, + "pre.women.b1966_1980.ylevel": { + "truth": 0.04398552907354234, + "filled": 0.044065206484431366, + "gap": 0.0018098073146592952, + "tolerance": 0.01981382685946912, + "passes": true + }, + "pre.women.b1966_1980.yr_cross": { + "truth": 0.318767197977468, + "filled": 0.339060948106113, + "gap": 0.020293750128644983, + "tolerance": 0.02460765229653575, + "passes": true + }, + "pre.women.b1966_1980.yzero": { + "truth": 0.42453709903219916, + "filled": 0.42469779049242196, + "gap": 0.0003784382015451504, + "tolerance": 0.01544025098036175, + "passes": true + } + } + } + ], + "candidate_code": { + "path": "src/populace_dynamics/estimates/epuf_fill.py (untracked when scored; v1 is the state hashed here)", + "sha256_v1": "ae0f005f465048b62b5147d79632446b3e7792dc01a2187ea089c03db30f8c8e" + }, + "pre_chain": "fitted, never scored" +} diff --git a/docs/amendments/gate_epuf_fill_registration_proposal.md b/docs/amendments/gate_epuf_fill_registration_proposal.md new file mode 100644 index 00000000..0cd02705 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_registration_proposal.md @@ -0,0 +1,807 @@ +# gate_epuf_fill registration: career fills scored on held-out EPUF careers + +- **Registration id**: `2026-10-03-epuf-career-fill` +- **Gate**: `gate_epuf_fill` (registered, `locked: false`; not in + `gates.yaml` until the lock ceremony in section 14 completes) +- **Surface**: the two rules the career assembler + (`populace_dynamics.estimates.career.build_career`) uses to fill years the + PSID did not record, and their learned replacements, scored on held-out + persons of SSA's 2006 Earnings Public-Use File (EPUF). +- **Ceremony stage**: ROUND-3 FIXES, then lock on Max's ratification + (decision d927). The TEST scoring needs the candidates' module, which PR + #516 adds, so #516 merges before TEST is read. Referee round 3 + (`reviews/gate_epuf_fill_round3_confirmation_20261004.md`) returned LOCK + AFTER LISTED FIXES with no rebuild; its fixes are in section 12b. + Earlier stage: ROUND-2 FIXES. Referee round 1 + (`reviews/gate_epuf_fill_round1_referee_20261004.md`) returned AMEND + BEFORE LOCK, and amendment 1 answered it (section 12). Referee round 2 + (`reviews/gate_epuf_fill_round2_verification_20261004.md`) returned LOCK + AFTER LISTED FIXES; this document carries those fixes (section 12a). The + registered floor build is v3; its floors, tolerances and partition + equal v2's exactly. +- **What has been seen**: + - Candidate fills exist and have been scored on DEV. Every such score is + disclosed: those before amendment 1 in + `docs/amendments/gate_epuf_fill_dev_scores_before_amendment_1.json`, + those after it in + `docs/amendments/gate_epuf_fill_dev_scores_after_amendment_1.json`, + and those after round 2's fixes, through 2026-10-04 07:45 UTC, in + `docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl`. The + log stays open, and is closed with a timestamp in the lock record. + - The bytes of the candidate code behind each disclosed score are kept + in `docs/amendments/gate_epuf_fill_candidate_code/` from 07:45 UTC on. + Earlier versions were not kept. Their SHA-256 values are in the + disclosures, and their differences are described there. + - No TEST person has been read. TEST can be read only through + `epuf_fill_gate.test_part()`, which refuses until `gates.yaml` locks + this gate. +- **Code**: + - `src/populace_dynamics/harness/epuf_fill_gate.py` (the rules); + - `tests/harness/test_epuf_fill_gate.py`; + - `scripts/build_epuf_fill_psid_scale.py` (the PSID counts); + - `scripts/build_epuf_fill_gate_floors.py` (floors, bite, doses, + oracles); + - `src/populace_dynamics/harness/epuf_fill_scoring.py` (the one + registered TEST scoring, pinned to the v3 build's SHA-256); + - `tests/test_epuf_fill_gate_floors.py` (pins the builds). + +## What this gate is, in plain words + +A gate here is a pass-or-fail test whose rules and thresholds are fixed and +published before anything is scored against it. + +The PSID asked about income only every other year from 1997, and its +earnings panel starts in 1968. The career assembler fills those holes with two +fixed rules: + +1. each odd year is the mean of the two years around it; and +2. nothing counts before 1968, or before age 22, whichever is later. + +EPUF records capped taxable earnings in every year from 1951 to 2006 for a +1 percent sample of Social Security numbers. So we can hide the years the +PSID would be missing, fill them, and compare the filled careers with the +real ones. This gate does that on persons no fill has seen. + +- **Cell pass**: the fill moves the cell by less than the sampling error the + PSID itself carries there. +- **Gate pass**: the fill passes every cell that gates. + +**What a pass certifies, and what it does not.** +- It certifies a fill on EPUF's administrative, capped, disclosure-processed + earnings. +- In use, the years a fill conditions on are PSID survey reports of + uncapped labour income, converted to a share of the wage base. The gate + does not measure the gap between the two sources. PR #509's registration + reports that gap. + +Max approved the direction on 2026-10-03: "at a minimum seems like we could +use it to improve the odd-year filling, we're overstating that form of +persistence with our current method". + +## 1. Data + +- **Source**: EPUF 2006 (`populace_dynamics.data.epuf`, SHA-256-pinned, from + PR #509). + - 4,384,254 persons. + - Capped taxable earnings 1951-2006, after SSA's disclosure operator. + - Sex and year of birth; no family links, deaths or migration. + - Capped earnings are the AIME's concept. +- **Reading** (`epuf_fill_gate.epuf_matrix`): + - Every annual row must join one demographic person, once per year. + - TRAIN and DEV are read directly. TEST is read only through `test_part()` + after lock. +- **Persons**: sex coded 1 or 2. 3,054 persons with unspecified sex are + dropped. +- **Wage bases**: EPUF's own (`epuf_operator.wage_base`). +- **NAWI**: `cola_track_a.statutory.captured_ssa_parameters().nawi`. +- **PSID scale**: the default-spec PSID-2010 cohort + (`cohorts.psid2010.build_psid2010_cohort`), 11,405 members. Its counts are + in `runs/epuf_fill_gate_psid_scale_v2.json` (section 5). + +## 2. Split + +Each person's position `u` in [0, 1) is computed by +`epuf_fill_gate.split_part`: + +1. take SHA-256 of `populace_dynamics.epuf_fill.split.v1|` followed by the + decimal person id; +2. read the first eight bytes, big-endian; +3. divide by 2^64. + +| Part | Rule | Persons | Use | +|---|---|---:|---| +| TRAIN | u < 0.6 | 2,629,944 | Fills learn from it; exploration; dry runs | +| DEV | 0.6 <= u < 0.8 | 875,829 | Floors, checks on bite, doses, oracles; candidate development | +| TEST | u >= 0.8 | 878,481 | Read once, through `test_part()`, after lock | + +- The split was fixed before any EPUF statistic was computed in this work. +- The counts include persons of unspecified sex. + +## 3. Masks, and what a fill is given + +Two families of years, as the PSID would leave a career: + +- **`odd`**: the odd years 1997, 1999, 2001, 2003 and 2005. The assembler + also fills 2007-2011 and the 2013 seam, which lie past EPUF's last year. +- **`pre`**: every year from 1951 before `max(1968, birth_year + 22)`. + - For cohorts born 1930-1945 this is 1951-1967. + - For cohorts born 1946-1980 it is every year before age 22. + - EPUF has no earnings below age 15 for cohorts born after 1937. + +**The scoring path** (`epuf_fill_gate.score_candidate`): +1. Every fill is given shares of the wage base with **both families' cells + unknown** (NaN), together with each person's sex, year of birth and id. +2. It fills its own family's cells (`family_mask`). The odd family owns the + masked odd years inside the career. An odd year before the career + start belongs to the pre-career rule, which fills it in use (round 2, + finding 2). +3. Each draw must return finite shares in [0, 1] on those cells and leave + every other cell exactly as given; otherwise `FillOutputInvalid` is + raised. +4. A share of 1 becomes exactly the wage base. +5. The filled cells replace the truth's cells of the fill's own family, and + every other cell stays true. The other family's cells are therefore true + when a family is scored, but unknown to its fill. +6. The current rules are scored the same way, as fills (`current_rule`). + - `CurrentOddFill` takes the mean of the neighbouring years' earnings in + dollars, as a share of its own year's wage base, capped at 1. + - With one neighbour unknown it uses that neighbour; with both unknown, + zero. + - `CurrentPreFill` gives zero. + + **The age-22 case.** The scoring path hides pre-career years, so at age + 22, whose year before is pre-career, `CurrentOddFill` always takes the + one known neighbour. The assembler does that only when the PSID did not + record the year before. `build_career` takes any observed PSID year from + 1968 on as a neighbour, whatever the age. + - In the PSID-2010 cohort, 274 of the 513 age-22 neighbour-mean fills + (53 percent) have the age-21 year recorded. That is 0.8 percent of + its 65,588 filled years. + - So the current rule has two readings there: always falling back, and + always averaging the true year before (`current_odd_fill` on the + truth). + - For the "improves" tier, each cell's current gap is the smaller of the + two readings' gaps (section 7.3). + +## 4. Cells + +### 4.1 Groups and populations + +- Every cell belongs to one **floor group**, `..` + (`group_of`). There are 40 groups (`groups()`). +- `group_rows(group)` defines exactly the persons a group's cells count. + The cells and the floors (section 5) both take their persons from it. +- Populations read no cell their own family masks: + - `odd` universe: positive in a recorded even year 1996-2006. + - `odd_career` universe: positive in a year from + `max(1968, birth_year + 22)` through 2006 that is not a masked odd year. + - `pre` universe: positive in a year from `max(1968, birth_year + 22)` + through 2006. +- Within a population, some denominators read the masked year itself: + `atcap` (positive at `t`), the quantiles of positive shares, and every + "positive in both years" correlation. A fill that gets the zero rate + wrong therefore also moves those cells. + +### 4.2 Family `odd` + +**Age-band groups** (`odd..`): +- Bands: 22-29, 30-44, 45-59, 60-74, and 22-74, which pools the four. +- The assembler fills odd years only inside the career, so units start at + age 22. +- Population: `odd` universe members with a masked odd year at an age in + the band. +- Units: their person-years at a masked year `t` at an age in the band. + +| Statistic | Definition | Scale | +|---|---|---| +| `r1` | Spearman of `t` with `t-1` and with `t+1`, positive in both; mean of the pairs | gap | +| `r3` | Spearman of `t` with `t-3` and with `t+3` (recorded), positive in both; mean | gap | +| `r2`, `r4` | Spearman of `t` with `t+2` / `t+4` (both masked), positive in both; mean | gap | +| `zint` | Share zero at `t`, among units positive at `t-1` and `t+1` | log ratio | +| `zexit` | Share zero at `t`, among units positive at exactly one of them | log ratio | +| `wint` | Share positive at `t`, among units zero at both | log ratio | +| `atcap` | Share at the wage base, among units positive at `t` | log ratio | +| `level` | Mean share of the wage base at `t`, zeros included | log ratio | +| `q10`, `q50`, `q90` | Quantiles of the positive shares at `t` | log ratio | + +**Cohort groups** (population: `odd_career` universe members of the cohort): +- **Born 1936-1940, 1941-1945, and 1936-1945 pooled**: `aime_p10`, `p25`, + `p50`, `p75` and `p90`, the percentiles of the AIME under the 35-year rule + through age 61 (`aime_35`). A Hypothesis test pins `aime_35` to + `ss.statutory_aime.aime`. +- **Born 1946-1955, 1956-1965, 1966-1980, and 1946-1980 pooled**: the same + percentiles of the **partial AIME** (`partial_aime`). + - It is the AIME formula applied to every year through 2006, indexed to + NAWI in 2006. + - It is not a statutory AIME. It measures how a fill moves the AIME's + ingredients for cohorts EPUF cannot follow to 61. + +### 4.3 Family `pre` + +Population: `pre` universe members of the cohort. + +**Born 1930-1934, 1935-1939, 1940-1945, and 1930-1945 pooled** (masked +years 1951-1967): + +| Statistic | Definition | +|---|---| +| `aime_p10` to `aime_p90` | Percentiles of the AIME | +| `pzero` | Share of masked person-years at ages 18 and over with no earnings | +| `plevel` | Mean share of the wage base in those person-years | +| `pr_in` | Spearman of 1962 with 1967 (both masked) | +| `pr_cross` | Spearman of 1965 (masked) with 1970 (recorded) | + +**Born 1946-1955, 1956-1965, 1966-1980, and 1946-1980 pooled** (masked +years: ages 21 and under): + +| Statistic | Definition | +|---|---| +| `yzero` | Share of person-years at ages 15-21 with no earnings | +| `ylevel` | Mean share of the wage base at ages 15-21 | +| `yr_cross` | Spearman of earnings at age 21 with earnings at age 24 | +| `paime_p10` to `paime_p90` | Percentiles of the partial AIME | + +Every correlation is computed among persons positive in both years. + +**Pooled groups.** A fill whose bias is just inside the tolerance in every +cohort or band, all in the same direction, would bias a pooled PSID +estimate by more than the pooled estimate's own sampling error. The pooled +groups gate that bias directly, with floors at the pooled PSID size. + +There are 326 cells in all. + +## 5. Gap, floor and tolerance + +**Gap** (`gap`): +- For a correlation, the filled value minus the true value. +- For every other cell, the log of the filled value over the true value. +- A candidate's filled value is the mean of the cell over the 20 draw seeds + `DRAW_SEEDS = 7100..7119`. Its gap is the gap of that mean (`score`). + +**Floor** (`floor_group`): +- Each of `N_FLOOR_REPLICATES = 200` replicates draws two disjoint samples + of `n` persons from the group's DEV population (`group_rows`). It uses + the stream `default_rng([5000, group index, 200, n])`, where the group + index is the group's position in `groups()`. +- It computes every cell on both samples and records + `[m(A) - m(B)] / sqrt(2)` on the cell's scale. +- The cell's `sigma` is the root mean square of its replicates. For two + independent samples that estimates one sample's standard error: the + sampling error a PSID-sized sample carries in that cell. + +**Sample size `n`** (from `runs/epuf_fill_gate_psid_scale_v2.json`): +- **Age-band groups**: the number of PSID-2010 career years the assembler + filled with the neighbour mean at an age in the band, among members + positive in a recorded even year 1996-2010. This is divided by the mean + number of masked units per DEV population member in the band, so a sample + carries about as many masked units as the PSID fills. +- **Cohort groups**: the number of PSID-2010 members of that sex and cohort + who are positive in a recorded year from `max(1968, birth_year + 22)` + through 2010. +- The counts are unweighted. Several things make the floors narrower than + the PSID's real sampling error, and so make the gate stricter on + balance: + - the PSID's effective sample size is below its count; + - its persons are clustered in families; + - its filled-year count includes 2007-2011 and the 2013 seam. + + Matching units rather than persons also makes it stricter. In every + band, DEV has fewer masked units per person than the PSID-2010 cohort: + 2.45 against 2.70 for men 22-29, and 4.51 against 5.89 for men 22-74. + So a sample carries more persons than the PSID's, and its floor is + narrower. + +**Tolerance**: `tau = K * sigma` with `K = 1`. A gating cell passes if its +gap is finite and `|gap| <= tau`. + +**What K = 1 means.** The tolerance is a **materiality threshold**: +- Suppose a fill's bias in a cell equals one PSID standard error. Then the + root mean square error of a PSID-sized estimate built on the fill is at + most the square root of 2 times the PSID's own sampling error. +- A fill within the tolerance is never the larger source of error in that + cell. The pooled groups extend that to estimates pooled across cohorts or + bands. +- The comparison's own noise is much smaller than `tau` (see the operating + characteristic below). So this is not, in the sense other gates use, + "the noise of the comparison". +- Max's ratification of this reading is queued (section 14). + +**Operating characteristic**: +- The gap compares the truth and the fill on the same TEST persons, so it + carries no error from which persons were sampled. +- A fill that draws from the true conditional law has a gap whose standard + deviation is at most about `sqrt(1 + 1/20) * sqrt(n / N) * sigma`. Here + `N` is the group's population on a part, and the stored `noise_ratio` is + `sqrt(n / N)`. +- The verdict is therefore close to a step at `|bias| = sigma`. With some + 300 gating cells, a faithful fill fails one by chance with negligible + probability. A fill whose bias in a cell is near `sigma` can land on + either side of it. + +## 6. Which cells gate, and the checks before lock + +A cell gates (`partition`) if, on DEV, all of these hold: +1. its true value is finite, and positive for a log-ratio cell; +2. its floor `sigma` is finite and positive; +3. in at least 95 percent of floor replicates, the smaller sample of the + pair has at least 20 events. A share's events are the smaller of its + hits and misses; a correlation's are the pairs in its smallest pair-year; + a quantile's are the positive values behind it. + +Every other cell is reported without a verdict. The partition is fixed at +the floor build and recorded in the artifact. + +**Bite.** Before lock, on DEV, through the scoring path: +- **B1**: the current odd-year rule must fail at least one gating `odd` + cell by more than `2 * tau`. +- **B2**: the current pre-career rule must fail at least one gating `pre` + cell by more than `2 * tau`. + +If either check fails, the gate does not lock. + +**Dosed perturbations** (report-only). These are perturbations of the true +DEV matrix, and each moves only its own family's cells. Each is scored and +its gaps are reported in tolerances. The dosed ones also report the dose at +which the most sensitive cell reaches one and two tolerances, and the same +for the most sensitive AIME cell, each with the cell named. + +| Id | Family | Perturbation | What it tests | +|---|---|---|---| +| D1 | `odd` | Copy the share of the wage base at `t+1` into `t` | Copying a neighbour | +| D2 | `odd` | Permute `t` within sex and five-year age band | The marginal law | +| D3 | `odd` | Shrink positive log shares toward their stratum median by `lambda` = 0.75 and 0.5 | Dispersion | +| D4 | `pre` | Scale the true blocks by 0.95 and 0.90 | Level bias in the AIME and `plevel` | +| D5 | `pre` | Permute whole blocks within sex and birth year only | Matching | +| D6 | `pre` | Permute each masked year independently within sex and birth year | Persistence | + +**Oracles** (report-only). Two fills permute true values among similar DEV +persons, over `ORACLE_SEEDS = 7200..7219`. Each conditions only on what a +fill is given: +- **O1** (`odd_oracle_fill`) permutes the true value of each masked odd year + inside the career among persons who share all of: sex; `odd` universe + membership; five-year age band at `t`; and the bins of `t-1` and `t+1`. A + neighbour inside the pre-career mask counts as unknown. +- **O2** (`pre_career_oracle_fill`) permutes whole masked blocks among + persons who share all of: sex; birth year; `pre` universe membership; the + number of positive years among the first five recorded years (a masked + odd year is unknown); and the quintile of their mean share. + +## 7. Candidates + +Each family has a primary candidate and a registered alternative. +- All four are fitted on TRAIN only and developed against DEV. +- Each is registered with its code commit and the SHA-256 of its fitted + artifact before TEST is read. **The registered commit and SHA-256 + govern.** The descriptions below are summaries. +- Every candidate produces shares in [0, 1] through the scoring path of + section 3. Its draws come from counter-based uniforms keyed by the fill, + the draw seed, the person id and the year, so a person's draw does not + depend on which other persons are filled. + +### 7.1 Odd years + +**Primary: a two-part conditional draw (QRF-style).** +- It draws the share of the wage base at `t`: a probability of a zero year, + then conditional quantiles of a positive share. +- It conditions on the recorded shares around `t` (`t-1`, `t+1`, `t-3`, + `t+3` and further where recorded), sex and age. +- A copula across a person's masked years carries the dependence the + conditioning leaves. + +**Alternative: kNN triples.** It draws the share at `t` from one of the `k` +nearest TRAIN person-years in the shares at `t-1` and `t+1`, sex and age. + +### 7.2 Pre-career years + +**Primary: rank-kNN donor careers.** +- **Donor pool**: TRAIN persons of the same sex and birth year who are in + the `pre` universe. +- **Match vector**: a person's percentile rank within the donor pool in the + first five years from their career start that they have recorded. +- **Draw**: one of the `k` nearest donors is chosen by the seeded uniform, + and the whole masked block is copied from that donor. +- **Precedent**: gate 1's passing candidate was rank-kNN + (`runs/gate1_rank_knn_v5.json`: 10-year autocorrelation 0.499-0.533 + against a reference of 0.539). + +**Alternative: a chained one-sided fill.** It draws year `y` from year +`y+1`, sex and age, working backward from the career start. +- Gate 1's chained weighted QRF baseline failed long persistence + (`runs/gate1_qrf_baseline_v1.json`: 10-year autocorrelation 0.309-0.368 + against 0.539, tolerance 0.07). +- This alternative is registered so that the choice of borrowing whole + careers is tested rather than assumed. + +### 7.3 Adoption rule + +This rule is coded in `adoption_tier` and `adopt`, applied by +`epuf_fill_scoring.score_registered`. Each candidate and the current rule +are scored on the same TEST persons. + +**The current gap.** For the odd family the current gap is the smaller, +cell by cell, of its two readings' gaps (section 3, item 6), and its +failing count is the smaller of the two. Both readings' scores are +reported. A gating cell whose TEST truth is undefined (not finite, or not +positive for a log ratio) is reported and dropped from every score. + +A candidate's tier is: +- **certified** if it passes the gate; +- **improves** if it fails the gate but meets both of these: + - in every gating cell its gap is finite and at most + `max(tau, min(|current gap|, 3 * tau))`, where a non-finite current gap + counts as infinite; + - it fails strictly fewer gating cells than the current rule; +- **not adopted** otherwise. + +In each family, the candidate in the better tier is adopted, and the +primary is adopted on a tie. If both are not adopted, the current rule +stays. +- An adopted candidate in the **improves** tier is recorded as an + uncertified improvement, and no text may call it certified. +- The cap of three tolerances (amendment 1) means "improves" requires a + fill to be near the tolerance everywhere, not merely finite where the + current rule is not. + +## 8. Implementation behind the rule + +- `estimates/career.py` stays byte-identical. The first-estimates + birth-evidence reducer seals every file under `src/` against its reviewed + commit (`scripts/first_estimates_birth_evidence.py`, + `_assert_input_identity`), and `career.py` is reachable from it. +- The gate's rules module (`harness/epuf_fill_gate.py`) and its TEST + scoring module (`harness/epuf_fill_scoring.py`) are listed in the + reducer's `POST_REVIEW_SOURCE_EXCLUSIONS`, and a test proves both are + unreachable from the reducer. +- The candidate fills will live in a new opt-in module under the same + exclusion. Its provenance values are `gap_epuf_drawn` and + `pre_career_epuf_donor`, so a year's provenance always names the rule + that produced it. +- `cohorts/psid2010.py`, which is outside the seal, gains a spec field for + the fill. Its default stays the current rule until a new registration of + the DYNASIM projections adopts the learned fills (section 11). + +## 9. Evidence before the first floor build + +**TRAIN exploration** of the current rules and of conditional-permutation +oracles. None of it fitted a candidate, built a DEV floor or read a TEST +person. + +The current odd-year rule: +- raises `r1` by 0.047-0.103 and `r2` by 0.068-0.157; +- removes every zero year where a career starts or stops (`zexit`: true + 0.39-0.52, filled 0); +- removes every zero year between two working years (`zint`: true + 0.016-0.037, filled 0); +- moves the median AIME of the 1936-1945 cohorts by 0.2 percent or less. + +The conditional-permutation oracles: +- One conditioning on `t-1` and `t+1` only understates `r2` by 0.006-0.032. +- Adding `t-3` and `t+3` cuts that to 0.001-0.015. + +On DEV, the current pre-career rule lowers the pre-universe median AIME +(log gaps from the v1 artifact, converted to percent): + +| Cohort | Men | Women | +|---|---:|---:| +| 1930-1934 | 32% | 34% | +| 1935-1939 | 20% | 25% | +| 1940-1945 | 8% | 13% | + +Zeroing only the years before age 22 lowers median AIME by 2-3 percent for +men and 7-11 percent for women born 1930-1945 (TRAIN, all persons of the +cohort). The PSID-2010 cohort's members born 1946 or later lose those years +under the current rule. That is why family `pre` covers every year before +`max(1968, birth_year + 22)`, not only 1951-1967. + +## 10. The first build (v1), superseded + +`runs/epuf_fill_gate_floors_v1.json` (built once on DEV at `d21aa659`) is +frozen. Referee round 1 found that four of its groups were priced on the +wrong population, among other findings (section 12). +- Under v1's rules, 130 of 136 cells gated. +- B1 failed 48 of 70 gating `odd` cells by more than two tolerances, and B2 + failed 54 of 60 gating `pre` cells. +- O1 failed 16 cells: seven of eight `r2` cells, six `r4` cells, and three + young-band `wint`/`zexit` cells. +- O2 failed 4 `yr_cross` cells, each by under 1.2 tolerances. +- `tests/test_epuf_fill_gate_floors.py` keeps v1 internally consistent. + +### 10a. Results of the registered (v3) build + +**Builds.** +- The amended build ran once on DEV at `23bee82c` (v2). +- After referee round 2's fixes, the registered build ran once at + `cb76ad15` (v3, `runs/epuf_fill_gate_floors_v3.json`, SHA-256 + `d403a824...`), with its bound files clean. +- The v3 build's truth, floors, tolerances and partition equal v2's + exactly. Round 2's fixes changed how fills are scored, not the floors. + `tests/test_epuf_fill_gate_floors.py` checks that equality. +- The numbers below are v3's. + +**Partition**: 319 of 326 cells gate. Seven are report-only: +- too few events: `wint` for men 22-29, men 60-74 and women 60-74; + `atcap` for women 22-29 and 60-74; `zint` for women 60-74; +- undefined floor: `odd.men.a45_59.q90`. Men's 90th-percentile positive + share is the cap in every sample. + +**Men's `q90` sits at the cap in three more bands** (22-29 excluded): 22-74, +30-44 and 60-74. +- There `q90` is exactly 1 while more than 10 percent of positive shares + are at the cap. +- Its floor counts how often a sample's at-cap share crosses 10 percent. +- So in those bands `q90` gates only as a lower bound on the at-cap share + near 10 percent, much as `atcap` does. No cell measures men's upper-tail + dispersion. +- Future registrations should compute `q90` among positive shares below + the cap. + +**Tolerances** (`K * sigma`) range by statistic: + +| Statistic | Tolerance | Statistic | Tolerance | +|---|---|---|---| +| `r1` | 0.0025-0.0098 (0.00248 at the lower bound) | `aime_p50` | 0.042-0.125 (log) | +| `r2` | 0.0046-0.0222 | `paime_p50` | 0.021-0.043 (log) | +| `r3` | 0.0047-0.0206 | `pzero` | 0.029-0.128 (log) | +| `r4` | 0.0070-0.0382 | `plevel` | 0.029-0.094 (log) | +| `zint` | 0.045-0.178 (log) | `pr_in` | 0.041-0.133 | +| `zexit` | 0.016-0.052 (log) | `pr_cross` | 0.044-0.122 | +| `wint` | 0.071-0.150 (log) | `yzero` | 0.009-0.023 (log) | +| `level` | 0.011-0.052 (log) | `ylevel` | 0.015-0.030 (log) | +| `q50` | 0.012-0.074 (log) | `yr_cross` | 0.015-0.037 | + +The noise ratio `sqrt(n / N)` runs from 0.087 to 0.212. + +**Bite: both checks hold, so the gate can lock.** Both are scored as fills +through the scoring path. +- **B1**, the current odd-year rule, fails 96 of 183 gating `odd` cells by + more than two tolerances. +- **B2**, the current pre-career rule, fails 112 of 136 gating `pre` cells + by more than two tolerances. +- On the pooled 1930-1945 cohorts, B2 lowers the median AIME by 20 percent + for men (5.3 tolerances) and 22 percent for women (4.0 tolerances). +- On the single cohorts the falls are: + +| Cohort | Men | Women | +|---|---:|---:| +| 1930-1934 | 32% | 34% | +| 1935-1939 | 20% | 25% | +| 1940-1945 | 8% | 13% | + +**Dosed perturbations** (report-only; gating cells failed, and failed by +more than two tolerances): + +| Perturbation | Failed | Beyond 2 tol. | +|---|---:|---:| +| D1 copy the share at `t+1` | 63 / 183 | 47 | +| D2 marginal draws | 127 / 183 | 116 | +| D3 shrink, lambda 0.75 | 55 / 183 | 37 | +| D3 shrink, lambda 0.5 | 81 / 183 | 62 | +| D4 scale pre blocks by 0.95 | 12 / 136 | 5 | +| D4 scale pre blocks by 0.90 | 16 / 136 | 12 | +| D5 blocks permuted within sex and birth year | 75 / 136 | 54 | +| D6 pre years permuted independently | 83 / 136 | 61 | + +- **D3's dose.** One tolerance is reached at `lambda` = 0.996, and the cell + is `odd.men.a22_74.q90`, the cap-bound cell above. The most sensitive + AIME cell, `odd.men.b1966_1980.paime_p90`, needs a shrink of 12.5 + percent. +- **D4's dose and the AIME.** + - A pre-career level bias of 1.4 percent moves `pre.men.b1946_1980.ylevel` + one tolerance. That is the scale factor seen directly. + - The AIME cells cannot see such a bias. Scaling every pre-career block + by 0.90 moves the pooled 1930-1945 median AIME for men by 2.2 percent + (0.53 tolerances), and no 1930-1945 AIME cell moves more than 0.61 + tolerances. + - The most sensitive AIME cell, `pre.women.b1966_1980.paime_p25`, needs a + 13.5 percent bias. + - `plevel` and `ylevel` see a pre-career level bias; the AIME cells see + only large ones. + +**Oracles** (report-only): +- **O1** fails 32 of 183 gating `odd` cells: 8 `r2`, 10 `r4`, 9 `r3`, 3 + `r1`, 1 `wint` and 1 `zint`. Conditioning on `t-1` and `t+1` alone misses + multi-year persistence. +- **O2** fails 6 of 136 gating `pre` cells: 5 `yr_cross` and 1 `aime_p10`. + +## 11. What changes downstream + +- The DYNASIM projection comparisons (registered one-shot benchmarks) build + their cohort with `cohorts.psid2010`, so their AIMEs inherit the fill. + Adopting a learned fill for them needs a new registration and a new run, + not a silent rerun. That decision is queued for Max. +- The first-estimates path (`build_career_inclusion`) is sealed historical + evidence and keeps the current rule. + +## 12. Amendment 1: referee round 1 and the response + +Round 1 (`reviews/gate_epuf_fill_round1_referee_20261004.md`, an +independent Opus 5.5 lane) returned AMEND BEFORE LOCK. + +Before amending, every DEV score of a candidate produced so far was +committed: `docs/amendments/gate_epuf_fill_dev_scores_before_amendment_1.json`. +Five scoring events: +- an odd-year primary at 29 and then 26 of 70 cells failing; +- the kNN alternative at 24 of 70; +- the donor primary at 1 of 60, after a first run that stopped on unfilled + cells. + +Every amendment below is justified on the truth side. Two changes did +loosen cells, both forced by round 1's findings and the pre-set events +rule, and neither rescues a candidate: +- Men's young-band `wint` gated in v1 (as 18-29), where `odd_knn v0` + failed it by 4.1 tolerances. With units from age 22 it has too few + events and is report-only. The pooled `a22_74` `wint` still gates. +- The odd family's AIME tolerances widened once their floors were priced + on the right population (finding 1). For men born 1936-1940 the median's + tolerance went from 0.051 to 0.082. The disclosed candidates' odd AIME + gaps were at most 0.004. + +| # | Finding | Response | +|---|---|---| +| 1 | Odd AIME floors priced on the wrong population | `group_rows` defines each group's population once; cells and floors both use it; a test checks every group's count | +| 2 | The 18-29 band scored ages 18-21, which the assembler never fills with the mean | Odd units start at 22; the band is `a22_29` | +| 3 | Same-direction bias across cells unchecked | Pooled groups (bands 22-74; cohorts 1936-45, 1930-45, 1946-80) at pooled PSID size; AIME p10 and p90 | +| 4 | The pre fill read true odd years | Every fill is given the union of both masks as unknown (section 3) | +| 5 | No registered scoring path or validity checks | `score_candidate` with `FillOutputInvalid`; TEST only through `test_part()` after lock | +| 6 | "Improves" nearly vacuous where the current gap is infinite | The allowance is capped at three tolerances | +| 7 | Candidates existed contrary to the record | Disclosed; this header corrected; section 7 defers to the registered commit and SHA-256 | +| 8 | Bite only through degenerate failures | Dosed perturbations D1-D6, reported with doses at one and two tolerances | +| 9 | Unmeasured modes | `r3` (masked to recorded at ±3); positive-share quantiles `q10`/`q50`/`q90`; partial AIME for cohorts born 1946-1980 | +| 10 | Claims | Universes and denominators reworded (section 4.1); AIME table in percent from DEV (section 9); infinite and NaN gaps distinguished; scope sentence added | +| 11 | "Conservative" sample size | Restated as stricter on balance (section 5) | +| 12 | Tolerance as materiality; house formula on full-DEV halves | Recorded in sections 5 and 13; Max's ratification queued | +| 13 | `epuf_matrix` join and `atcap` exactness | Join asserted; a share of 1 becomes exactly the wage base | +| 14 | Dry-run provenance | `runs/epuf_fill_gate_floors_train_dryrun_v0.json` committed. It was built from the uncommitted working tree when HEAD was `61ade83a`, an ancestor of the rules commit `14045be4`, before universe membership in the oracle strata and the improves tier were added | +| 15 | Tests | Added: each group counts exactly its rows; populations invariant to their family's mask; units from age 22; scoring-path validity and union-mask inputs; refused TEST reads; the EPUF join | + +**Forks ledger** (rules changed after a floor build or a candidate score): + +| Change | Seen before it | Why it is not a self-rescue | +|---|---|---| +| Oracle strata gain universe membership | TRAIN dry run | Oracles are report-only | +| "Improves" tier added | TRAIN dry run (an oracle failing `r2`/`r4`) | Adoption only; no certification changes | +| Amendment 1 (all of the above) | v1 DEV floors; candidate DEV scores (disclosed) | Every change comes from referee round 1 and is truth-side. It adds cells and pooled groups, tightens adoption, and removes units the PSID does not fill | +| Amendment 1's rules, dry-run on TRAIN | A TRAIN dry run (`runs/epuf_fill_gate_floors_train_dryrun_v2pre.json`) | See the next paragraph | +| Round 2's fixes (section 12a) | v2 DEV floors; candidate DEV scores after amendment 1 (disclosed) | Every fix comes from referee round 2. None changes a floor, tolerance or partition; the v3 build reproduces v2's exactly | + +**The amendment-1 dry run.** +- The run started at 05:48:03 UTC (5 replicates, one oracle seed) from the + uncommitted working tree, when HEAD was `f91d43a1`. That was 16 minutes + before the rules commit `7061200d` at 06:04:18. +- Its cell ids, groups, constants (bar the replicate count) and output + schema are identical to the v2 build's. Between it and `7061200d`, the + tests and this document changed; no change to the rules module or the + builders is recorded. +- It showed the gate lockable, and `odd.men.a45_59.q90` with an undefined + floor. +- No candidate DEV score existed under amendment 1's rules until after + `7061200d`. The first was at about 06:19 UTC + (`docs/amendments/gate_epuf_fill_dev_scores_after_amendment_1.json`). + +### 12a. Referee round 2 and the fixes + +Round 2 (`reviews/gate_epuf_fill_round2_verification_20261004.md`) +verified that amendment 1 fixes round 1's findings 1-4, 6, 9-12 and 13. It +found 5, 7, 8, 14 and 15 partly fixed or needing more, and it raised +these: + +| # | Finding | Fix | +|---|---|---| +| 1 | Men's `q90` sits at the wage base in three bands | Disclosed in section 10a: there `q90` gates only as a lower bound on the at-cap share near 10 percent, and no cell measures men's upper-tail dispersion. Future registrations should compute `q90` below the cap | +| 2 | The odd fill was scored on pre-career odd years it never fills | The odd family owns only odd years inside the career (`family_mask`); the current rules, oracle and perturbations follow. Floors unchanged | +| 3 | Undisclosed TRAIN dry run; unlogged DEV scoring | The dry run is committed and in the forks ledger. Every DEV score after `7061200d` is disclosed | +| 4 | No pinned TEST entry point | `epuf_fill_scoring.score_registered`, pinned to the v3 build's SHA-256, reads TEST only through `test_part`, scores the current rule and both candidates, and returns the tiers | +| 5 | Claims the artifact contradicts | D3's and D4's dose cells are now named in the artifact, with the AIME doses (section 10a). The sample-size sentence is corrected (section 5), as are the "none loosens" sentence (section 12) and `current_odd_fill`'s docstring | +| 6 | D1 copied capped dollars across years | D1 copies the share of the wage base | +| 7 | Tests and provenance | Added: pool equals the cells' count for every cohort group; `epuf_cells.py` and `epuf_operator.py` bound to the build; the scored matrix keeps every non-owned cell true; writes into the other family's cells are refused; no unit at age 21; the annual year range is asserted; the build records whether the bound files were clean | +| 8 | Notes | Recorded in section 13 | + +### 12b. Referee round 3 and the fixes + +Round 3 (`reviews/gate_epuf_fill_round3_confirmation_20261004.md`) +confirmed round 2's eight findings fixed or disclosed, and confirmed that +v3's floors equal v2's. It returned LOCK AFTER LISTED FIXES, with no +rebuild: + +| # | Finding | Fix | +|---|---|---| +| 1 | Making the current odd rule a fill moved its young-band gaps, and so the improves allowances | Disclosed below. The neutral rule of sections 3 and 7.3 is registered: each cell's current gap is the smaller of the two readings' | +| 2 | The DEV log was not closed | `pre_chain3`'s check is logged. The log runs through 07:45 UTC and is closed at lock. Candidate code bytes are kept from 07:45 UTC | +| 3 | Gaps in the TEST entry point | `score_registered` now does all of these: builds the wage bases and NAWI itself and records their hash; drops and reports a gating cell whose TEST truth is undefined; loads each candidate through `load_fill` with its registered SHA-256; and scores both readings of the current rule. The lock record pins `epuf_fill_scoring.py`'s SHA-256 | +| 4 | Note for d927: "improves" allows up to three tolerances | Recorded in section 13 and in decision d927 | + +**What finding 1 disclosed.** Between v2 and v3 the current odd rule's +gaps moved in the 22-29 band. The two cells that loosened were +`odd.women.a22_29.level` (0.11 to 1.23 tolerances) and `odd.men.a22_29.level` +(0.31 to 1.09). Four cells tightened: men's 22-29 `q10`, `q50`, `q90` and +`r3`. +- The change was made at 07:16 UTC. By then, DEV event 8 (07:05 UTC) had + shown `odd_qrf_sex2` failing `odd.women.a22_29.level` by 1.10 tolerances, + where v2's allowance blocked "improves" and v3's would not. +- The neutral rule restores v2's allowance on those cells, because v2's + reading had the smaller gap there. It keeps the four tightenings. + +## 13. Considered and rejected + +1. **Floors from two halves of all DEV persons, and the house tolerance + `round(mean |e| + 4 sd |e|, 3)` on them.** + - With same-person scoring, the house formula on full-DEV halves gives a + tolerance of roughly `4.5 * sqrt(2n / N) * sigma`, about 0.55-1.35 + sigma here. That is the same order as `K = 1`, and it is tied to the + comparison's real noise. + - The registered tolerance instead prices materiality at the PSID's + size, which matches the house's deployment-scale floor (`gates.yaml` + noise-floor notes). + - The two readings agree in order. This registration takes the + materiality reading and asks Max to ratify it. +2. **Halves of all DEV persons with `K = 1`.** Each half would have about + 440,000 persons. The gate would demand a fitted model's bias fall below + noise no PSID-based estimate can perceive. +3. **Scoring the fill by its AIME alone.** On EPUF the current odd-year + rule moves the median AIME by 0.2 percent or less while distorting + persistence and zero years badly. +4. **Masking only 1951-1967.** It would leave untested the age-22 part of + the rule, which lowers AIME for every PSID-2010 member born after 1945. +5. **Weighting the PSID counts.** Effective sample sizes would loosen the + floors. +6. **A sign rule over cells** (the mean standardized gap over a + statistic's cells within `1/sqrt(k)`). Pooled groups do the same job + with ordinary cells and floors. + +**Known nits in bound files, left as they are.** Editing these files would +unbind the registered build, so both are recorded here instead (code review +of PR #515): +- `test_part` raises `AttributeError` rather than `TestPartLocked` on a + malformed `gates.yaml`. TEST stays unread either way. +- The builder counts the dosed perturbations' "beyond two" cells with a + literal 2, equal to `BITE_MULTIPLE`. + +**Notes from round 2, for Max's ratification and for future registrations.** +- At the boundary, a fill adds up to 41 percent (the square root of 2) to + the root mean square error of a PSID-sized estimate. +- Single-cohort AIME tail tolerances are wide. For example, + `pre.women.b1930_1934.aime_p10` is 0.391 log, so a 48 percent + overstatement passes it. The pooled groups carry the AIME's materiality: + their p10 tolerances are about 0.17. +- Estimates pooled across both sexes are not gated directly. A bias of 0.9 + sigma in the same direction in both sexes' pooled cells is about 1.3 + sigma on a both-sex estimate. +- Floor seeds depend on a group's position in `groups()`, so adding groups + re-rolls every floor. Amendment 1 moved tolerances by up to about 7 + percent this way, and no disclosed result flipped. Future amendments + should key seeds by group name. +- "Improves" allows a candidate to miss a cell by up to three tolerances. + At that boundary a fill adds up to the square root of 10 (about 3.2 + times) the PSID's own sampling error to a PSID-sized estimate's root mean + square error, against the square root of 2 for a certified fill. + - On DEV, the leading pre-career candidate (`pre_donor`, 07:22 UTC) was + in that tier, missing the pooled `pre.women.b1930_1945.aime_p10` by + +0.33 log (2.0 tolerances). + - Its later version (`pre_donor2`) fixed that cell. + - Section 7.3 says an "improves" adoption is uncertified. +- Pooled floors sample EPUF's cohort mix, not the PSID's. For men born + 1930-1945 that is 28/29/43 percent across the three cohorts, against the + PSID's 18/26/55. This is harmless. + +## 14. Ceremony checklist + +- [x] Rules pushed before any DEV floor (`14045be4`) +- [x] PSID scale counts v1 and floor build v1 on DEV (superseded) +- [x] Adversarial referee round 1: AMEND BEFORE LOCK +- [x] DEV candidate scores disclosed before amending +- [x] Amendment 1 rules (this document, the code and tests) pushed before + the v2 floor build +- [x] PSID scale counts v2 and floor build v2 on DEV (`23bee82c`); + lockable +- [x] Referee round 2 (verification): LOCK AFTER LISTED FIXES +- [x] Round 2's fixes; DEV candidate scores after amendment 1 disclosed +- [x] Floor build v3 on DEV at `cb76ad15` (round 2's fixes; floors equal + to v2's); the TEST scoring module pins its SHA-256 +- [x] Referee round 3 (confirmation): LOCK AFTER LISTED FIXES, no rebuild +- [x] Round 3's fixes (section 12b) +- [ ] Max ratifies the materiality reading of the tolerance (queued + decision) +- [ ] Ratifying merge; lock flip in `gates.yaml` +- [ ] Candidates registered (code commit, fitted-artifact SHA-256) +- [ ] TEST read through `test_part()` and scored once; result published + whether it passes or fails diff --git a/docs/design/gate_epuf_fill_block_draft.yaml b/docs/design/gate_epuf_fill_block_draft.yaml new file mode 100644 index 00000000..eca137dd --- /dev/null +++ b/docs/design/gate_epuf_fill_block_draft.yaml @@ -0,0 +1,63 @@ +# Draft gates.yaml block for gate_epuf_fill (NOT in gates.yaml yet). +# +# A gate here is a pass-or-fail test whose rules and thresholds are fixed and +# published before anything is scored against it. This block is copied into +# gates.yaml, with locked: true, only after Max ratifies the materiality +# reading of the tolerance (decision d927) and the ratifying merge lands. +# Until then epuf_fill_gate.test_part() refuses to read TEST. +gate_epuf_fill: + id: epuf_career_fill + registration_id: 2026-10-03-epuf-career-fill + status: registered_awaiting_ratification + locked: false + covers: >- + The career assembler's two fill rules (odd income years 1997-2011 and + the 2013 seam as neighbour means; nothing before max(1968, birth year + + 22)) and their learned replacements, scored on held-out persons of SSA's + 2006 Earnings Public-Use File: one-year and multi-year rank persistence, + zero-year rates by age, positive-share quantiles, and the AIME (and a + partial AIME through 2006) under the statutory 35-year rule. + holdout_basis: [EPUF2006_TEST_by_salted_person_hash] + proposal: docs/amendments/gate_epuf_fill_registration_proposal.md + rules: src/populace_dynamics/harness/epuf_fill_gate.py + rules_commit: 7061200d # amendment 1; round-2 fixes at cb76ad15 + floor_run: runs/epuf_fill_gate_floors_v3.json + floor_run_sha256: d403a824416f00524fadceefb897f5bdcaa197c12ebee0b1f1fd52304d98e25e + floor_build_commit: cb76ad15 + psid_scale: runs/epuf_fill_gate_psid_scale_v2.json + psid_scale_sha256: b35e12fae91aa4c219f719ce64b139962ea3bdba3cd322ff55a8a7b30abcdff1 + test_scoring: src/populace_dynamics/harness/epuf_fill_scoring.py + test_scoring_sha256: e21750be4e1d37d19275966546ec938a6a6979d863bee132cdfdcc18cae18ebb + thresholds: + locked: false + k_tolerance: 1.0 + tolerance: >- + Each gating cell's tolerance is K times its sampling standard error + at the PSID-2010 cohort's size (the root mean square of [m(A) - + m(B)] / sqrt(2) over 200 pairs of disjoint real DEV samples), read + from the floor run and never rewritten. A materiality threshold: a + fill within it is never the larger source of error in that cell. + gating_cells: 319 + report_only_cells: 7 + bite: >- + B1 (current odd-year rule) fails 96 of 183 gating odd cells and B2 + (current pre-career rule) 112 of 136 gating pre cells by more than two + tolerances on DEV. + adoption: >- + certified (passes every gating cell) > improves (fails, but every + gating cell within max(tau, min(|current gap|, 3 tau)), the current + gap being the smaller of the odd rule's two readings, and strictly + fewer failing cells) > not adopted; the primary wins ties; families + adopt separately. + ceremony_record: + round_1_referee: reviews/gate_epuf_fill_round1_referee_20261004.md (AMEND BEFORE LOCK) + amendment_1: docs/amendments/gate_epuf_fill_registration_proposal.md section 12 + round_2_verification: reviews/gate_epuf_fill_round2_verification_20261004.md (LOCK AFTER LISTED FIXES) + round_2_fixes: section 12a + round_3_confirmation: reviews/gate_epuf_fill_round3_confirmation_20261004.md (LOCK AFTER LISTED FIXES, no rebuild) + round_3_fixes: section 12b + ratification: pending decision d927 (Max) + dev_disclosures: + - docs/amendments/gate_epuf_fill_dev_scores_before_amendment_1.json + - docs/amendments/gate_epuf_fill_dev_scores_after_amendment_1.json + - docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl diff --git a/reviews/gate_epuf_fill_pr515_code_review_20261004.md b/reviews/gate_epuf_fill_pr515_code_review_20261004.md new file mode 100644 index 00000000..44e2a3bf --- /dev/null +++ b/reviews/gate_epuf_fill_pr515_code_review_20261004.md @@ -0,0 +1,122 @@ +**Verdict: APPROVE WITH NITS.** This is provisional: several checks you asked for need `git` and `pytest`, and this session only had file read, search and web fetch tools. Those checks are listed under "Not verified" with the exact commands. + +I found no correctness bugs in the gate rules, the scripts or the TEST guard. The proposal's v3 numbers match the artifact. The findings are about provenance and test strength. + +## Findings + +1. **Medium: a run with injected inputs produces the same record as the registered run.** `src/populace_dynamics/harness/epuf_fill_scoring.py:181-182, 197-223` + - **Evidence:** + - When `fills=` is passed, the loader is skipped (`:198-200`), so no candidate's SHA-256 is checked. The record still writes each spec's `path` and `sha256` under `"candidates"` as if they had been loaded. + - When `matrix=` is passed, `test_part` and the lock check are skipped. The record still carries `registration_id`. + - Nothing in the output says either input was injected. A test-style run cannot be told apart from the one registered TEST scoring. + - `epuf_fill.load_fill(..., sha256=None)` also skips its hash check silently (`estimates/epuf_fill.py:1657`), so a spec with a `None` hash goes through unverified. + - **Fix:** + - Add `"injected": {"matrix": matrix_given, "fills": fills_given}` to the record, or leave out `registration_id` when either is set. + - Refuse a spec whose SHA-256 is `None` or not 64 hex characters. + - This should land before the TEST run; it does not need to block the merge. + +2. **Low: #515's TEST entry point imports a module that, by every sign, first appears in #516.** `epuf_fill_scoring.py:201` + - **Evidence:** + - `populace_dynamics.estimates.epuf_fill` is imported lazily. + - The proposal (§8) says the candidate fills "will live in a new opt-in module". + - Its exclusion lines (`scripts/first_estimates_birth_evidence.py:358-361`) sit in a separate block whose comment matches #516's commit message. + - `test_registered_candidates_are_hash_checked` uses `importorskip`, so on #515 alone that test is skipped. + - **Effect:** at `70553e70`, calling `score_registered` without `fills` raises `ModuleNotFoundError`. + - **Fix:** say in #515's description that #516 must merge before the TEST run. Confirm with `git show 70553e70:src/populace_dynamics/estimates/epuf_fill.py`. + +3. **Low: `test_registered_candidates_are_hash_checked` never reaches a hash check.** `tests/harness/test_epuf_fill_scoring.py:150-159` + - **Evidence:** it points at a file that does not exist and accepts `FileNotFoundError`. `load_fill` reads the bytes before it compares hashes, so the mismatch branch never runs. + - **Fix:** write a real file and pass a wrong SHA-256. Expect `ValueError` matching `"SHA-256"`, and drop `FileNotFoundError` from the accepted errors. + +4. **Low: nothing tests round 3's "smaller of the two readings" rule end to end.** `tests/harness/test_epuf_fill_scoring.py:72-117` + - **Evidence:** + - The tolerance is 10 and the chosen cells all have finite gaps, so every score passes. + - `primary.n_failing == fallback.n_failing` is therefore just `0 == 0`. + - The `two_sided` reading never changes a tier. + - `test_combined_current_takes_the_smaller_gap_per_cell` tests only the helper. + - **Fix:** add a case where `fallback` and `two_sided` differ in a gating cell (young-band `level`, at a tight tolerance). Assert that `adoption_tier(candidate, combined_current(...))` returns `improves` where the fallback reading alone would give `not_adopted`. Also compare gaps cell by cell, not only failing counts. + +5. **Nit: the odd-oracle test does not check what its name says.** `tests/harness/test_epuf_fill_gate.py:383-396` + - **Evidence:** + - The test checks only that each column keeps the same set of values. Nothing checks that values move only within a stratum. + - The last assertion, `filled >= 0`, is always true, and its comment describes a different property. + - **Fix:** check that each moved value's source row has the same oracle key. + +6. **Nit: a malformed `gates.yaml` raises the wrong error.** `epuf_fill_gate.py:1424-1433` + - **Evidence:** an empty file, or `thresholds` that is not a mapping, raises `AttributeError` instead of `TestPartLocked`. TEST still stays unread, so this only affects the error message. + - **Fix:** use `document = yaml.safe_load(...) or {}` and guard with `isinstance` checks. + +7. **Nit: the tolerance table in the proposal rounds a few bounds inconsistently.** `docs/amendments/gate_epuf_fill_registration_proposal.md:541-549` + - **Evidence:** + - The `q50` upper bound is 0.074465 in the artifact; the table says 0.075. + - The `pzero` upper bound is 0.128484; the table says 0.129. + - The `r1` lower bound is 0.002476; the table says 0.0025. + - **Fix:** round to nearest throughout. + +8. **Nit: the D-series count hardcodes the bite multiple.** `scripts/build_epuf_fill_gate_floors.py:228` + - **Evidence:** `n_beyond_two` uses the literal `2` instead of `g.BITE_MULTIPLE`. The output does not change, because the constant is 2.0. + +## What I verified (by reading the code and the committed JSON) + +- **Rules module:** + - **Masks and current rules:** + - `family_mask("odd")` is the odd years minus pre-career years. + - `CurrentOddFill` falls back to the one known neighbour at age 22, with no out-of-range neighbour (2005 + 1 = 2006). + - **Universes and units:** + - Each universe reads no cell its own family fills, so group membership does not change when a fill is scored. + - Band units start at age 22, so every unit is a cell the odd family owns. + - **Year indexing:** + - `yr_cross` indexes stay within 1967–2004. + - `pr_in` compares two masked years and `pr_cross` a masked year with a recorded one, for 1930–1945. + - `r3` drops 2008. + - **Scoring path:** + - `score_candidate` refuses non-finite values, values outside [0, 1], and writes to cells the fill does not own, including the other family's cells. + - A share of 1 becomes exactly the wage base. + - **Partition and adoption:** the partition checks its reasons in the documented order, and `adoption_tier` and `adopt` match §7.3. + - **Oracles:** the permutation keys cannot collide, and `_permute_within` is correct. + - **TEST guard:** `epuf_matrix` refuses both `TEST` and `None` without the token. `test_part` needs both `locked` flags and the registration id. + - **Join checks:** the EPUF join assertions and the `pair` encoding (offset under 100) are sound. +- **Scoring module:** + - Undefined gating cells are dropped from the candidates' scores and both current-rule readings alike. + - `combined_current` takes the smaller gap and the smaller failing count, as round 3 specified. + - `_ROOT` and `parents[3]` resolve to the repository root. +- **Scripts:** + - The age-band sample size counts units the same way the cells do. + - Each perturbation and oracle replaces only its own family's cells. + - `registered_build` is true only for a DEV build with 200 replicates and all oracle seeds. + - The build records whether the bound files were clean. +- **Exclusions:** both new `src/` modules appear in `POST_REVIEW_SOURCE_EXCLUSIONS` (`:356-357`). The exact-tuple test and the reachability test both list them (`test_birth_evidence_artifact.py:217-218, 476-477`). +- **Claims against `runs/epuf_fill_gate_floors_v3.json`:** + - **Build and partition:** + - The build ran at `cb76ad15` with clean bound files. + - 319 of 326 cells gate (183 odd, 136 pre), and the 7 report-only cells and their reasons match §10a. + - Men's `q90` is exactly 1 in the 22–74, 30–44, 45–59 and 60–74 bands. + - **Checks on bite and perturbations:** + - The bite checks fail 96/183 and 112/136 cells by more than two tolerances. + - B2's pooled median-AIME falls work out to 5.3 and 4.0 tolerances. + - D1–D6 failing and beyond-two counts all match. + - The doses match: λ ≈ 0.9965 at `odd.men.a22_74.q90`; 12.5% at `paime_p90`; 1.4% at `ylevel`; 13.5% at `paime_p25`. + - **Oracles:** O1 fails 32/183 and O2 fails 6/136. + - **Ranges:** the noise ratio runs 0.087–0.212. All tolerance ranges match apart from finding 7. `pre.women.b1930_1934.aime_p10` is 0.391, and the pooled p10 tolerances are about 0.17. +- **Expected tier-count change from static counting** (the conftest rules classify by path and source): + - `test_epuf_fill_gate.py`: 47 tests, unit tier. + - `test_epuf_fill_scoring.py`: 6 tests, unit tier. + - `test_epuf_fill_gate_floors.py`: 27 tests, artifact tier (it names `"runs"` and `.json`). That is 21 tests run once for each of 3 builds, plus 6 that run only on the registered build. + - If all three files are new in #515, expect +53 unit and +27 artifact. +- **CI history:** CI checks out with `fetch-depth: 0`, so `test_registered_build_is_bound_to_its_rules` really runs there rather than skipping. + +## Not verified (needs `git` or `pytest`) + +``` +git diff --stat 5129ac32 70553e70 +git diff --name-only 5129ac32 70553e70 -- src/populace_dynamics/estimates/career.py # expect empty +git diff --diff-filter=M --name-only 5129ac32 70553e70 -- 'runs/*.json' # expect empty +git diff 5129ac32 70553e70 -- tests/tier_counts.json # expect unit +53, artifact +27 +git diff --name-only cb76ad15 70553e70 -- src/populace_dynamics/harness/epuf_fill_gate.py src/populace_dynamics/harness/epuf_cells.py src/populace_dynamics/harness/epuf_operator.py scripts/build_epuf_fill_gate_floors.py scripts/build_epuf_fill_psid_scale.py # expect empty, or the bound-file test fails +git show 70553e70:src/populace_dynamics/estimates/epuf_fill.py # finding 2 +PYTHONPATH=src .venv/bin/python -m pytest tests/harness/test_epuf_fill_gate.py tests/harness/test_epuf_fill_scoring.py tests/test_epuf_fill_gate_floors.py tests/estimates/test_birth_evidence_artifact.py -q -p no:cacheprovider +``` + +- **CI shard 1:** I could not confirm that the inherited depletion-cut pin is the only failure. The PR checks page did not load its results, and `gh` was not available. +- **Sealed files:** the proposal (§8) says `career.py` stays byte-identical, but I could not diff it. +- **Bound files:** nothing in `epuf_fill_gate.py` mentions round 3, which suggests it is unchanged since `cb76ad15`. Only the diff above can confirm it. \ No newline at end of file diff --git a/reviews/gate_epuf_fill_pr515_pr516_confirmation_review_20261004.md b/reviews/gate_epuf_fill_pr515_pr516_confirmation_review_20261004.md new file mode 100644 index 00000000..34ab8168 --- /dev/null +++ b/reviews/gate_epuf_fill_pr515_pr516_confirmation_review_20261004.md @@ -0,0 +1,123 @@ +# Confirmation review: PRs #515 and #516 after code review + +**Verdicts (provisional):** +- **PR #515: APPROVE WITH NITS.** +- **PR #516: APPROVE WITH NITS.** + +Every finding from both reviews is either fixed or left on purpose and recorded. I found no new correctness bug in the fixes. I found two new Low findings, both best fixed before the TEST run rather than before merge, and a few nits. + +**This session had read and search tools only, with no shell.** So I could not run `git diff`, `git show 61349f0b:…`, `sha256sum` or pytest. Everything below comes from reading the files at `2cebe01e`. The checks that need a shell are listed at the end; the verdicts hold only if they pass. + +## Prior findings + +### PR #515 review + +| # | Sev | Finding | Status | Evidence | +|---|---|---|---|---| +| 1 | Med | A run with injected inputs looks like the registered one; a `None` SHA was accepted | **Fixed** | `epuf_fill_scoring.py:196-205` requires a 64-character lowercase hex SHA-256 for every role before anything loads. `:219-224` adds `injected` and sets `registered_test_scoring` to false whenever either input is injected. Tests at `test_epuf_fill_scoring.py:150-159` and `:177-194`. `load_fill(sha256=None)` still skips its check silently (`epuf_fill.py:1695`), but `score_registered` can no longer reach that path. | +| 2 | Low | #515 alone cannot import `epuf_fill` | **Disclosed** | `registration_proposal.md:12` says "#516 merges before TEST is read". I could not see the GitHub PR description. | +| 3 | Low | The hash-check test never reached a hash check | **Fixed** | `test_epuf_fill_scoring.py:162-174` writes a real file with the wrong bytes and expects `ValueError` matching "SHA-256". | +| 4 | Low | Nothing tested the "smaller of the two readings" rule end to end | **Fixed in substance** | `:197-290` checks the per-cell minimum on real fallback and two-sided readings. `:312-334` shows the combined reference changing a tier. The review asked for the opposite direction, which `adoption_tier` makes impossible; the new test checks the correct one. Its docstring is wrong (N3). | +| 5 | Nit | The odd-oracle test did not check strata | **Fixed** | `test_epuf_fill_gate.py:399-421` compares the multiset of values within each stratum. Its key matches `odd_oracle_fill` (`epuf_fill_gate.py:1262-1269`; the age edges are `range(20,85,5)` in both). | +| 6 | Nit | A malformed `gates.yaml` raises `AttributeError` | **Left on purpose, recorded** | `epuf_fill_gate.py:1427-1429` is unchanged. Recorded in `registration_proposal.md:754-755`. | +| 7 | Nit | Tolerance table rounding | **Fixed** | `registration_proposal.md:542-550`: q50 0.074, pzero 0.128, r1 0.0025 (0.00248 at the lower bound). | +| 8 | Nit | Literal `2` in the builder | **Left on purpose, recorded** | `build_epuf_fill_gate_floors.py:228` is unchanged. Recorded in `registration_proposal.md:756-757`. | + +### PR #516 review + +| # | Sev | Finding | Status | Evidence | +|---|---|---|---|---| +| 1 | Med | Learned odd fills break on PSID gap years with no visible neighbour | **Fixed** | `psid2010_epuf_fill.py:139-157`: `odd_years` defaults to `SCORED_ODD_YEARS` (1997-2005). Every gap year is hidden from the fills, whether filled or not. A gap year is filled only if a neighbour in `given` is finite, and `given` already hides pre-career years and `boundary_2014`. Both `OddKnnFill` (`epuf_fill.py:970-990`) and `OddForestFill` (`:657`) handle a right neighbour only, so neither can return NaN on a row that reaches them. The docstring states the extrapolation (`:11-20`). Tests: the seam and the no-neighbour start at `test_psid2010_epuf_fill.py:188-224`, and the default scope at `:164-185`. | +| 2 | Med | The donor cache was keyed by `id(self)` | **Fixed** | `bank_digest` (`epuf_fill.py:1225-1240`) hashes sex, birth year, match vector, block and `k`. The cache key (`:1249-1255`) uses that digest instead of `id`. A test covers it, weakly (N4). | +| 3 | Med-low | The fit script overwrote staged files before refusing | **Fixed** | `fit_epuf_fills.py:169-173` refuses an existing manifest before reading TRAIN. `:188-196` never replaces a staged file with other bytes and writes new files with `"xb"`. | +| 4 | Low | float32 thresholds can leave a leaf empty | **Fixed for the quantile forest** | `epuf_fill.py:536-541` raises at fit if a leaf is empty; `test_epuf_fill.py:198-200` checks it. Thresholds stay float32, so the bytes are unchanged. There is no comparison against `apply`, which is acceptable because the stored traversal is used for both fitting and drawing. The zero forest is not checked (N5). | +| 5 | Low | Uncoded sex got no copula | **Fixed** | `BySexFill.fill` passes `np.full(len(rows), value)` (`:1643-1653`). Tested at `test_epuf_fill.py:182-195`. | +| 6 | Low | "Byte-reproducible" depends on the environment | **Fixed** | `fit_epuf_fills.py:218-224` records the Python version, zlib runtime and platform, and the manifest carries them. The registration document qualifies the claim (`:26-31`). | +| 7 | Low | Provenance gaps in the scripts | **Fixed** | Score script: pins the manifest SHA-256 (`:35-38`, `:76-80`), records `code_files_clean` (`:82-89`), writes file names only (`:115-117`), writes a started marker (`:97-109`), and checks the lock first (`:92-96`). Fit script: `CODE_FILES` now includes `epuf_fill_gate.py` and `epuf_operator.py` (`:45-50`). | +| 8a | Nit | Stale `PreDonorFill` docstring | **Fixed** | `:1137-1153` now describes seven features, `bank_size`, and "scaled by seven". | +| 8b | Nit | `start_year` was really the last year | **Fixed** | Renamed to `last_year` and documented (`:99`, `:107-108`). | +| 8c | Nit | `bank_block` was not clipped | **Fixed** | `np.clip(values, 0, 1)` at `:1218-1220`. | +| 8d | Nit | `ok` always True in `PreChainFill` | **Fixed** | It is no longer in `PreChainFill.fill` (`:1508-1564`). | +| 8e | Nit | `match_vector` used PSID years after 2006 | **Fixed** | `_BANK_LAST_YEAR` limits the summaries (`:1078`, `:1089-1093`). | +| 8f | Nit | The copula test did not check the coded-sex row | **Fixed** | `test_epuf_fill_candidates_manifest.py:101-105` asserts each coded-sex row is nonzero in the registered manifest. | +| 8g | Nit | `n_units` applies per sex | **Fixed** | `candidates_registration.md:57`. | + +## New findings + +**N1. Low: the started marker can record a TEST read that never happened, and block the one run.** `scripts/score_epuf_fill_test.py:97-113`, `epuf_fill_scoring.py:206-221` +- **What happens:** the marker is written before `score_registered` runs. `score_registered` loads and hash-checks the candidate files before it calls `test_part`. So a wrong `--fills-dir`, an unstaged file or a SHA mismatch leaves `.started.json` behind although TEST was never read. +- **Effect:** a rerun with the same `--output` fails on `open("x")`. The operator then has to delete a marker whose whole purpose is to be a trace of the read. +- **Fix:** hash-check the staged files against the manifest before writing the marker, or give the marker a `phase` field and update it after `test_part` returns. `score_epuf_fill_test.py` is not a pinned file, so this needs no refit. It should land before the TEST run. + +**N2. Low: no test ties `REGISTERED_MANIFEST_SHA256` to the committed manifest, and the score script has no tests.** `scripts/score_epuf_fill_test.py:36-38` +- **Fix:** in `tests/test_epuf_fill_candidates_manifest.py`, assert that the SHA-256 of `MANIFEST` equals the script's constant. Also test that the script refuses another manifest, refuses before lock, and refuses an existing output. + +**N3. Nit: a test's docstring contradicts the rule it tests.** `tests/harness/test_epuf_fill_scoring.py:198-205` +- **What's wrong:** it says the combined reference "admits it as 'improves' where the fallback alone would not". `adoption_tier` (`epuf_fill_gate.py:1072-1083`) makes that impossible: a smaller current gap and a smaller failing count can only tighten the tier. That is what the test at `:312` correctly checks. +- **Also:** the test never asserts that some gating cell actually differs between the two readings. +- **Fix:** reword the docstring, and assert that at least one cell's fallback and two-sided gaps differ. + +**N4. Nit: `_BANK_DIGESTS` keeps every digested `PreDonorFill` alive forever.** `epuf_fill.py:1069`, `:1229-1240` +- **Effect:** the registered run uses one fill, but anything that fits fills in a loop holds every bank in memory. +- **Test gap:** the cache test (`test_epuf_fill.py:162-179`) keeps both fills alive, so it never reproduces the original address-reuse bug. It only checks that the digests differ. +- **Fix:** use `functools.cached_property`, which works on a frozen dataclass without slots. This file is pinned, so either record the nit or fix it together with a refit. + +**N5. Nit: the zero forest has no empty-leaf check.** `epuf_fill.py:831-833` +- **Effect:** an empty leaf silently gets `p = 0` through `hits / np.maximum(total, 1)`. This is the same float32 cause as #516's finding 4. +- **Fix:** raise there as `fit` now does for the quantile forest. If no leaf is empty, the bytes do not change. The file is pinned, so either record the nit or fix it with a refit. + +**N6. Nit: the score record can claim the registered scoring under a non-default `gates_path` or `data_dir`.** `epuf_fill_scoring.py:223` +- **Effect:** neither argument is recorded, so a run pointed at a substitute `gates.yaml` would still set `registered_test_scoring` to true. The score script never passes them. +- **Fix:** add both to `injected`. Separately, the score script's `CODE_FILES` (`:40-45`) leaves out `epuf_cells.py` and `epuf_operator.py`, which scoring uses. + +**N7. Nit, older than the fixes: the module docstring overstates the pre-career fill.** `psid2010_epuf_fill.py:21-23` +- **What's wrong:** it says every year before the career start becomes the pre-career draw. Years the PSID recorded before the career start keep their value, because only `pre_mask & ~in_career` is added (`:190`). +- **Fix:** reword the docstring. + +## Questions 2-4 + +**2. The fixes introduce no new bug that I found.** Item by item: +- `score_registered`: the SHA-256 check runs before any load, and the `injected` flags are correct. +- `fill_careers`: the `odd_years` scope, the no-visible-neighbour rule and `last_year` all index correctly (`years[columns]`, with the `inside` filter applied before `gap`). +- `bank_digest` and `_NEAREST_CACHE`: the digest is keyed by content and the cache is bounded at 4 entries. Only N4 remains. +- `BySexFill` routing is correct. +- The empty-leaf check is correct where it applies (N5 is the gap). +- `match_vector`'s 2006 limit is correct. +- The fit script refuses before reading TRAIN and never overwrites a staged file. +- The score script pins the manifest and checks the lock before writing the marker. N1 is about where the marker sits. + +**3. The refit:** +- **Manifest:** `code_commit` is `598e4436…`, which comes after the fixes commit `358ad0d2`, and `code_files_clean` is `true`. +- **Artifact SHA-256 values:** they match the registration document's table. The untracked `.diag/epuf-fill/dev_registered_dryrun.json`, which looks like the dry run's output record, also contains all four. That supports the claim that the DEV dry run loaded these same bytes. I could not compare against `git show 61349f0b:runs/epuf_fill_candidates_v1.json`. The committed log records only the old manifest's SHA, `8d421536…` (`dev_scores_after_round_2.jsonl:17`). +- **Manifest pin:** the score script pins `a3043113…`, the same value the registration document gives. I could not compute the file's actual hash. + +**4. The DEV dry-run claim holds.** No fix changes a draw for a scored person: +- **Uncoded sex:** EPUF codes sex 3 as unspecified (`data/epuf.py:11`). Every population is restricted to `_coded` sex, 1 or 2 (`epuf_fill_gate.py:472-473`), so the persons whose draws changed are never in a cell. +- **`match_vector`:** EPUF's years end at 2006, so the new limit changes nothing on EPUF. +- **The cache:** only its key changed, not the values. +- **The clip and the leaf check:** neither changes the bytes, which the matching SHA-256 values confirm. +- **Other persons:** draws are keyed by person, and an existing test (`test_epuf_fill.py:100-113`) shows they do not depend on the other persons' order. +- **The PSID path:** EPUF scoring does not use it. + +**5. Merge readiness.** Apart from the known blockers (#509 first, ratification d927, CI shard 1), both PRs are ready once the commands below pass. N1 and N2 should land before the TEST run, not necessarily before merge. N4 and N5 touch the pinned `epuf_fill.py`, so the simplest course is to record them as #515's two nits were recorded. + +## What I verified, and what I could not + +**Verified by reading at `2cebe01e`:** +- every prior finding, against the code and tests cited above; +- both proposal notes for the nits left in bound files; +- the manifest fields and the registration document's SHA-256 values, environment and claims; +- the birth-evidence exclusions (`first_estimates_birth_evidence.py:356-361`); +- that the oracle test's strata match the code. + +**Not run (needs a shell):** +``` +git diff e1f10808 2cebe01e --stat +git diff 598e4436 2cebe01e -- src/populace_dynamics/estimates/epuf_fill.py src/populace_dynamics/harness/epuf_fill_gate.py src/populace_dynamics/harness/epuf_operator.py scripts/fit_epuf_fills.py # expect empty +git show 61349f0b:runs/epuf_fill_candidates_v1.json | grep '"sha256"' # expect 37a9ea76…, 8c7d323d…, 8bb48b02…, 3c31fbd3… +shasum -a 256 runs/epuf_fill_candidates_v1.json # expect a304311343f3…62d0 +git diff 70553e70 e1f10808 -- src/populace_dynamics/harness/epuf_fill_gate.py scripts/build_epuf_fill_gate_floors.py # expect empty (bound files) +PYTHONPATH=src .venv/bin/python -m pytest tests/estimates/test_epuf_fill.py tests/cohorts/test_psid2010_epuf_fill.py tests/test_epuf_fill_candidates_manifest.py tests/harness/test_epuf_fill_scoring.py tests/harness/test_epuf_fill_gate.py tests/test_epuf_fill_gate_floors.py tests/estimates/test_birth_evidence_artifact.py -q -p no:cacheprovider +``` + +I made no edits, commits or GitHub posts, and I read no EPUF microdata. This session had no Write tool or plan-mode exit, so this review is the whole output. \ No newline at end of file diff --git a/reviews/gate_epuf_fill_round1_referee_20261004.md b/reviews/gate_epuf_fill_round1_referee_20261004.md new file mode 100644 index 00000000..09459bbc --- /dev/null +++ b/reviews/gate_epuf_fill_round1_referee_20261004.md @@ -0,0 +1,265 @@ +# Referee report: `gate_epuf_fill` registration, round 1 + +Branch `epuf-career-fill-20261003`, head `01966003`. Independent adversarial referee. + +**What I could not do.** This session had no shell. I could only read files, so I ran neither test file and no `git log/show/diff`. Three things are therefore unverified: +- the push-time chronology; +- whether `test_bound_files_are_unchanged_since_the_build` still passes after `7a49b520` ("tier counts"); +- whether the dry-run commit `61ade83a` is an ancestor of `14045be4`. + +All the evidence below comes from reading the code and doing arithmetic by hand on the committed artifacts. I read no EPUF microdata and called nothing that reads EPUF. + +## Verdict: AMEND BEFORE LOCK + +Most of the machinery is sound: the split, the masks, `aime_35`, the floor algebra, the partition rule, the oracles' permutation logic and the scoring arithmetic. The artifact also confirms the operating-characteristic claim. But three findings change a floor or a cell definition, so lock-after-fixes is not possible: +- **Finding 1:** the floor pool for the odd-family AIME groups is a different population from the one the cell scores. +- **Finding 2:** the 18-29 band scores ages 18-21, which the assembler never fills with the neighbour mean. +- **Finding 3:** per-cell tolerances leave common-direction bias across cells unchecked. That matters most for the AIME distribution, which Max asked about. + +Several more rule fixes do not touch floors (findings 4-7). One record problem must be fixed before any amendment: candidates already exist (finding 7). + +--- + +## Findings + +### 1. BLOCKING: the floor pool for the odd-family AIME groups differs from the population the cell scores + +**Evidence (verified).** +- The cell's population: `odd_cells` computes the AIME quartiles over `career_universe`, persons positive in any unmasked year 1968-2006 (`epuf_fill_gate.py:425-443`). +- The floor's pool: `group_eligible` builds it from `family_universe("odd")`, persons positive in an even year 1996-2006 (`epuf_fill_gate.py:684`, `:653-654`). +- Every pool member is in the career universe, but the reverse fails. People born 1936-45 who stopped working before 1996 are in the cell but never in a floor sample. +- The PSID count, 145 for men born 1936-40, uses the career-universe definition (`build_epuf_fill_psid_scale.py:63-64, 81-92`). +- The artifact shows the gap between the two populations: + +| Group | Cell `n` (truth) | Floor `pool_size` | +|---|---:|---:| +| `odd.men.b1936_1940` | 13,101 | 8,868 | +| `odd.women.b1936_1940` | 12,064 | 7,627 | +| `odd.men.b1941_1945` | 15,958 | 12,103 | +| `odd.women.b1941_1945` | 15,125 | 11,245 | + +**Effect (inferred).** +- The `pre` groups for nearly the same cohorts use the same universe for pool and cell, at similar `n`. +- Their median tolerances are 0.087 (men 1935-39, n=131), 0.125 (women 1935-39), 0.066 (men 1940-45) and 0.096 (women 1940-45). +- The odd-family medians are 0.051, 0.088, 0.051 and 0.070, roughly 60-75 percent of those. +- So the 12 odd-family AIME tolerances are priced on the wrong population, probably too tight. The `noise_ratio` for these groups also uses the wrong N. + +**Required fix.** +- Make `group_eligible` return exactly the population the cell counts: for odd cohort groups, the career universe intersected with sex and cohort. +- Rebuild the four groups' floors. +- Add a test that, for every group, the pool equals the set of persons the group's cells count. Pool ⊆ universe is not enough. + +### 2. MAJOR: the `a18_29` band scores ages 18-21, which the assembler never fills with the neighbour mean + +**Evidence (verified).** +- `build_career` emits years only from `coverage_start = max(1968, birth_year + 22)` (`career.py:1004, 1070-1077`). +- The PSID careers frame is built from `record.years` (`psid2010.py:2005-2012`). So every PSID `gap_imputed` unit behind `odd.*.a18_29` is at age 22-29. +- EPUF's `a18_29` units run from 18 to 29 with no career-start restriction (`epuf_fill_gate.py:119-124, 363-368`). +- At ages 18-21 the assembler's actual rule is "zero" (family `pre`, which `yzero`/`ylevel` already score), not "neighbour mean". So these units are scored twice, under two different current rules. +- The pool and the per-person unit rate behind `n = 1,321` and `n = 1,778` are measured on a different age range from the PSID count they are scaled to. + +**Effect (inferred).** +- The band's truth values and sigmas are driven by teenagers' entry dynamics, which are exactly where O1 fails (`wint` and `zexit` at 18-29). +- An odd-year candidate is judged on units it will never fill in deployment. + +**Required fix.** +- Restrict odd units to `t >= birth_year + 22`, making the band `a22_29`. +- Rebuild the `odd.men.a18_29` and `odd.women.a18_29` floors, the sample sizes and the partition for those 14 cells. +- Optionally do the same for post-claim units in 60-74. The PSID fills no years after claiming (`career.py:1003, 1051-1053`), while EPUF fills all of them. + +### 3. MAJOR: per-cell `K = 1` leaves common-direction bias unchecked, including bias in the AIME distribution + +**Evidence (verified from `runs/epuf_fill_gate_floors_v1.json`).** +- The `pre` AIME tolerances are wide: + - median: 0.066-0.125 log; + - 25th percentile: 0.129-0.192 log, that is, 14-21 percent. +- A fill that understates every pre-1946 cohort's median AIME by 6 percent passes all six median cells. +- The current pre-career rule itself passes `pre.men.b1940_1945.aime_p75` (gap −0.033, 0.88 tolerances; artifact line 57404). It sits at only 1.26 tolerances on that cohort's median, an 8 percent understatement (line 57395). +- The odd-family AIME cells cannot bind. The current rule's gaps there are at most 0.25 percent against tolerances of 3.3-13.7 percent. This is because they cover cohorts whose AIME window holds only 1-5 masked years, at ages 57-61. + +**Effect (inferred).** +- PSID-based estimates are pooled across cohorts and sexes; the DYNASIM comparisons use the whole cohort. Their sampling error is about `1/sqrt(k)` of the per-cell error. +- A bias of 0.9 cell-sigma in the same direction in all six `pre` cohort-sex cells is therefore about 2 sigma on a pooled 1930-1945 estimate. The PSID would detect it, yet the gate passes it. +- "A fill within the tolerance is never the larger source of error" holds cell by cell, not for aggregates. + +**Required fix (truth-side, so it can be done before seeing any candidate).** +- Add pooled gating cells, each with its floor at the pooled PSID size: + - per family and sex, pooled over cohorts: the AIME quartiles for 1930-45, plus `pzero`/`plevel` and `yzero`/`ylevel`; + - per sex, pooled over bands: `r1`, `r2`, `level`, `zexit`. +- Alternatively, register a sign rule: for example, the mean standardized gap over a statistic's cells must lie within `1/sqrt(k)`. +- Also consider AIME p10 and p90 cells. The current pre-career rule hits the lower tail hardest: −63 percent at p25 for women born 1930-34 (gap −0.84 log). + +### 4. MAJOR: in family `pre`, the fill reads true odd years 1997-2005, which the PSID never records + +**Evidence (verified).** +- The two families are masked separately (§3), so a `pre` fill is handed the true odd years. +- The candidate `PreDonorFill.donors` matches on `_first_recorded(readable, …)`, where `readable` hides only `fill_mask`, the pre mask (`estimates/epuf_fill.py:994-995`). +- O2 matches the same way (`epuf_fill_gate.py:995-1006`). +- For people born after about 1970 the career starts in 1992 or later, so the "first five recorded years" include 1997, 1999 and so on. On the PSID those years are neighbour means or learned fills. +- `yr_cross` at age 24 falls on an odd year for half of the people born 1973-1981, and is read as truth. + +**Effect (inferred).** The `b1966_1980` cells, and partly `b1956_1965`, score a fill with information it will not have in deployment. The gate overstates its quality there. + +**Required fix.** +- Register that each family's fill receives the union of both masks as unknown (NaN) inputs, while the cells score only the family's own mask. +- Floors and partition do not change, because truth cells are unaffected. O2 should be rerun under the same rule. + +### 5. MAJOR (ceremony): no registered scoring path, and the fill-validity invariants are unenforced + +**Evidence (verified).** +- `epuf_fill_gate.py` scores matrices handed to it. Nothing in the registered code: + - hides masked values from a candidate; + - checks that unmasked cells come back unchanged; + - checks that filled values are finite and lie in `[0, wage base]`. +- The claim that the true and filled matrices share every universe (§4) holds only if the fill leaves the even years alone. `odd_cells` recomputes the universe, and reads `t±1`, from the filled matrix (`:355-356, 370-401`). +- A fill that edits recorded years could move `r1`, `zint`, `zexit` and `wint`. +- The working-tree DEV scorer overwrites unmasked cells (`.diag/epuf-fill/dev_score.py:27`), but it is unregistered. +- The candidates receive the full true matrix and rely on `known = … & ~fill_mask` (`epuf_fill.py:619, 757`). +- The prior round flagged the same problem (`reviews/gate_epuf_round1_referee_20261002.md` finding 7). + +**Required fix.** +- Add a registered `score_candidate(fill, part)` that: + 1. passes NaN in every cell of the union mask; + 2. asserts that the output is finite, `0 <= x <= cap`, and equal to the input on unmasked cells; + 3. reads TEST through a single audited call. +- Pin it in `BOUND_FILES` and in the lock record. Test each assertion. + +### 6. MAJOR: the "improves" tier is nearly vacuous wherever the current rule's gap is non-finite + +**Evidence (verified).** +- `adoption_tier` allows `|gap| <= max(|current gap| or inf, tau)` (`epuf_fill_gate.py:878-883`). +- The current rules have non-finite gaps in 21 gating `odd` cells (`zint`, `zexit`, `wint`) and in 30 gating `pre` cells (`plevel`, `pr_in`, `pr_cross`, `ylevel`, `yr_cross`). In all 51, any finite gap is "no larger". +- The current pre-career rule's `pzero`/`yzero` gaps are 0.6-1.5 log. +- The current rule fails 59 of 60 `pre` cells (line 57802). So in family `pre`, "improves" reduces to: no non-finite gap, and AIME quartiles no worse than zeroing. +- A pre-career fill missing `pr_cross` by five tolerances would be adopted as an "uncertified improvement". + +**Forking path.** This tier was added after the TRAIN dry run showed an oracle failing `r2`/`r4` (§9). It does not change certification, so it is not a self-rescue of the gate. It is a loosening of adoption, made after seeing results that stand in for candidates. + +**Required fix (no floor change).** Cap the allowance at `max(tau, min(|current gap|, M * tau))` with `M` fixed now, for example 3. Alternatively, require `|gap| <= max(tau, |oracle gap| + tau)` using the stored DEV oracle gaps. Either makes "improves" mean near-oracle performance. + +### 7. MAJOR (record and forking paths): candidates already exist, contrary to the proposal + +**Evidence (verified).** +- The proposal header says: "No candidate has been fitted" (line 10). +- The working tree contains: + - all four registered fills, implemented in the untracked `src/populace_dynamics/estimates/epuf_fill.py`; + - fitted blobs in `.diag/epuf-fill/cache/{odd_quantile,odd_knn,pre_donor,pre_chain}.npz`; + - `dev_fit.py`, which fits on TRAIN; + - `dev_score.py`, which scores on DEV against the registered tolerances. +- No DEV score output is present, and no TEST cache exists: only `part0.npz` and `part1.npz`. +- Developing candidates on DEV is allowed (§2). But any amendment from this round will now be written with candidate behaviour knowable. +- The implementation also departs from §7.1: + - it uses an AR(1) copula, not a "person-level" one; + - it adds context at offsets 5-9 (`epuf_fill.py:233, 249-281`). + +**Required fix.** +- Correct the ceremony stage. +- Before amending, commit (or hash into the record) any DEV candidate scores already produced, so the next referee can check that the amendments do not favour them. +- Justify every amendment on truth-side grounds only. +- Align §7 with the code, or state that §7 describes candidates loosely and the registered commit plus SHA-256 governs. + +### 8. MAJOR: the bite checks hold only through degenerate failures; the AIME cells have no dosed demonstration + +**Evidence (verified).** +- B1's 48 failures include 21 infinite ones (filled zero shares). +- B2's 54 include 30 NaN or infinite ones and 12 infinite-ratio `pzero`/`yzero` ones. +- B1 and B2 show that the gate rejects the current rules, which is what Max asked for. They say nothing about the cells' detection point for realistic errors. The prior round's central flaw was exactly a bite with no operating characteristic. +- The oracles help only in part, because O1 and O2 preserve several cells exactly by construction: + - O1's strata nest inside the 30-44, 45-59 and 60-74 bands, so `zint`, `zexit`, `wint`, `atcap` and `level` there have gap 0.0 (lines 57880-58092). + - O2 preserves `pzero`, `plevel`, `pr_in`, `yzero` and `ylevel` exactly. + +**Required fix.** Before lock, register and report truth-side dosed bites with their gap in tolerances. For example: +- O2 with blocks scaled by 0.90 and by 0.95: an AIME and `plevel` level bias; +- O1 with draws shrunk toward the stratum median: dispersion; +- "copy `t+1`"; +- marginal draws within sex and age. + +Report each one's dose at 1 and 2 tolerances, as in the prior round's finding 3. Report only; no new pass condition is needed. + +### 9. MINOR: unmeasured failure modes + +The failure modes named in the prompt are caught (inferred from the cell definitions): +- **"Copy `t+1`":** gives `zint` and `wint` of 0, an infinite gap, and pushes `r1` up. +- **Marginal draws:** `r1` collapses. +- **Right levels, wrong ranks:** `r1` catches it. + +Not measured: +- the marginal dispersion of the share at `t`, since only the mean (`level`) and `atcap` are scored; +- persistence from a masked year to recorded years beyond ±1; +- any AIME effect for cohorts born 1946 and later. For most of the PSID-2010 cohort the gate's AIME coverage is nil. + +Consider the following, gating or report-only: +- p10, p50 and p90 of the positive share at `t`; +- a partial-AIME proxy for people born 1946-65: top-k indexed earnings through 2006. + +### 10. MINOR: claims not supported as written + +- **§4, "No universe or conditioning denominator reads a masked year".** The universes do not, but: + - `atcap`'s denominator is `centre > 0` (`:418`); + - every correlation subset is "positive in both", which reads `t`. + + Reword, and note that zero-selection error moves `atcap` and the `r` cells. +- **§9's AIME table reports log points as percentages.** Men 1930-34: the DEV log gap is −0.384, a 32 percent fall, not 38 (line 57270). In percent the table should read 32/20/8 and 34/25/13. +- **§10, "1.3 to 5.1 tolerances at the median".** The medians run 1.26-4.56; 5.1 is women 1930-34 at p25 (line 57537). +- **§10, "(infinite)" for `pr_in`, `pr_cross` and `yr_cross`.** These are NaN: there are no positive pairs. +- **§10, "Young workers' re-entry (`wint`) needs more than the neighbours too".** This is unsupported. O1's 15-19 age stratum straddles the band's age-18 edge (`_ORACLE_AGE_EDGES`, `:170`). Its `wint`/`zexit` failures appear only in the one band whose strata do not nest (inferred). +- **§8, "That module is listed in the reducer's `POST_REVIEW_SOURCE_EXCLUSIONS`".** It is not. Only `harness/epuf_fill_gate.py` is (`first_estimates_birth_evidence.py:356`), and `epuf_fill.py` is untracked under `src/`. +- **Scope.** A pass certifies the fill on EPUF's administrative, capped, disclosure-perturbed earnings. In deployment the conditioning years are PSID survey reports of uncapped labour income. Say so in "What this gate is". + +### 11. MINOR: the "conservative" sample-size claim + +Three things affect the size behind the age-band floors: +- the PSID count includes filled years 2007-2013 (`_STRUCTURAL_GAP_YEARS = 1997..2011` plus 2013; `career.py:35, 1058-1062`); +- it includes single-neighbour copies; +- matching units rather than persons gives EPUF samples more units per person in some bands: 2.99 against the PSID's 2.70 at 18-29. + +Most of this, together with unweighted counts and the PSID's clustering by family, makes the gate stricter. The person-clustering effect runs slightly the other way. State the claim as "stricter on balance". + +The "unweighted counts are stricter" claim holds (inferred): the Kish design effect is at least 1, so the effective n is at most the count. + +### 12. MINOR: §12 conflates two rejected alternatives; Max should ratify the reading of "floor" + +- With same-person scoring, the house formula applied to full-DEV halves gives a tolerance of roughly `4.5 * sqrt(2n/N) * sigma`, about 0.55-1.35 sigma (inferred from the stored noise ratios of 0.087-0.212). That is the same order as `K = 1` and is tied to the comparison's real noise. +- The proposal's PSID-size sigma is a defensible materiality yardstick. It matches the house's deployment-scale floor (`gates.yaml:22-30`). But it is not "the noise of the comparison" in the sense other gates use. +- Required: record this alternative in §12, and have Max explicitly ratify that the tolerance is a materiality threshold. + +### 13. MINOR: code robustness + +- **`epuf_matrix` (`:1066-1072`).** The `searchsorted` join assumes every annual `person_id` is in the demographic file. The reader does not validate this (`data/epuf.py`). A missing id would write its earnings into a neighbouring person's row, possibly a DEV row; an id past the end would raise. Assert `ids[position] == annual ids` and no duplicate (person, year) rows. If the assertion ever fires, rebuild the floors. +- **`atcap` uses `centre >= cap`.** Require candidates to emit exactly `share * cap` at the cap, or compare with a tolerance. + +### 14. MINOR: provenance of the dry run + +- `.diag/epuf-fill/floors_train_dry.json` was built at `61ade83a` (04:52 UTC). The DEV build was at `d21aa659` (05:04 UTC). +- `61ade83a` is not in the listed branch history, and I could not check whether it is reachable. +- Commit the dry-run artifact as lineage and state its commit relationship to `14045be4`. + +### 15. MINOR: tests + +The existing tests are good for what they cover. Add: +1. pool equals the cell's counted population (finding 1); +2. fill-validity and masked-input tests for the scoring path (finding 5); +3. the `epuf_matrix` join assertion (finding 13); +4. a universe-invariance test for `career_universe` and the AIME cells; +5. a test that the `odd` units exclude ages under 22 after finding 2's fix. + +--- + +## What I verified + +- **Split, masks, current rules.** `split_part` matches the stated hash and thresholds. `odd_mask`, `pre_career_mask`, `current_odd_fill` and `current_pre_career_fill` match `career.py`'s rules on EPUF; the one-neighbour fallback is moot there. +- **`aime_35`.** It follows the statute for people born 1929-1945: 35 computation years, every year from 1951 through age 61, indexing to NAWI at age 60, later years nominal, and a floor over 420 months. The test pins it to `ss.statutory_aime.aime`. I could not run that test. +- **Cell helpers.** `_share`, `_spearman_mean` (events = smallest pair-year) and `weighted_spearman` (mid-ranks for ties) are correct. The `r1`/`r2`/`r4` pair counts are 10/4/3. +- **Floor, partition, scoring, oracles.** `floor_group` is correct: disjoint samples, `/sqrt 2`, RMS, seeds. `partition` follows the stated order. `score` and `adopt` match the text. The O1/O2 key packing has no collisions, and `_permute_within` permutes only within keys. +- **Counts.** The artifact has 136 cells and 24 groups. 130 gate: 70 `odd`, 60 `pre`. The six report-only cells are as listed. +- **Tolerances.** Every range in §10's table matches the artifact. +- **Sample sizes and noise ratios.** They recompute; for example, men 18-29 gives 3,949 / 2.989 = 1,321 and `sqrt(1321/80398)` = 0.128. The range is 0.087-0.212. +- **Bite counts.** + - B1: 48 of 70. `r1` 7.0-16.9 tolerances, `r2` 3.7-12.8, `r4` 6 of 8, `atcap` 8.1-17.2. + - B2: 54 of 60, with 12 of 18 AIME quartiles beyond 2 tolerances. + - O1: 16 failures (7 `r2`, 6 `r4`, 2 `wint`, 1 `zexit`), with gaps of 0.015-0.043. + - O2: 4 `yr_cross` failures at 1.11-1.20 tolerances. + - The TRAIN dry run's 131 gating cells, 49/71 and 55/60 match §9. +- **Operating characteristic.** A faithful fill's gap has SD of at most about `sqrt(1 + 1/20) * sqrt(n/N) * sigma`. The artifact supports this: O1's per-seed SD over sigma is about 0.10 for men 30-44 `r2` and about 0.13 for women 60-74 `r4`, both below their noise ratios. So with K = 1, the chance that a zero-bias fill fails any of 130 cells is negligible, and the verdict is close to a step at `|bias| = sigma`. +- **Leakage.** No TEST value reaches a floor, a tolerance, the partition or the oracles. The builders accept only TRAIN or DEV, and the PSID builder reads no EPUF. In `.diag`, candidate fitting uses TRAIN (`part0`) and scoring uses DEV (`part1`); there is no TEST cache. +- **The two dry-run rule changes.** Universe membership in the oracle strata affects report-only output, so it is harmless. The "improves" tier changes adoption, not certification; see finding 6. \ No newline at end of file diff --git a/reviews/gate_epuf_fill_round2_verification_20261004.md b/reviews/gate_epuf_fill_round2_verification_20261004.md new file mode 100644 index 00000000..0339909f --- /dev/null +++ b/reviews/gate_epuf_fill_round2_verification_20261004.md @@ -0,0 +1,240 @@ +# Referee report: `gate_epuf_fill` amendment 1, round 2 (verification) + +Branch `epuf-career-fill-20261003`, head `f8623f4d`. Independent referee, read-only. + +**What I could not do.** This session had file-read and search tools but no shell. So: + +- **Tests not run.** I ran neither test file. Every statement below about the tests comes from reading them. +- **No `git diff`.** I could not diff `7061200d` against `23bee82c`, so I have not checked that the rules files are byte-identical at the two commits. `test_registered_build_is_bound_to_its_rules` would check it, but I could not run that test. +- **Chronology from git's logs.** I got the commit order from the reflog and the remote-tracking log, read as plain files under `.git/`. +- **No EPUF microdata.** I read none, including the `.diag/epuf-fill/cache/*.npz` part caches, and called no EPUF reader. + +Claims are marked **(V)** for verified (I read it, or did the arithmetic on stored values) or **(I)** for inferred. + +## Verdict: LOCK AFTER LISTED FIXES + +Amendment 1 genuinely fixes round 1's three blocking or cell-defining findings: + +- **Finding 1:** each group's floor is now priced on exactly the population its cells score. +- **Finding 2:** the youngest band now starts at age 22. +- **Finding 3:** pooled groups now gate same-direction bias. + +The v2 artifact recomputes wherever I checked it. Both bite checks hold by a wide margin. No amendment looks tailored to the disclosed DEV scores. + +Two things stop me saying lock as-is: + +- **A scoring defect of the same kind as round 1's finding 2.** The odd-year fill is scored on pre-career odd years that it never fills in use (new finding 2). Fixing this changes no floor, tolerance or partition, because floors are built from true data only. +- **Gaps in the record.** + - An amended-rules dry run on TRAIN, from 16 minutes before the rules commit, is not disclosed (new finding 3). + - Candidate scoring on DEV since the rules commit has no log. + +New finding 1, the three men's `q90` cells that sit at the wage base, is a real definitional flaw. But it changes almost no verdict, so I ask for disclosure rather than a rebuild. + +## Round 1's findings + +| # | Finding | Status | Evidence | +|---|---|---|---| +| 1 | Odd-family AIME floors priced on the wrong population | **Fixed** | Cells and floors both use `group_rows` (`epuf_fill_gate.py:450-481, 645-659`; builder `:316-321`). (V) The floor `pool_size` equals the truth cell's `n` in all 30 cohort groups. For example, `odd.men.b1936_1940` is 13,101 in both, against 8,868 in v1. (V) The test checks the cell side only (new finding 7). | +| 2 | 18-29 band scored ages the assembler never fills with the mean | **Fixed** | `ODD_AGE_BANDS` now starts at 22 (`:142-148`). (V) The PSID count stays at 3,949, as it should: the PSID never filled below 22. DEV units per person are now 2.454. (V) The optional post-claim part was not done; that is acceptable. A residual of the same kind remains: new finding 2. | +| 3 | Same-direction bias across cells unchecked | **Fixed** | Pooled groups (`a22_74`, `b1936_1945`, `b1930_1945`, `b1946_1980`) and AIME p10/p90 cells added. (V) The pooled median-AIME tolerance for men is 0.042, against 0.075-0.088 for the single cohorts. (V) Pooling is within sex only; see note 8. | +| 4 | Pre fill read true odd years | **Fixed** | `score_candidate` puts NaN on the union of both masks (`:1071-1073`). (V) A test asserts it (`test_epuf_fill_gate.py:543-547`). (V) O2 treats odd years as unknown (`:1217-1219`). (V) | +| 5 | No registered scoring path or validity checks | **Partly fixed** | Added:
• `score_candidate` checks finiteness, `[0, 1]` and that other cells are unchanged (`:1086-1098`) (V);
• TEST is read only through `test_part()` (V);
• the module is in `BOUND_FILES` (V).
Still missing: a registered TEST entry point that ties the truth, tolerances and partition to the v2 artifact by hash (new finding 4). | +| 6 | "Improves" tier nearly vacuous | **Fixed** | `allowed = max(tau, min(baseline, 3 tau))` (`:1001`), with tests for the cap (`:428-461`). (V) | +| 7 | Candidates existed, contrary to the record | **Fixed for the earlier scores; record still incomplete** | The disclosure was committed in `f91d43a1` (05:40:24 UTC) and pushed (05:40:27), before the rules commit `7061200d` (06:04:18). (V, reflog) Still undisclosed: the v2 dry run on TRAIN, and the scoring on DEV since `7061200d` (new finding 3). | +| 8 | Bite checks only through degenerate failures | **Partly fixed** | D1-D6 are run and stored, and the table in §10a matches the artifact. (V) But:
• both summary doses are driven by trivial or degenerate cells, and one is attributed to the wrong cell;
• no AIME dose is reported;
• D1 has a cap artefact.
See new findings 1, 5 and 6. | +| 9 | Unmeasured failure modes | **Fixed, with a new flaw** | `r3`, `q10`/`q50`/`q90` and `partial_aime` added. (V) Three men's `q90` cells are degenerate (new finding 1). | +| 10 | Claims not supported as written | **Fixed** | Section 4.1 reworded. The AIME table now reads 32/20/8 and 34/25/13. The "1.3 to 5.1" claim and the young-worker `wint` claim are gone. The section 8 exclusion claim is true (`first_estimates_birth_evidence.py:356`). The scope sentence is added. (V) | +| 11 | "Conservative" sample size | **Fixed, but one new sentence is now false** | The sentence "Matching units rather than persons works slightly the other way in some bands" no longer holds; see new finding 5(c). | +| 12 | Tolerance as materiality | **Fixed** (recorded in section 13.1; Max's ratification still pending) | (V) | +| 13 | `epuf_matrix` join; `atcap` exactness | **Fixed** | Join and duplicate checks at `:1305-1316`, with a test. The share-of-1 rule maps to exactly the wage base (`:1101`). (V) Minor gap: the annual year range is not asserted (new finding 7). | +| 14 | Dry-run provenance | **Fixed for v0** | `61ade83a` is the parent of `14045be4` in the reflog. (V) The v2 dry run is the same problem again (new finding 3). | +| 15 | Tests | **Mostly fixed** | All five requested tests exist. Several are weaker than their names suggest (new finding 7). | + +## New findings + +### 1. MODERATE: three men's `q90` cells gate while their true value sits at the wage base + +**Evidence.** + +- **True values and tolerances (V).** + + | Cell | Truth | Tolerance | True `atcap` | + |---|---:|---:|---:| + | `odd.men.a22_74.q90` | 1.0 | 0.00345 | 0.1037 | + | `odd.men.a30_44.q90` | 1.0 | 0.00374 | 0.1057 | + | `odd.men.a60_74.q90` | 1.0 | 0.0407 | 0.1002 | + + Truth values are at `floors_v2.json:106, 166, 286`. + +- **The floors are point masses (V).** Most replicates of `odd.men.a22_74.q90` are exactly `0.0`, with occasional jumps of 0.003-0.016 (lines 8792-8993). Sigma here measures how often a sample's at-cap share crosses 10 percent. It is not a sampling error in any useful sense. +- **The cell is a cliff at 10 percent (I).** While more than 10 percent of positive shares are at the cap, `q90` is exactly 1. It measures nothing about the men's upper tail, which is what round 1's finding 9 asked for. Once the share drops below 10 percent, the gap jumps to several tolerances. +- **D2 sits next to the cliff (V).** D2 (marginal draws) lands at `atcap` = 0.1005 for men 22-74, 0.05 points above it, and gets `q90` gap 0 (lines 142900-142943). +- **It drives a reported dose (V).** The stored D3 dose at one tolerance, 0.0035252 (line 152849), equals 0.25 / 70.917. The 70.917 is `odd.men.a22_74.q90` under λ = 0.75 (line 144646). So §10a's sentence "That cell is `atcap`" is wrong: the cell is this degenerate `q90`. + +**Effect on verdicts (I).** Small. + +- **Men 22-74:** the `atcap` cell's own lower bound is 0.1037 · e^−0.0369 = 0.0999. That is the same cliff, so `q90` duplicates `atcap` here. +- **Men 30-44:** the `atcap` bound is 0.1007, above the cliff, so `q90` never binds alone. +- **Men 60-74:** the tolerance of 0.0407 reflects real crossings in samples of 834. +- **A faithful fill:** passes, because its at-cap share tracks the truth's to within about 0.1 points (I). + +The same feature already made `odd.men.a45_59.q90` report-only. It was visible in the undisclosed dry run on TRAIN (new finding 3), and is not tailoring. + +**Required fix (no floor change).** + +- In §10a, disclose next to the `a45_59` note that men's `q90` sits at the cap in the other three bands. In those bands it gates only as a lower bound on the at-cap share near 10 percent, and the gate measures no upper-tail dispersion for men. +- Correct the D3 sentence. +- Recommended for future registrations: compute `q90` among positive shares below the cap. + +### 2. MODERATE: the odd-year fill is scored on pre-career odd years that it never fills in use + +**Evidence.** + +- **The odd fill owns every odd year (V).** `family_mask("odd")` returns every person's five odd years (`:290-291`), and `score_candidate` makes the odd fill own all of them (`:1072, 1088-1092`). +- **That includes pre-career years (I).** People born 1976-1980 have one to three odd years before age 22. For those born 1978-1980, the neighbouring even years are pre-career too, so they are also NaN to the fill. +- **Those values enter gating cells (V).** `partial_aime` sums every year through 2006 (`:385`). So the odd fill's guesses for those years feed `odd.{men,women}.b1966_1980.paime_*` and `.b1946_1980.paime_*`: 20 gating cells. Their tolerances are 0.018-0.078. +- **In use, those years are zero (V).** The assembler fills them under the pre-career rule. + +This is the class of problem behind round 1's finding 2: a fill judged on cells it will never fill in use. The size is probably well under one tolerance (I), but the direction penalises a fill for not imitating the pre-career fill. B1, O1 and D1-D3 have the same scope. + +**Required fix.** + +- Make the odd family's own mask `odd_mask & ~pre_career_mask`, both in `score_candidate` and in the current-rule, oracle and perturbation paths. +- Rerun the bite, oracle and perturbation sections on DEV. Floors, tolerances and partition do not change, because `floor_group` samples true data only. +- Add a test that pre-career odd cells come back NaN and are scored at their true values. + +### 3. MAJOR (record): an undisclosed dry run on TRAIN before the rules commit, and unlogged candidate scoring on DEV since then + +**Evidence.** + +- **The dry run (V).** `.diag/epuf-fill/floors_v2_train_dry.json` has `code_commit` `f91d43a1`, `built_at_utc` 05:48:03, 5 replicates, and uses `psid_scale_v2_try.json`. It was built from an uncommitted working tree 16 minutes before the rules commit `7061200d` (06:04:18, pushed 06:04:21). Its partition already showed `odd.men.a45_59.q90` as `floor_undefined` and the other men's `q90` cells gating. +- **The forks ledger (§12) omits it (V).** Round 1's finding 14 established that dry runs are disclosed as lineage. +- **No log of DEV scoring since the rules commit (V).** + - `dev_score2.py` appends events to `.diag/epuf-fill/dev_events.jsonl`, which does not exist. + - `quick_gaps.py`, `fit_by_sex.py` and `o1_from_train.py` score candidates or oracles on DEV (`part1`) and only print. + - `quick_gaps.py` calls `g.score_candidate`, which exists only in amendment 1's code, with no tolerances. When it first ran is not recorded. If it ran before 06:04, candidate gaps on the new cells were known while the rules were still open (I, unresolved). + +**Required fix, before applying any other fix in this report.** + +- Commit the v2 dry run as lineage and add it to the forks ledger. +- State any rule difference between its working tree and `7061200d`. +- State when `quick_gaps.py` first ran. +- Disclose every candidate DEV score since `7061200d`, as round 1's finding 7 did. The fixes above would otherwise be written with v2 candidate scores knowable. + +### 4. MINOR: no pinned entry point for scoring on TEST (round 1's finding 5, residual) + +**Evidence (V).** + +- `score_candidate` takes `truth`, `tolerance` and `gating` from the caller (`:1044-1055`). Nothing ties them to `runs/epuf_fill_gate_floors_v2.json`. +- Nothing scores the current rule through the same path. +- The test passes empty tolerances and an empty gating list (`test_epuf_fill_gate.py:532-542`). + +**Required fix.** Add a registered `score_test(fill, family)`, or an equivalent lock-record procedure, that: + +1. reads the matrix through `test_part()`; +2. computes the truth with `family_cells` on that same matrix; +3. loads the tolerances and partition from the v2 artifact, checked against a pinned SHA-256; +4. scores the current rule as a fill on the same persons; +5. returns `adoption_tier`. + +Test it on synthetic matrices. + +### 5. MINOR: claims in the proposal that the artifact contradicts + +- **(a) D3 dose.** It comes from `odd.men.a22_74.q90`, not from `atcap` (finding 1). (V) +- **(b) D4 dose and the AIME.** + - The "1.4 percent" dose is `pre.men.b1946_1980.ylevel`: |log 0.9| / 0.01521 = 6.93 tolerances, matching `max_gap_in_tolerances` at line 149139. That is trivially the scale factor. (V) + - A 10 percent cut to every pre-career block moves the pooled median AIME for men by −2.2 percent, or 0.53 tolerances. No AIME cell moves more than 0.61 tolerances (lines 149142-149266). (V) + - Say plainly that the AIME cells cannot see a 10 percent pre-career level bias, and that `plevel`/`ylevel` can. This bears on Max's AIME question. +- **(c) Matching units rather than persons.** Section 5 says it "works slightly the other way in some bands". In v2, DEV has fewer units per person than the PSID in every band, so the matching makes the gate stricter everywhere. (V) Examples: + - men 22-29: 2.45 against 3,949 / 1,462 = 2.70; + - men 22-74: 4.51 against 5.89; + - men 60-74: about equal, 3.20 against 3.19. +- **(d) "None loosens a cell those scores failed" (§12).** This is overstated in two ways. (V) Both changes are forced by round 1's findings and the pre-set events rule, and neither rescues a candidate. Say so. + - The men's young-band `wint` cell gated in v1 (as 18-29), and `odd_knn v0` failed it by 4.1 tolerances. It is now report-only (`too_few_events`). The pooled `a22_74` `wint` still gates. + - The odd-family AIME tolerances widened, as round 1 predicted. For men born 1936-40, the median went from 0.051 to 0.082. +- **(e) `current_odd_fill` docstring.** "The single-neighbour fallback … never applies" is false for units at age 22, whose `t-1` is pre-career. (V) + +### 6. MINOR: D1 copies capped earnings across years with different wage bases + +**Evidence (V).** Copying year `t+1`'s capped earnings into `t` creates shares above 1. The filled `q90` is 1.0103 (line 141235). The scoring path would reject such a fill. This inflates D1's `atcap`, `q90` and `level` failures. + +**Fix.** Copy the share, `x[t+1] / cap[t+1] · cap[t]`. Report-only. + +### 7. MINOR: tests and provenance + +- **(a) No test ties the builder's pool to `group_rows`.** The artifact agrees in all 30 cohort groups (V). Add `pool_size == truth[".*_p50"].n` to `test_sizes_seeds_and_sigmas_recompute`. +- **(b) `BOUND_FILES` is incomplete.** It omits `harness/epuf_cells.py` (`weighted_spearman`, `CellValue`) and `harness/epuf_operator.py` (`wage_base`). Both shape every cell. Add them. +- **(c) The `score_candidate` tests miss the scoring.** They check inputs only. Nothing checks that the scored matrix has the fill's own cells replaced and every other cell true. Nothing checks that a fill writing into the other family's NaN cells is refused. The code does refuse it (`:1093-1098`) (V). +- **(d) A test that checks a constant.** `test_odd_units_start_at_age_22_and_bands_pool` asserts `min(low) == 22` rather than excluding a unit at age 21. +- **(e) No dirty-tree flag.** The builder records `HEAD` but not whether the bound files were dirty at run time. +- **(f) Year range not asserted.** `epuf_matrix` does not check that annual years lie in 1951-2006. A year outside that range would index negatively at `:1327`. + +### 8. NOTES (no fix required) + +- **Floor seeds depend on the order of `groups()`.** Adding groups re-rolled every floor: v1 → v2 tolerances moved by up to about 7 percent. For example, `odd.men.a60_74.atcap` went from 0.1168 to 0.1247 (V). No disclosed result flipped (V). Future amendments should key seeds by group name. +- **Pooled floors sample EPUF's cohort mix.** For men born 1930-45, the pooled DEV pool is 28/29/43 percent across the three cohorts. The PSID's mix is 18/26/55 (V). This is harmless. +- **No pooling across sexes.** A 0.9 sigma bias in the same direction in both sexes' pooled cells is about 1.3 sigma on a both-sex estimate (I). This is a residual limit on the materiality claim, and section 5 does not overclaim it. + +## Ruling on the record (item 4) + +**Chronology (V, reflog and remote log).** + +- Disclosure `f91d43a1`: 05:40:24 UTC. +- Rules `7061200d`: 06:04:18, pushed 06:04:21. +- PSID v2 counts: built at `7061200d` (artifact `code_commit`). +- `23bee82c`: committed 06:09:30, pushed 06:09:52. +- v2 floor build: started 06:10:08 at `23bee82c`. +- `f8623f4d`: pushed 06:47:36. +- The candidates branch `epuf-career-fill-candidates-20261004` has no commits. + +I could not check that the rules are identical at `7061200d` and `23bee82c` (I). + +**Tailoring.** No amendment looks tailored to the disclosed scores. + +- **They mostly work against the disclosed candidates.** + - `pre_donor v1`'s `yr_cross` gaps are positive in all six cohort-sex cells, +0.018 to +0.028. The new pooled tolerances are 0.0146 (men) and 0.0157 (women). (V) So it would probably fail the pooled cells (I). + - Its one failure, women 1940-45 `pzero`, now has a tighter tolerance: 0.0515 → 0.0470 (V). + - The union mask removes the true odd years it used to read. + - The cap on "improves" demotes `odd_quantile v1`, whose `r2` gaps were about 5.7 tolerances (I). +- **The loosenings were forced.** The odd-family AIME tolerances and the young men's `wint` cell changed because of round 1's findings 1 and 2. The candidates' odd AIME gaps were at most 0.004 (V). +- **The record still needs new finding 3.** + +## Ruling on the tolerance (item 5) + +**`tau = 1 * sigma` at the PSID's size is acceptable to lock as a materiality threshold.** + +- **The verdict is close to a step at sigma.** The comparison's own noise is at most sqrt(1.05) times 0.087-0.212 sigma (stored noise ratios, V). So a fill with no bias fails essentially never, and a fill with bias near sigma can go either way. +- **The materiality bound is correct.** At the boundary, the root mean square error of a PSID-sized estimate built on the fill is √2 times the PSID's own sampling error. +- **The floors are strict.** They use unweighted counts, so they are tighter than the PSID's real sampling error. +- **The pooled groups carry the bound to aggregates** across cohorts and bands, within each sex. + +Max should ratify with three facts in view: + +1. **√2 means up to 41 percent more error.** At the boundary, the fill can add up to 41 percent to the root mean square error. +2. **Single-cohort AIME tail tolerances are very wide.** `pre.women.b1930_1934.aime_p10` has a tolerance of 0.391 log, so a 48 percent overstatement passes. The pooled groups carry the AIME materiality: their p10 tolerances are 0.17. +3. **Estimates pooled across both sexes are not gated directly.** + +## What I verified + +- **Rules code.** I read all of `epuf_fill_gate.py`: + - the masks, union mask, universes and `group_rows`; + - the band, `r3`, quantile, AIME and partial-AIME cells; + - `floor_group`, `partition`, `score`, `adoption_tier`, `score_candidate`; + - the oracles, `epuf_matrix` and `test_part`. + + `r3` uses only recorded even years (2008 excluded). Universes are invariant to their own family's fill. +- **Builders.** Sample sizes, pools, seeds and bite logic. The dose algebra takes the smallest linear dose over finite cells. +- **PSID scale v2 additivity.** + - Men's bands: 3,949 + 11,148 + 9,472 + 2,670 = 27,239. + - Women's bands: 5,318 + 13,927 + 10,663 + 2,773 = 32,681. + - Every pooled cohort count equals the sum of its parts. +- **v2 floors recomputed by hand.** + - All 40 groups: n = PSID count / units per person, for example 3,949 / 2.4543 = 1,609. + - All 30 cohort groups: `pool_size` equals the truth's `n`. + - Spot-checked noise ratios, e.g. sqrt(1,609 / 65,387) = 0.1569; the range is 0.087-0.212. +- **Partition.** 326 cells: 190 `odd`, 136 `pre`. 319 gate. The 7 report-only cells match §10a. Against v1, the report-only changes are men's young-band `atcap` (now gates) and men's young-band `wint` (now report-only). +- **Bite and results counts.** B1 97/183, B2 112/136. B2's pooled median AIME: men 1,813 / 2,266 (−20 percent, 5.27 tolerances), women 601 / 771 (−22 percent, 4.01). O1 32, O2 6. D1-D6 failing and beyond-two counts all match §10a. The total of `"passes": false` across all scored results, 822, is consistent. +- **Disclosure file.** All five events, and the v1-vs-v2 tolerance comparison for every cell a disclosed candidate failed. +- **Leakage.** There is no TEST read path besides `test_part` (searched with grep). The cache holds only `part0`/`part1`. The candidate module reads no EPUF. +- **Tests.** I read both files in full. I did not run them. \ No newline at end of file diff --git a/reviews/gate_epuf_fill_round3_confirmation_20261004.md b/reviews/gate_epuf_fill_round3_confirmation_20261004.md new file mode 100644 index 00000000..130d01b3 --- /dev/null +++ b/reviews/gate_epuf_fill_round3_confirmation_20261004.md @@ -0,0 +1,151 @@ +**Verdict: LOCK AFTER LISTED FIXES** + +Round 2's eight findings are fixed or disclosed. The v3 floors match v2's, and the TEST entry point is pinned and reads TEST only through the lock. Three things need fixing before lock, and none of them needs a new floor build: + +- **New finding 1:** one round-2 fix widened the "improves" allowance on a cell where the best disclosed odd-year candidate was blocked. The proposal doesn't say so. +- **New finding 2:** the record of DEV scores isn't closed. A newer candidate has been checked against DEV and isn't logged. +- **New finding 3:** small gaps in the TEST entry point. + +**What I could not do.** This session had file-reading tools only, with no shell. So: + +- **Tests not run.** I did not run the three test files. What I say about tests comes from reading them. +- **No `git diff` or `git show`.** I took commit times from the reflog, read as plain files under `.git/logs`. +- **No EPUF microdata.** I read none, including the `cache/*.npz` files, and called no EPUF reader. + +Each claim is marked **(V)** for verified (I read it, or did the arithmetic on stored values) or **(I)** for inferred. + +## Round 2's findings + +| # | Finding | Status | Evidence | +|---|---|---|---| +| 1 | Men's `q90` sits at the wage base | **Disclosed** | §10a discloses the 22-74, 30-44 and 60-74 bands and the "lower bound on the at-cap share" reading. It recommends computing `q90` below the cap in future. (V) The D3 sentence is corrected: v3's artifact names `odd.men.a22_74.q90` as D3's dose cell (`floors_v3.json:152851`). (V) | +| 2 | Odd fill scored on pre-career odd years | **Fixed** | `family_mask("odd")` is now `odd_mask & ~pre_career_mask` (`epuf_fill_gate.py:302-303`). (V) `score_candidate` uses it (`:1152`). (V) The perturbations restore every non-owned cell to its true value (`build_epuf_fill_gate_floors.py:211-212`), and O1 is masked to the family's own cells (`:445-453`). (V) Tests: `test_odd_family_owns_only_career_odd_years` and `test_score_candidate_scores_own_cells_and_keeps_the_rest_true` (V, read only). Bite, perturbations and oracles were rerun in v3. (V) | +| 3 | Undisclosed TRAIN dry run; unlogged DEV scoring | **Fixed, with a residual** | `runs/epuf_fill_gate_floors_train_dryrun_v2pre.json` is committed. Its header matches `.diag/.../floors_v2_train_dry.json` (commit `f91d43a1`, 05:48:03 UTC, 5 replicates). (V) It is in the forks ledger, with the paragraph in §12. (V) The time `quick_gaps.py` was written (06:18:42 UTC) is stated. (V) Eight events after `7061200d` are disclosed in `..._after_amendment_1.json`, and three in `..._after_round_2.jsonl`. The jsonl is identical to `.diag/epuf-fill/dev_events.jsonl`. (V) Residuals:
• "No change to the rules module or the builders is recorded" says nothing was recorded; it does not say nothing changed.
• The dry run used a different set of PSID counts (`psid_scale_v2_try.json`, SHA-256 `e2a175…`, not the registered `b35e12…`), and the disclosure doesn't mention it. (V)
• Later DEV checks are unlogged; see new finding 2. | +| 4 | No pinned TEST entry point | **Fixed** (minor gaps: new finding 3) | `score_registered` (`epuf_fill_scoring.py:83-154`) does each step round 2 asked for:
• loads v3 and checks it against a pinned SHA-256 (`d403a824…`), and requires `registered_build` and `lockable` (V);
• reads TEST through `g.test_part` (V);
• computes the truth with `family_cells` on that same matrix (V);
• scores `current_rule(family)` through `score_candidate` (V);
• returns the tiers and `adopt` (V).
`test_the_scoring_module_pins_the_registered_build` checks the hash. (V, read only) | +| 5 | Claims the artifact contradicts | **Fixed** | (a) D3 dose cell corrected (V). (b) §10a now says plainly that the AIME cells can't see a 10 percent pre-career bias, while `plevel`/`ylevel` can, and gives the AIME doses (12.5 and 13.5 percent, matching `:152854, 152862`) (V). (c) §5 now says matching units makes the gate stricter in every band (V). (d) §12 now admits the two loosenings (V). (e) The `current_odd_fill` docstring is corrected (V). | +| 6 | D1 copied capped dollars | **Fixed** | D1 now copies the share (`builder:233-239`). (V) No filled value above 1 remains in v3; v2 had 4. (V) D1 now fails 63 of 183 cells, down from 92. (V) | +| 7 | Tests and provenance | **Fixed** | (a) Pool equals the cells' `n` for cohort groups (`test_epuf_fill_gate_floors.py:151-159`). (b) `BOUND_FILES` now includes `epuf_cells.py` and `epuf_operator.py`. (c) Tests that the scored matrix keeps every non-owned cell true, and that a fill writing into the other family's cells is refused. (d) `test_a_masked_year_at_age_21_is_no_odd_unit`. (e) `bound_files_clean` is recorded and is `true` in v3. (f) The annual year range is asserted (`:1394-1397`). All (V) by reading. | +| 8 | Notes | **Recorded** | §13 records all of them. (V) | + +## The floors are unchanged + +- **Same structure (V).** Every top-level section of v3 spans exactly the same number of lines as in v2, shifted by one. The extra line is the new `bound_files_clean` key. + - truth: 1,632 lines; + - groups: 132,882 lines; + - partition: 328 lines; + - tolerance: 321 lines. +- **Tolerances (V).** I read both tolerance blocks in full: 319 entries, matching entry for entry by eye. +- **Partition (V).** The seven report-only cells and their reasons are identical. +- **Truth (V for samples).** The values I sampled are identical. +- **Group records (I).** I did not compare the replicates value by value. The identical extents and `test_registered_floors_equal_the_v2_floors_exactly` (read, not run) support equality. +- **Unchanged inputs (V).** The EPUF and PSID-scale hashes and the constants are unchanged. +- **What changed (V).** Only the fill scoring: + - B1: 97 cells beyond two tolerances became 96, and 99 failing became 100; + - D1-D3 counts moved; + - B2, O1 (32 failing) and O2 (6 failing) are unchanged. + +## New findings + +### 1. MODERATE: the change to `CurrentOddFill` moved the current rule's gaps, and so the "improves" allowances, in the young bands. One loosened cell is where the best disclosed odd candidate was blocked. + +**Evidence.** + +- **The current rule's gaps moved between v2 and v3 (V).** In v2 the current rule averaged the true age-21 year at age 22. In v3 it is a fill that sees age 21 as unknown and copies `t+1`. Its gaps in tolerances: + + | Cell | v2 | v3 | Allowance effect | + |---|---:|---:|---| + | `odd.women.a22_29.level` | 0.11 | 1.23 | looser: 1.00 → 1.23 τ | + | `odd.men.a22_29.level` | 0.31 | 1.09 | looser: 1.00 → 1.09 τ | + | `odd.men.a22_29.q10` | 1.47 | 0.60 | tighter | + | `odd.men.a22_29.q50` | 3.79 | 1.94 | tighter | + | `odd.men.a22_29.q90` | 2.62 | 2.01 | tighter | + | `odd.men.a22_29.r3` | 3.64 | 2.63 | tighter | + | `odd.men.a22_29.zexit` | ∞ | 36.4 | none (still above 3) | + + The rows come from `floors_v2.json:135206-135294, 136049` and `floors_v3.json:135207-135295, 136050`. + +- **Where the disclosed candidate stood.** + - **When it was scored (V).** Event 8 (`odd_qrf_sex2`, 07:05 UTC) was scored against v2 tolerances. The round-2 fix commit `cb76ad15` came at 07:16:52 UTC (reflog epoch 1791098212), so these scores were known while the fixes were being written. + - **The binding cell (V).** Event 8 fails `odd.women.a22_29.level` by +0.0203 against a tolerance of 0.0185, or 1.10 τ. + - Under v2's current rule the allowance there is 1.00 τ, so the cell blocks "improves". + - Under v3's it is 1.23 τ, so the cell no longer blocks. + - **Its other binding cell (V).** `odd.women.b1966_1980.paime_p25` fails by 1.015 τ against a current gap of 0. Finding 2's fix touches exactly that cohort, so it may also move this cell. I can't tell from the record, because no odd candidate has been scored on DEV since the fixes. +- **Not clearly tailoring (I).** + - The change follows from round 2's findings 4 and 5(e), and it is applied evenly. + - Its effects go both ways: four cells are tighter and two looser. + - It affects only the "improves" tier, not certification. +- **The "faithful" claim is only partly true (V).** + - `build_career` takes any observed PSID year from 1968 on as a neighbour (`career.py:1010-1031, 968-969`), with no age filter. + - `family_earnings_panel` keeps heads and spouses at any age (`data/family.py:790-800`). + - So the assembler uses the two-sided mean at age 22 whenever the person was a PSID head or spouse at 21. It falls back to one neighbour only when that row is missing. + - §3 and the `CurrentOddFill` docstring say the fallback is what `_impute_gap` does, without that qualification. + +**Required fix (no rebuild).** + +- In §12a, disclose: + - the v2→v3 change in the current rule's young-band gaps (the table above); + - its effect on the improves allowances; + - that the change was made while event 8's cell-level failures were known. +- Qualify the faithfulness claim in §3 and the docstring. +- Then do one of these: + - **(a)** Report the share of PSID-2010 age-22 filled units that have an observed age-21 year. This is computable from the PSID with no EPUF read. Have Max accept the always-fallback choice. + - **(b)** Register a neutral rule: where the always-mean and always-fallback current rules differ, the improves allowance uses the smaller of the two current gaps. + +### 2. MINOR (record): the DEV disclosure isn't closed + +**Evidence.** + +- **A newer candidate exists (V).** `.diag/epuf-fill/cache/pre_chain3.npz` is there. +- **It has been checked against DEV truth (V).** `.diag/epuf-fill/debug_chain.py` fills DEV persons (`part1`) with it and prints the fill's mean share and zero rate against the truth at ages 15-22. Those are the `ylevel`/`yzero` statistics, which gate. +- **It isn't logged (V).** Neither `dev_events.jsonl` nor the committed `..._after_round_2.jsonl` mentions it. Yet the header says every DEV score is disclosed. +- **Timing (I).** It probably ran after `7eab0914` (07:34:31 UTC). +- **Uncommitted code (V).** The disclosed `code_sha256` values point at `src/populace_dynamics/estimates/epuf_fill.py`, which is untracked, so those bytes can't be audited later. + +DEV development is allowed, and the floors are frozen, so this threatens no floor (I). It matters only if the rules change again. + +**Fix.** + +- Log `pre_chain3` and every other DEV check up to lock, and close the log in the lock record with a timestamp. +- Make the header's claim "through