From 93122f5c9aa9d2c433dba7805558eedb14661d9a Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 04:30:54 -0400 Subject: [PATCH 01/13] EPUF career fills: the four candidates and their PSID-2010 application estimates/epuf_fill.py holds gate_epuf_fill's candidate fills, fitted on EPUF TRAIN only: odd years by a two-part random-forest draw (a probability forest for a zero year, a quantile regression forest for the positive share) per sex with a person-level copula calibrated on held-out TRAIN persons, against kNN triples; pre-career years by rank-kNN donor careers, against a chained one-sided draw. Artifacts are byte-reproducible npz files staged outside git and loaded by SHA-256. cohorts/psid2010_epuf_fill.py applies them to a built PSID-2010 cohort's careers behind new provenance values (gap_epuf_drawn, pre_career_epuf_donor); career.py and the registered runs are untouched. Both modules are excluded from the birth-evidence reducer's seal. The DEV log and kept code bytes are updated. Co-Authored-By: Claude Opus 5.5 --- .../epuf_fill_36e045bdc946.py | 1942 +++++++++++++++ .../epuf_fill_565e3ba2717b.py | 2066 ++++++++++++++++ .../epuf_fill_5af343166b51.py | 2127 ++++++++++++++++ .../epuf_fill_c5e4a37417ec.py | 2131 +++++++++++++++++ ...e_epuf_fill_dev_scores_after_round_2.jsonl | 11 + scripts/first_estimates_birth_evidence.py | 4 + scripts/fit_epuf_fills.py | 225 ++ .../cohorts/psid2010_epuf_fill.py | 191 ++ src/populace_dynamics/estimates/epuf_fill.py | 1665 +++++++++++++ tests/cohorts/test_psid2010_epuf_fill.py | 158 ++ .../estimates/test_birth_evidence_artifact.py | 4 + tests/estimates/test_epuf_fill.py | 159 ++ 12 files changed, 10683 insertions(+) create mode 100644 docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_36e045bdc946.py create mode 100644 docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_565e3ba2717b.py create mode 100644 docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_5af343166b51.py create mode 100644 docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_c5e4a37417ec.py create mode 100644 scripts/fit_epuf_fills.py create mode 100644 src/populace_dynamics/cohorts/psid2010_epuf_fill.py create mode 100644 src/populace_dynamics/estimates/epuf_fill.py create mode 100644 tests/cohorts/test_psid2010_epuf_fill.py create mode 100644 tests/estimates/test_epuf_fill.py diff --git a/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_36e045bdc946.py b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_36e045bdc946.py new file mode 100644 index 00000000..bece8a62 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_36e045bdc946.py @@ -0,0 +1,1942 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddQuantileFill` (odd years, primary): a two-part conditional + draw. The probability of a zero year and the conditional quantiles of a + positive share, relative to the neighbours' level, by sex, age, the + shares at ``t-1`` and ``t+1`` and the context at ``t-3``, ``t+3`` and + further; a Gaussian AR(1) copula correlates a person's draws across + masked years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "OddQuantileFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _invert(table: np.ndarray, rows: np.ndarray, value: np.ndarray): + """The level at which each row's quantile function reaches ``value``. + + Where the function is flat at ``value`` (a run of equal quantiles), the + middle of the run's levels. + """ + + points = table.shape[1] + levels = np.linspace(0.0, 1.0, points) + out = np.empty(len(rows)) + for start in range(0, len(rows), 200_000): + block = slice(start, start + 200_000) + curve = table[rows[block]].astype(np.float64) + target = np.asarray(value[block], dtype=np.float64)[:, None] + below = (curve < target).sum(axis=1) + above = (curve <= target).sum(axis=1) + flat = below < above + result = np.empty(len(curve)) + result[above == 0] = 0.0 + result[below >= points] = 1.0 + middle = flat & (above > 0) & (below < points) + result[middle] = 0.5 * ( + levels[below[middle]] + levels[above[middle] - 1] + ) + between = ~flat & (below > 0) & (below < points) + index = np.flatnonzero(between) + left = curve[index, below[index] - 1] + right = curve[index, below[index]] + share = np.where( + right > left, + (target[index, 0] - left) + / np.where(right > left, right - left, 1), + 0.5, + ) + result[index] = levels[below[index] - 1] + share * ( + levels[below[index]] - levels[below[index] - 1] + ) + out[block] = result + return out + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + buffer = io.BytesIO() + np.savez_compressed(buffer, **arrays) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: the two-part conditional draw +# -------------------------------------------------------------------------- +_N_SHARE_BINS = 20 + + +def _share_bin(value: np.ndarray, edges: np.ndarray) -> np.ndarray: + """0 zero, 1..20 quantile bins of a positive share below the cap, 21 cap.""" + + bins = np.searchsorted(edges, value, side="right") + 1 + bins = np.where(value <= 0, 0, bins) + return np.where(value >= 1.0, _N_SHARE_BINS + 1, bins) + + +def _coarse_age(band: np.ndarray) -> np.ndarray: + """Age bands grouped: under 30, 30-44, 45-59, 60 and over.""" + + return np.digitize(band, [4, 7, 10]) + + +@dataclass(frozen=True) +class OddQuantileFill: + """Two-part conditional draw for masked odd years, with an AR(1) copula. + + A unit's reference level ``m`` is the geometric mean of its positive + neighbours' shares (the one positive neighbour's share if only one is, + 1 if neither is). Its cell is the finest of seven nested keys with at + least ``MIN_CELL`` TRAIN units, built from sex, five-year age band, the + bins of the shares at ``t-1`` and ``t+1`` (zero, 20 quantile bins of a + positive share below the cap, at the cap), the context at ``t-3`` and + ``t+3`` (missing, zero, below or above the median positive share), and + whether any share at offsets 5, 7 or 9 is positive. In the cell: ``p0`` + the share of zero years, and 65 quantiles of ``log(x_t / m)`` among + positive years. A uniform ``u`` maps to zero if ``u < p0``, else to + ``min(m * exp(Q((u - p0) / (1 - p0))), 1)``. The uniforms of a person's + consecutive masked years (two years apart) are joined by a Gaussian + AR(1) copula with correlation ``rho`` by sex and age band, learned on + TRAIN from the probability integral transforms of consecutive units. + """ + + share_edges: np.ndarray + context_median: float + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + rho: np.ndarray + stream: str = "epuf_fill.odd_quantile.v1" + name: str = "odd_quantile" + + # -- keys --------------------------------------------------------------- + @staticmethod + def _parts(context: OddContext, share_edges, median): + left = np.nan_to_num(context.left, nan=-1.0) + right = np.nan_to_num(context.right, nan=-1.0) + # A missing neighbour takes the other's value (the PSID fallback). + left = np.where(left < 0, right, left) + right = np.where(right < 0, left, right) + positive_left = np.where(left > 0, left, 1.0) + positive_right = np.where(right > 0, right, 1.0) + level = np.where( + (left > 0) & (right > 0), + np.sqrt(positive_left * positive_right), + np.where(left > 0, positive_left, positive_right), + ) + + def context_code(value): + return np.where( + np.isnan(value), + 0, + np.where(value <= 0, 1, np.where(value < median, 2, 3)), + ) + + return { + "sex": context.sex, + "age": _age_band(context.age), + "left_bin": _share_bin(left, share_edges), + "right_bin": _share_bin(right, share_edges), + "context3": 4 * context_code(context.left3) + + context_code(context.right3), + "wide": context.wide, + "level": level, + "valid": ~(np.isnan(context.left) & np.isnan(context.right)), + } + + @staticmethod + def _keys(parts) -> list[np.ndarray]: + sex = parts["sex"] + age = parts["age"] + coarse = _coarse_age(age) + left = parts["left_bin"] + right = parts["right_bin"] + context3 = parts["context3"] + wide = parts["wide"] + + # Nested keys from finest to coarsest; a dropped component is held + # at a sentinel (age 16-20 marks the coarse bands, 21 none). + def key(s, a, lb, rb, c3, w): + return ((((s * 22 + a) * 23 + lb) * 23 + rb) * 17 + c3) * 3 + w + + return [ + key(sex, age, left, right, context3, wide), + key(sex, age, left, right, context3, 2), + key(sex, age, left, right, 16, 2), + key(sex, 16 + coarse, left, right, 16, 2), + key(sex, 21, left, right, 16, 2), + key(0 * sex, 21, left, right, 16, 2), + key(0 * sex, 21, np.minimum(left, 1), np.minimum(right, 1), 16, 2), + ] + + def _cells(self, parts) -> tuple[np.ndarray, np.ndarray]: + """(level, row) of each unit's finest populated cell.""" + + keys = self._keys(parts) + level = np.full(len(keys[0]), -1, dtype=np.int64) + row = np.full(len(keys[0]), -1, dtype=np.int64) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + if (level < 0).any(): + raise ValueError("a unit has no populated cell at any level") + return level, row + + def _quantile(self, parts, u: np.ndarray) -> np.ndarray: + level, row = self._cells(parts) + out = np.zeros(len(u)) + for index in np.unique(level): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate(self.level_quantiles[index], row[take], v) + share = np.minimum(parts["level"][take] * np.exp(residual), 1.0) + out[take] = np.where(positive, share, 0.0) + return out + + # -- fitting -------------------------------------------------------------- + @classmethod + def fit( + cls, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + unit_years: tuple[int, ...], + rho_seed: int = 0, + ) -> tuple[OddQuantileFill, dict[str, object]]: + """Fit on complete TRAIN shares; every year of ``unit_years`` a unit. + + Every person-year of ``unit_years`` whose two neighbours are inside + the matrix is a training unit (all years are recorded on TRAIN). + """ + + shares = np.asarray(shares, dtype=np.float64) + known = np.isfinite(shares) + n = len(shares) + rows_list, years_list = [], [] + for year in unit_years: + rows_list.append(np.arange(n)) + years_list.append(np.full(n, year)) + rows = np.concatenate(rows_list) + unit_year = np.concatenate(years_list) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + target = _take(shares, rows, _column_of(years, unit_year)) + neighbours = np.concatenate([context.left, context.right]) + inside = neighbours[(neighbours > 0) & (neighbours < 1.0)] + share_edges = np.quantile( + inside, np.linspace(0, 1, _N_SHARE_BINS + 1)[1:-1] + ) + median = float(np.median(inside)) + parts = cls._parts(context, share_edges, median) + keys = cls._keys(parts) + positive = target > 0 + residual = np.log(np.where(positive, target, 1.0)) - np.log( + parts["level"] + ) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zeros_unique, zeros = np.unique( + key[in_cells & ~positive], return_counts=True + ) + totals = count[count >= MIN_CELL] + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zeros_unique)] = zeros + p0 = p0 / totals + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + # A populated cell with no positive unit draws only zeros. + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0.astype(np.float64)) + level_quantiles.append(full) + provisional = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=np.zeros((4, 16)), + ) + rho, rho_diagnostics = provisional._fit_rho( + parts, target, rows, unit_year, rho_seed + ) + fill = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=rho, + ) + cells = [len(k) for k in level_keys] + return fill, { + "n_units": int(len(target)), + "cells_per_level": cells, + **rho_diagnostics, + } + + def _pit(self, parts, target, rows, unit_year, seed) -> np.ndarray: + """Randomised probability integral transforms of true shares.""" + + level, row = self._cells(parts) + jitter = hash_uniform( + "epuf_fill.odd_quantile.pit", seed, rows, unit_year + ) + out = np.empty(len(target)) + for index in np.unique(level): + take = np.flatnonzero(level == index) + p0 = self.level_p0[index][row[take]] + zero = target[take] <= 0 + out[take[zero]] = jitter[take[zero]] * p0[zero] + positive = take[~zero] + residual = np.log(target[positive]) - np.log( + parts["level"][positive] + ) + # At the cap the residual is censored: spread it over the mass + # the quantile function puts at or above the cap. + at_cap = target[positive] >= 1.0 + v = _invert(self.level_quantiles[index], row[positive], residual) + cap_v = v.copy() + cap_v[at_cap] = v[at_cap] + jitter[positive][at_cap] * ( + 1.0 - v[at_cap] + ) + p0_positive = p0[~zero] + out[positive] = p0_positive + (1.0 - p0_positive) * cap_v + return np.clip(out, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, parts, target, rows, unit_year, seed): + """AR(1) correlation of consecutive units' normal scores (t, t+2). + + On a 5 percent sample of persons (by seed): every unit's + probability integral transform under the fitted cells, its normal + score, and the correlation of the scores of ``t`` and ``t+2`` for + the same person, by sex and age band at ``t``. + """ + + persons = np.unique(rows) + rng = np.random.default_rng(seed) + chosen = persons[rng.random(len(persons)) < 0.05] + index = np.flatnonzero(np.isin(rows, chosen)) + sub = {k: v[index] for k, v in parts.items()} + z = ndtri( + self._pit(sub, target[index], rows[index], unit_year[index], seed) + ) + first_year = int(unit_year.min()) + n_years = int(unit_year.max()) - first_year + 1 + position = np.searchsorted(chosen, rows[index]) + grid = np.full((len(chosen), n_years), np.nan) + grid[position, unit_year[index] - first_year] = z + sex_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + sex_grid[position, unit_year[index] - first_year] = sub["sex"] + age_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + age_grid[position, unit_year[index] - first_year] = sub["age"] + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex = sex_grid[:, :-2].ravel() + age = age_grid[:, :-2].ravel() + both = np.isfinite(now) & np.isfinite(later) + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = both & (sex == s) & (age == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + overall = float(np.corrcoef(now[both], later[both])[0, 1]) + four_now = grid[:, :-4].ravel() + four_later = grid[:, 4:].ravel() + four = np.isfinite(four_now) & np.isfinite(four_later) + lag4 = float(np.corrcoef(four_now[four], four_later[four])[0, 1]) + return rho, { + "rho_persons": int(len(chosen)), + "rho_pairs": int(both.sum()), + "rho_overall": overall, + "lag4_normal_score_correlation": lag4, + "lag4_ar1_prediction": overall**2, + } + + # -- filling -------------------------------------------------------------- + def fill( + self, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + person_key: np.ndarray, + fill_mask: np.ndarray, + seed: int, + ) -> np.ndarray: + """Fill the masked cells; masked cells with no known neighbour stay NaN.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + parts = self._parts(context, self.share_edges, self.context_median) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + rho = self.rho[np.clip(parts["sex"], 0, 3), parts["age"]] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + valid = parts["valid"] + drawn = np.full(len(rows), np.nan) + if valid.any(): + sub = {k: v[valid] for k, v in parts.items()} + drawn[valid] = self._quantile(sub, ndtr(z[valid])) + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + # -- persistence ---------------------------------------------------------- + def to_bytes(self) -> bytes: + arrays = { + "kind": np.array(self.name), + "share_edges": self.share_edges, + "context_median": np.array(self.context_median), + "rho": self.rho, + } + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> OddQuantileFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + share_edges=arrays["share_edges"], + context_median=float(arrays["context_median"]), + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + rho=arrays["rho"], + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: a quantile regression forest (QRF) draw +# -------------------------------------------------------------------------- + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +#: The reference level of a unit with no positive share around it. +_DEFAULT_LEVEL = 0.3 + + +def reference_level(context: OddContext) -> np.ndarray: + """The level a unit's share is drawn relative to. + + The geometric mean of the positive shares at ``t-1`` and ``t+1``; else + of those at ``t-3`` and ``t+3``; else the mean positive share at + offsets 5-9; else 0.3. + """ + + def geometric(a, b): + a = np.nan_to_num(a, nan=0.0) + b = np.nan_to_num(b, nan=0.0) + both = (a > 0) & (b > 0) + one = np.where(a > 0, a, b) + value = np.where(both, np.sqrt(np.where(both, a * b, 1.0)), one) + return np.where((a > 0) | (b > 0), value, np.nan) + + level = geometric(context.left, context.right) + level = np.where( + np.isnan(level), geometric(context.left3, context.right3), level + ) + level = np.where( + np.isnan(level) & (context.wide_mean > 0), context.wide_mean, level + ) + return np.where(np.isnan(level), _DEFAULT_LEVEL, level) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.61, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + A random forest (scikit-learn; split target ``log(share + 0.01)``) + partitions TRAIN units by :func:`odd_features`. Every TRAIN unit used + in the fit is passed down every tree, and each leaf keeps the sorted + true shares of the units that reach it (zeros and the cap included, + stored as shares times 65,535). The forest's conditional law of a unit + is the average over trees of its leaves' empirical laws, so the draw is + two-part by construction: zero, the cap and every share between keep + their own mass. A draw picks a tree by one seeded uniform and takes the + leaf's value at the quantile of a second, the copula uniform. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + stream: str = "epuf_fill.odd_forest.v3" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + known = np.isfinite(shares) + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x = odd_features(context) + y = target[chosen] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y + 0.01)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + drawn[valid] = self._value(unit["leaves"][valid], ndtr(z[valid])) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + mask &= years[None, :] >= start[:, None] + given = np.where(mask, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +_FOREST_ARRAYS = ( + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + known = np.isfinite(shares) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum): + members = np.flatnonzero(stratum == value) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=context.left[keep].astype(np.float32), + bank_right=context.right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:_DONOR_BANK] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + -1 for rows with no masked cell or no bank donor of their sex and + birth year. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + out = np.full(len(shares), -1, dtype=np.int64) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + recipients = targets[ + (sex[targets] == s) & (birth_year[targets] == b) + ] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + column = np.sort( + donor_match[:, j][np.isfinite(donor_match[:, j])] + ) + ranks_donor[:, j] = np.where( + np.isfinite(donor_match[:, j]), + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + nearest = np.argpartition(distance, k - 1, axis=1)[:, :k] + nearest_distance = np.take_along_axis(distance, nearest, 1) + order = np.lexsort((nearest, nearest_distance), axis=1) + nearest = np.take_along_axis(nearest, order, 1) + pick = np.minimum( + (u[recipients[block]] * k).astype(np.int64), k - 1 + ) + no_match = count.max(axis=1) == 0 + random_donor = np.minimum( + (u[recipients[block]] * len(donors)).astype(np.int64), + len(donors) - 1, + ) + out[recipients[block]] = donors[ + np.where( + no_match, + random_donor, + nearest[np.arange(len(nearest)), pick], + ) + ] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_quantile": OddQuantileFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_565e3ba2717b.py b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_565e3ba2717b.py new file mode 100644 index 00000000..fde0bfa1 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_565e3ba2717b.py @@ -0,0 +1,2066 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddQuantileFill` (odd years, primary): a two-part conditional + draw. The probability of a zero year and the conditional quantiles of a + positive share, relative to the neighbours' level, by sex, age, the + shares at ``t-1`` and ``t+1`` and the context at ``t-3``, ``t+3`` and + further; a Gaussian AR(1) copula correlates a person's draws across + masked years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "OddQuantileFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _invert(table: np.ndarray, rows: np.ndarray, value: np.ndarray): + """The level at which each row's quantile function reaches ``value``. + + Where the function is flat at ``value`` (a run of equal quantiles), the + middle of the run's levels. + """ + + points = table.shape[1] + levels = np.linspace(0.0, 1.0, points) + out = np.empty(len(rows)) + for start in range(0, len(rows), 200_000): + block = slice(start, start + 200_000) + curve = table[rows[block]].astype(np.float64) + target = np.asarray(value[block], dtype=np.float64)[:, None] + below = (curve < target).sum(axis=1) + above = (curve <= target).sum(axis=1) + flat = below < above + result = np.empty(len(curve)) + result[above == 0] = 0.0 + result[below >= points] = 1.0 + middle = flat & (above > 0) & (below < points) + result[middle] = 0.5 * ( + levels[below[middle]] + levels[above[middle] - 1] + ) + between = ~flat & (below > 0) & (below < points) + index = np.flatnonzero(between) + left = curve[index, below[index] - 1] + right = curve[index, below[index]] + share = np.where( + right > left, + (target[index, 0] - left) + / np.where(right > left, right - left, 1), + 0.5, + ) + result[index] = levels[below[index] - 1] + share * ( + levels[below[index]] - levels[below[index] - 1] + ) + out[block] = result + return out + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + buffer = io.BytesIO() + np.savez_compressed(buffer, **arrays) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: the two-part conditional draw +# -------------------------------------------------------------------------- +_N_SHARE_BINS = 20 + + +def _share_bin(value: np.ndarray, edges: np.ndarray) -> np.ndarray: + """0 zero, 1..20 quantile bins of a positive share below the cap, 21 cap.""" + + bins = np.searchsorted(edges, value, side="right") + 1 + bins = np.where(value <= 0, 0, bins) + return np.where(value >= 1.0, _N_SHARE_BINS + 1, bins) + + +def _coarse_age(band: np.ndarray) -> np.ndarray: + """Age bands grouped: under 30, 30-44, 45-59, 60 and over.""" + + return np.digitize(band, [4, 7, 10]) + + +@dataclass(frozen=True) +class OddQuantileFill: + """Two-part conditional draw for masked odd years, with an AR(1) copula. + + A unit's reference level ``m`` is the geometric mean of its positive + neighbours' shares (the one positive neighbour's share if only one is, + 1 if neither is). Its cell is the finest of seven nested keys with at + least ``MIN_CELL`` TRAIN units, built from sex, five-year age band, the + bins of the shares at ``t-1`` and ``t+1`` (zero, 20 quantile bins of a + positive share below the cap, at the cap), the context at ``t-3`` and + ``t+3`` (missing, zero, below or above the median positive share), and + whether any share at offsets 5, 7 or 9 is positive. In the cell: ``p0`` + the share of zero years, and 65 quantiles of ``log(x_t / m)`` among + positive years. A uniform ``u`` maps to zero if ``u < p0``, else to + ``min(m * exp(Q((u - p0) / (1 - p0))), 1)``. The uniforms of a person's + consecutive masked years (two years apart) are joined by a Gaussian + AR(1) copula with correlation ``rho`` by sex and age band, learned on + TRAIN from the probability integral transforms of consecutive units. + """ + + share_edges: np.ndarray + context_median: float + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + rho: np.ndarray + stream: str = "epuf_fill.odd_quantile.v1" + name: str = "odd_quantile" + + # -- keys --------------------------------------------------------------- + @staticmethod + def _parts(context: OddContext, share_edges, median): + left = np.nan_to_num(context.left, nan=-1.0) + right = np.nan_to_num(context.right, nan=-1.0) + # A missing neighbour takes the other's value (the PSID fallback). + left = np.where(left < 0, right, left) + right = np.where(right < 0, left, right) + positive_left = np.where(left > 0, left, 1.0) + positive_right = np.where(right > 0, right, 1.0) + level = np.where( + (left > 0) & (right > 0), + np.sqrt(positive_left * positive_right), + np.where(left > 0, positive_left, positive_right), + ) + + def context_code(value): + return np.where( + np.isnan(value), + 0, + np.where(value <= 0, 1, np.where(value < median, 2, 3)), + ) + + return { + "sex": context.sex, + "age": _age_band(context.age), + "left_bin": _share_bin(left, share_edges), + "right_bin": _share_bin(right, share_edges), + "context3": 4 * context_code(context.left3) + + context_code(context.right3), + "wide": context.wide, + "level": level, + "valid": ~(np.isnan(context.left) & np.isnan(context.right)), + } + + @staticmethod + def _keys(parts) -> list[np.ndarray]: + sex = parts["sex"] + age = parts["age"] + coarse = _coarse_age(age) + left = parts["left_bin"] + right = parts["right_bin"] + context3 = parts["context3"] + wide = parts["wide"] + + # Nested keys from finest to coarsest; a dropped component is held + # at a sentinel (age 16-20 marks the coarse bands, 21 none). + def key(s, a, lb, rb, c3, w): + return ((((s * 22 + a) * 23 + lb) * 23 + rb) * 17 + c3) * 3 + w + + return [ + key(sex, age, left, right, context3, wide), + key(sex, age, left, right, context3, 2), + key(sex, age, left, right, 16, 2), + key(sex, 16 + coarse, left, right, 16, 2), + key(sex, 21, left, right, 16, 2), + key(0 * sex, 21, left, right, 16, 2), + key(0 * sex, 21, np.minimum(left, 1), np.minimum(right, 1), 16, 2), + ] + + def _cells(self, parts) -> tuple[np.ndarray, np.ndarray]: + """(level, row) of each unit's finest populated cell.""" + + keys = self._keys(parts) + level = np.full(len(keys[0]), -1, dtype=np.int64) + row = np.full(len(keys[0]), -1, dtype=np.int64) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + if (level < 0).any(): + raise ValueError("a unit has no populated cell at any level") + return level, row + + def _quantile(self, parts, u: np.ndarray) -> np.ndarray: + level, row = self._cells(parts) + out = np.zeros(len(u)) + for index in np.unique(level): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate(self.level_quantiles[index], row[take], v) + share = np.minimum(parts["level"][take] * np.exp(residual), 1.0) + out[take] = np.where(positive, share, 0.0) + return out + + # -- fitting -------------------------------------------------------------- + @classmethod + def fit( + cls, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + unit_years: tuple[int, ...], + rho_seed: int = 0, + ) -> tuple[OddQuantileFill, dict[str, object]]: + """Fit on complete TRAIN shares; every year of ``unit_years`` a unit. + + Every person-year of ``unit_years`` whose two neighbours are inside + the matrix is a training unit (all years are recorded on TRAIN). + """ + + shares = np.asarray(shares, dtype=np.float64) + known = np.isfinite(shares) + n = len(shares) + rows_list, years_list = [], [] + for year in unit_years: + rows_list.append(np.arange(n)) + years_list.append(np.full(n, year)) + rows = np.concatenate(rows_list) + unit_year = np.concatenate(years_list) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + target = _take(shares, rows, _column_of(years, unit_year)) + neighbours = np.concatenate([context.left, context.right]) + inside = neighbours[(neighbours > 0) & (neighbours < 1.0)] + share_edges = np.quantile( + inside, np.linspace(0, 1, _N_SHARE_BINS + 1)[1:-1] + ) + median = float(np.median(inside)) + parts = cls._parts(context, share_edges, median) + keys = cls._keys(parts) + positive = target > 0 + residual = np.log(np.where(positive, target, 1.0)) - np.log( + parts["level"] + ) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zeros_unique, zeros = np.unique( + key[in_cells & ~positive], return_counts=True + ) + totals = count[count >= MIN_CELL] + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zeros_unique)] = zeros + p0 = p0 / totals + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + # A populated cell with no positive unit draws only zeros. + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0.astype(np.float64)) + level_quantiles.append(full) + provisional = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=np.zeros((4, 16)), + ) + rho, rho_diagnostics = provisional._fit_rho( + parts, target, rows, unit_year, rho_seed + ) + fill = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=rho, + ) + cells = [len(k) for k in level_keys] + return fill, { + "n_units": int(len(target)), + "cells_per_level": cells, + **rho_diagnostics, + } + + def _pit(self, parts, target, rows, unit_year, seed) -> np.ndarray: + """Randomised probability integral transforms of true shares.""" + + level, row = self._cells(parts) + jitter = hash_uniform( + "epuf_fill.odd_quantile.pit", seed, rows, unit_year + ) + out = np.empty(len(target)) + for index in np.unique(level): + take = np.flatnonzero(level == index) + p0 = self.level_p0[index][row[take]] + zero = target[take] <= 0 + out[take[zero]] = jitter[take[zero]] * p0[zero] + positive = take[~zero] + residual = np.log(target[positive]) - np.log( + parts["level"][positive] + ) + # At the cap the residual is censored: spread it over the mass + # the quantile function puts at or above the cap. + at_cap = target[positive] >= 1.0 + v = _invert(self.level_quantiles[index], row[positive], residual) + cap_v = v.copy() + cap_v[at_cap] = v[at_cap] + jitter[positive][at_cap] * ( + 1.0 - v[at_cap] + ) + p0_positive = p0[~zero] + out[positive] = p0_positive + (1.0 - p0_positive) * cap_v + return np.clip(out, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, parts, target, rows, unit_year, seed): + """AR(1) correlation of consecutive units' normal scores (t, t+2). + + On a 5 percent sample of persons (by seed): every unit's + probability integral transform under the fitted cells, its normal + score, and the correlation of the scores of ``t`` and ``t+2`` for + the same person, by sex and age band at ``t``. + """ + + persons = np.unique(rows) + rng = np.random.default_rng(seed) + chosen = persons[rng.random(len(persons)) < 0.05] + index = np.flatnonzero(np.isin(rows, chosen)) + sub = {k: v[index] for k, v in parts.items()} + z = ndtri( + self._pit(sub, target[index], rows[index], unit_year[index], seed) + ) + first_year = int(unit_year.min()) + n_years = int(unit_year.max()) - first_year + 1 + position = np.searchsorted(chosen, rows[index]) + grid = np.full((len(chosen), n_years), np.nan) + grid[position, unit_year[index] - first_year] = z + sex_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + sex_grid[position, unit_year[index] - first_year] = sub["sex"] + age_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + age_grid[position, unit_year[index] - first_year] = sub["age"] + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex = sex_grid[:, :-2].ravel() + age = age_grid[:, :-2].ravel() + both = np.isfinite(now) & np.isfinite(later) + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = both & (sex == s) & (age == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + overall = float(np.corrcoef(now[both], later[both])[0, 1]) + four_now = grid[:, :-4].ravel() + four_later = grid[:, 4:].ravel() + four = np.isfinite(four_now) & np.isfinite(four_later) + lag4 = float(np.corrcoef(four_now[four], four_later[four])[0, 1]) + return rho, { + "rho_persons": int(len(chosen)), + "rho_pairs": int(both.sum()), + "rho_overall": overall, + "lag4_normal_score_correlation": lag4, + "lag4_ar1_prediction": overall**2, + } + + # -- filling -------------------------------------------------------------- + def fill( + self, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + person_key: np.ndarray, + fill_mask: np.ndarray, + seed: int, + ) -> np.ndarray: + """Fill the masked cells; masked cells with no known neighbour stay NaN.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + parts = self._parts(context, self.share_edges, self.context_median) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + rho = self.rho[np.clip(parts["sex"], 0, 3), parts["age"]] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + valid = parts["valid"] + drawn = np.full(len(rows), np.nan) + if valid.any(): + sub = {k: v[valid] for k, v in parts.items()} + drawn[valid] = self._quantile(sub, ndtr(z[valid])) + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + # -- persistence ---------------------------------------------------------- + def to_bytes(self) -> bytes: + arrays = { + "kind": np.array(self.name), + "share_edges": self.share_edges, + "context_median": np.array(self.context_median), + "rho": self.rho, + } + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> OddQuantileFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + share_edges=arrays["share_edges"], + context_median=float(arrays["context_median"]), + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + rho=arrays["rho"], + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: a quantile regression forest (QRF) draw +# -------------------------------------------------------------------------- + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +#: The reference level of a unit with no positive share around it. +_DEFAULT_LEVEL = 0.3 + + +def reference_level(context: OddContext) -> np.ndarray: + """The level a unit's share is drawn relative to. + + The geometric mean of the positive shares at ``t-1`` and ``t+1``; else + of those at ``t-3`` and ``t+3``; else the mean positive share at + offsets 5-9; else 0.3. + """ + + def geometric(a, b): + a = np.nan_to_num(a, nan=0.0) + b = np.nan_to_num(b, nan=0.0) + both = (a > 0) & (b > 0) + one = np.where(a > 0, a, b) + value = np.where(both, np.sqrt(np.where(both, a * b, 1.0)), one) + return np.where((a > 0) | (b > 0), value, np.nan) + + level = geometric(context.left, context.right) + level = np.where( + np.isnan(level), geometric(context.left3, context.right3), level + ) + level = np.where( + np.isnan(level) & (context.wide_mean > 0), context.wide_mean, level + ) + return np.where(np.isnan(level), _DEFAULT_LEVEL, level) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.91, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + Two parts, both random forests (scikit-learn) on :func:`odd_features` + of TRAIN units inside the career, whose contexts see the career only: + + 1. a probability forest for a zero year: ``p0`` is the mean over trees + of the zero share of the unit's leaves; + 2. a quantile regression forest on positive shares (split target + ``log share``): every positive TRAIN unit used in the fit is passed + down every tree, and each leaf keeps the sorted true shares that + reach it (the cap included, stored as shares times 65,535). + + A draw maps the copula uniform ``u`` to zero below ``p0``; otherwise a + second seeded uniform picks a tree, and the share is that tree's leaf + value at the quantile ``(u - p0) / (1 - p0)``. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + zero_tree_offsets: np.ndarray + zero_node_left: np.ndarray + zero_node_right: np.ndarray + zero_node_feature: np.ndarray + zero_node_threshold: np.ndarray + zero_node_leaf: np.ndarray + zero_leaf_p: np.ndarray + stream: str = "epuf_fill.odd_forest.v4" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + # Units lie inside the career, and their contexts see the career + # only, as a fill's do (pre-career years are unknown to it). + inside = unit_year >= career_start(birth_year[rows]) + rows, unit_year = rows[inside], unit_year[inside] + pre_career = years[None, :] < career_start(birth_year)[:, None] + known = np.isfinite(shares) & ~pre_career + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x_all = odd_features(context) + y_all = target[chosen] + # Part one: the probability of a zero year, a probability forest. + from sklearn.ensemble import RandomForestClassifier + + zero_forest = RandomForestClassifier( + n_estimators=n_trees, + min_samples_leaf=4 * min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + zero_forest.fit(x_all, (y_all <= 0).astype(np.int8)) + zero_arrays = _forest_arrays( + zero_forest, x_all, (y_all <= 0).astype(np.float64) + ) + # Part two: the positive share, a quantile regression forest. + positive = y_all > 0 + x = x_all[positive] + y = y_all[positive] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + **zero_arrays, + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y_all)), + "n_positive_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _p_zero(self, x) -> np.ndarray: + """The probability forest's zero-year probability (mean over trees).""" + + x = np.asarray(x, dtype=np.float32) + n_trees = len(self.zero_tree_offsets) - 1 + total = np.zeros(len(x)) + for tree in range(n_trees): + start = self.zero_tree_offsets[tree] + stop = self.zero_tree_offsets[tree + 1] + node = _tree_leaves( + self.zero_node_left[start:stop].astype(np.int64), + self.zero_node_right[start:stop].astype(np.int64), + self.zero_node_feature[start:stop].astype(np.int64), + self.zero_node_threshold[start:stop], + x, + ) + total += self.zero_leaf_p[ + self.zero_node_leaf[start:stop][node].astype(np.int64) + ] + return total / n_trees + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + p_zero = np.ones(len(rows)) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + p_zero[valid] = self._p_zero(features) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "p_zero": p_zero, + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + u = ndtr(z) + p0 = unit["p_zero"] + valid = valid & (u >= p0) + v = (u[valid] - p0[valid]) / np.maximum(1.0 - p0[valid], 1e-12) + drawn[valid] = self._value(unit["leaves"][valid], v) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + pre_career = years[None, :] < start[:, None] + mask &= ~pre_career + given = np.where(mask | pre_career, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +def _forest_arrays(forest, x, y) -> dict[str, np.ndarray]: + """A fitted probability forest as arrays: nodes, and each leaf's mean y.""" + + offsets = [0] + lefts, rights, features, thresholds, leaf_index, means = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + n_leaves = int(is_leaf.sum()) + index = np.full(tree.node_count, -1, dtype=np.int64) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + total = np.bincount(local, minlength=n_leaves) + hits = np.bincount(local, weights=y, minlength=n_leaves) + means.append(hits / np.maximum(total, 1)) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + offsets.append(offsets[-1] + tree.node_count) + return { + "zero_tree_offsets": np.asarray(offsets, dtype=np.int64), + "zero_node_left": np.concatenate(lefts).astype(np.int32), + "zero_node_right": np.concatenate(rights).astype(np.int32), + "zero_node_feature": np.concatenate(features), + "zero_node_threshold": np.concatenate(thresholds), + "zero_node_leaf": np.concatenate(leaf_index).astype(np.int32), + "zero_leaf_p": np.concatenate(means).astype(np.float32), + } + + +_FOREST_ARRAYS = ( + "zero_tree_offsets", + "zero_node_left", + "zero_node_right", + "zero_node_feature", + "zero_node_threshold", + "zero_node_leaf", + "zero_leaf_p", + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + known = np.isfinite(shares) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum): + members = np.flatnonzero(stratum == value) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=context.left[keep].astype(np.float32), + bank_right=context.right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:_DONOR_BANK] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + -1 for rows with no masked cell or no bank donor of their sex and + birth year. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + out = np.full(len(shares), -1, dtype=np.int64) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + recipients = targets[ + (sex[targets] == s) & (birth_year[targets] == b) + ] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + column = np.sort( + donor_match[:, j][np.isfinite(donor_match[:, j])] + ) + ranks_donor[:, j] = np.where( + np.isfinite(donor_match[:, j]), + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + nearest = np.argpartition(distance, k - 1, axis=1)[:, :k] + nearest_distance = np.take_along_axis(distance, nearest, 1) + order = np.lexsort((nearest, nearest_distance), axis=1) + nearest = np.take_along_axis(nearest, order, 1) + pick = np.minimum( + (u[recipients[block]] * k).astype(np.int64), k - 1 + ) + no_match = count.max(axis=1) == 0 + random_donor = np.minimum( + (u[recipients[block]] * len(donors)).astype(np.int64), + len(donors) - 1, + ) + out[recipients[block]] = donors[ + np.where( + no_match, + random_donor, + nearest[np.arange(len(nearest)), pick], + ) + ] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_quantile": OddQuantileFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_5af343166b51.py b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_5af343166b51.py new file mode 100644 index 00000000..787f5aaa --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_5af343166b51.py @@ -0,0 +1,2127 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddQuantileFill` (odd years, primary): a two-part conditional + draw. The probability of a zero year and the conditional quantiles of a + positive share, relative to the neighbours' level, by sex, age, the + shares at ``t-1`` and ``t+1`` and the context at ``t-3``, ``t+3`` and + further; a Gaussian AR(1) copula correlates a person's draws across + masked years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "OddQuantileFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _invert(table: np.ndarray, rows: np.ndarray, value: np.ndarray): + """The level at which each row's quantile function reaches ``value``. + + Where the function is flat at ``value`` (a run of equal quantiles), the + middle of the run's levels. + """ + + points = table.shape[1] + levels = np.linspace(0.0, 1.0, points) + out = np.empty(len(rows)) + for start in range(0, len(rows), 200_000): + block = slice(start, start + 200_000) + curve = table[rows[block]].astype(np.float64) + target = np.asarray(value[block], dtype=np.float64)[:, None] + below = (curve < target).sum(axis=1) + above = (curve <= target).sum(axis=1) + flat = below < above + result = np.empty(len(curve)) + result[above == 0] = 0.0 + result[below >= points] = 1.0 + middle = flat & (above > 0) & (below < points) + result[middle] = 0.5 * ( + levels[below[middle]] + levels[above[middle] - 1] + ) + between = ~flat & (below > 0) & (below < points) + index = np.flatnonzero(between) + left = curve[index, below[index] - 1] + right = curve[index, below[index]] + share = np.where( + right > left, + (target[index, 0] - left) + / np.where(right > left, right - left, 1), + 0.5, + ) + result[index] = levels[below[index] - 1] + share * ( + levels[below[index]] - levels[below[index] - 1] + ) + out[block] = result + return out + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + """A compressed ``.npz`` whose bytes depend only on the arrays. + + ``numpy.savez_compressed`` stamps each member with the time of writing, + so two writes of the same fill differ. This writer fixes every member's + timestamp and order, so a fill's SHA-256 can be registered and refit. + """ + + import zipfile + + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive: + for name in sorted(arrays): + member = io.BytesIO() + np.lib.format.write_array( + member, np.asanyarray(arrays[name]), allow_pickle=False + ) + info = zipfile.ZipInfo( + f"{name}.npy", date_time=(1980, 1, 1, 0, 0, 0) + ) + info.compress_type = zipfile.ZIP_DEFLATED + archive.writestr(info, member.getvalue()) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: the two-part conditional draw +# -------------------------------------------------------------------------- +_N_SHARE_BINS = 20 + + +def _share_bin(value: np.ndarray, edges: np.ndarray) -> np.ndarray: + """0 zero, 1..20 quantile bins of a positive share below the cap, 21 cap.""" + + bins = np.searchsorted(edges, value, side="right") + 1 + bins = np.where(value <= 0, 0, bins) + return np.where(value >= 1.0, _N_SHARE_BINS + 1, bins) + + +def _coarse_age(band: np.ndarray) -> np.ndarray: + """Age bands grouped: under 30, 30-44, 45-59, 60 and over.""" + + return np.digitize(band, [4, 7, 10]) + + +@dataclass(frozen=True) +class OddQuantileFill: + """Two-part conditional draw for masked odd years, with an AR(1) copula. + + A unit's reference level ``m`` is the geometric mean of its positive + neighbours' shares (the one positive neighbour's share if only one is, + 1 if neither is). Its cell is the finest of seven nested keys with at + least ``MIN_CELL`` TRAIN units, built from sex, five-year age band, the + bins of the shares at ``t-1`` and ``t+1`` (zero, 20 quantile bins of a + positive share below the cap, at the cap), the context at ``t-3`` and + ``t+3`` (missing, zero, below or above the median positive share), and + whether any share at offsets 5, 7 or 9 is positive. In the cell: ``p0`` + the share of zero years, and 65 quantiles of ``log(x_t / m)`` among + positive years. A uniform ``u`` maps to zero if ``u < p0``, else to + ``min(m * exp(Q((u - p0) / (1 - p0))), 1)``. The uniforms of a person's + consecutive masked years (two years apart) are joined by a Gaussian + AR(1) copula with correlation ``rho`` by sex and age band, learned on + TRAIN from the probability integral transforms of consecutive units. + """ + + share_edges: np.ndarray + context_median: float + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + rho: np.ndarray + stream: str = "epuf_fill.odd_quantile.v1" + name: str = "odd_quantile" + + # -- keys --------------------------------------------------------------- + @staticmethod + def _parts(context: OddContext, share_edges, median): + left = np.nan_to_num(context.left, nan=-1.0) + right = np.nan_to_num(context.right, nan=-1.0) + # A missing neighbour takes the other's value (the PSID fallback). + left = np.where(left < 0, right, left) + right = np.where(right < 0, left, right) + positive_left = np.where(left > 0, left, 1.0) + positive_right = np.where(right > 0, right, 1.0) + level = np.where( + (left > 0) & (right > 0), + np.sqrt(positive_left * positive_right), + np.where(left > 0, positive_left, positive_right), + ) + + def context_code(value): + return np.where( + np.isnan(value), + 0, + np.where(value <= 0, 1, np.where(value < median, 2, 3)), + ) + + return { + "sex": context.sex, + "age": _age_band(context.age), + "left_bin": _share_bin(left, share_edges), + "right_bin": _share_bin(right, share_edges), + "context3": 4 * context_code(context.left3) + + context_code(context.right3), + "wide": context.wide, + "level": level, + "valid": ~(np.isnan(context.left) & np.isnan(context.right)), + } + + @staticmethod + def _keys(parts) -> list[np.ndarray]: + sex = parts["sex"] + age = parts["age"] + coarse = _coarse_age(age) + left = parts["left_bin"] + right = parts["right_bin"] + context3 = parts["context3"] + wide = parts["wide"] + + # Nested keys from finest to coarsest; a dropped component is held + # at a sentinel (age 16-20 marks the coarse bands, 21 none). + def key(s, a, lb, rb, c3, w): + return ((((s * 22 + a) * 23 + lb) * 23 + rb) * 17 + c3) * 3 + w + + return [ + key(sex, age, left, right, context3, wide), + key(sex, age, left, right, context3, 2), + key(sex, age, left, right, 16, 2), + key(sex, 16 + coarse, left, right, 16, 2), + key(sex, 21, left, right, 16, 2), + key(0 * sex, 21, left, right, 16, 2), + key(0 * sex, 21, np.minimum(left, 1), np.minimum(right, 1), 16, 2), + ] + + def _cells(self, parts) -> tuple[np.ndarray, np.ndarray]: + """(level, row) of each unit's finest populated cell.""" + + keys = self._keys(parts) + level = np.full(len(keys[0]), -1, dtype=np.int64) + row = np.full(len(keys[0]), -1, dtype=np.int64) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + if (level < 0).any(): + raise ValueError("a unit has no populated cell at any level") + return level, row + + def _quantile(self, parts, u: np.ndarray) -> np.ndarray: + level, row = self._cells(parts) + out = np.zeros(len(u)) + for index in np.unique(level): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate(self.level_quantiles[index], row[take], v) + share = np.minimum(parts["level"][take] * np.exp(residual), 1.0) + out[take] = np.where(positive, share, 0.0) + return out + + # -- fitting -------------------------------------------------------------- + @classmethod + def fit( + cls, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + unit_years: tuple[int, ...], + rho_seed: int = 0, + ) -> tuple[OddQuantileFill, dict[str, object]]: + """Fit on complete TRAIN shares; every year of ``unit_years`` a unit. + + Every person-year of ``unit_years`` whose two neighbours are inside + the matrix is a training unit (all years are recorded on TRAIN). + """ + + shares = np.asarray(shares, dtype=np.float64) + known = np.isfinite(shares) + n = len(shares) + rows_list, years_list = [], [] + for year in unit_years: + rows_list.append(np.arange(n)) + years_list.append(np.full(n, year)) + rows = np.concatenate(rows_list) + unit_year = np.concatenate(years_list) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + target = _take(shares, rows, _column_of(years, unit_year)) + neighbours = np.concatenate([context.left, context.right]) + inside = neighbours[(neighbours > 0) & (neighbours < 1.0)] + share_edges = np.quantile( + inside, np.linspace(0, 1, _N_SHARE_BINS + 1)[1:-1] + ) + median = float(np.median(inside)) + parts = cls._parts(context, share_edges, median) + keys = cls._keys(parts) + positive = target > 0 + residual = np.log(np.where(positive, target, 1.0)) - np.log( + parts["level"] + ) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zeros_unique, zeros = np.unique( + key[in_cells & ~positive], return_counts=True + ) + totals = count[count >= MIN_CELL] + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zeros_unique)] = zeros + p0 = p0 / totals + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + # A populated cell with no positive unit draws only zeros. + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0.astype(np.float64)) + level_quantiles.append(full) + provisional = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=np.zeros((4, 16)), + ) + rho, rho_diagnostics = provisional._fit_rho( + parts, target, rows, unit_year, rho_seed + ) + fill = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=rho, + ) + cells = [len(k) for k in level_keys] + return fill, { + "n_units": int(len(target)), + "cells_per_level": cells, + **rho_diagnostics, + } + + def _pit(self, parts, target, rows, unit_year, seed) -> np.ndarray: + """Randomised probability integral transforms of true shares.""" + + level, row = self._cells(parts) + jitter = hash_uniform( + "epuf_fill.odd_quantile.pit", seed, rows, unit_year + ) + out = np.empty(len(target)) + for index in np.unique(level): + take = np.flatnonzero(level == index) + p0 = self.level_p0[index][row[take]] + zero = target[take] <= 0 + out[take[zero]] = jitter[take[zero]] * p0[zero] + positive = take[~zero] + residual = np.log(target[positive]) - np.log( + parts["level"][positive] + ) + # At the cap the residual is censored: spread it over the mass + # the quantile function puts at or above the cap. + at_cap = target[positive] >= 1.0 + v = _invert(self.level_quantiles[index], row[positive], residual) + cap_v = v.copy() + cap_v[at_cap] = v[at_cap] + jitter[positive][at_cap] * ( + 1.0 - v[at_cap] + ) + p0_positive = p0[~zero] + out[positive] = p0_positive + (1.0 - p0_positive) * cap_v + return np.clip(out, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, parts, target, rows, unit_year, seed): + """AR(1) correlation of consecutive units' normal scores (t, t+2). + + On a 5 percent sample of persons (by seed): every unit's + probability integral transform under the fitted cells, its normal + score, and the correlation of the scores of ``t`` and ``t+2`` for + the same person, by sex and age band at ``t``. + """ + + persons = np.unique(rows) + rng = np.random.default_rng(seed) + chosen = persons[rng.random(len(persons)) < 0.05] + index = np.flatnonzero(np.isin(rows, chosen)) + sub = {k: v[index] for k, v in parts.items()} + z = ndtri( + self._pit(sub, target[index], rows[index], unit_year[index], seed) + ) + first_year = int(unit_year.min()) + n_years = int(unit_year.max()) - first_year + 1 + position = np.searchsorted(chosen, rows[index]) + grid = np.full((len(chosen), n_years), np.nan) + grid[position, unit_year[index] - first_year] = z + sex_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + sex_grid[position, unit_year[index] - first_year] = sub["sex"] + age_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + age_grid[position, unit_year[index] - first_year] = sub["age"] + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex = sex_grid[:, :-2].ravel() + age = age_grid[:, :-2].ravel() + both = np.isfinite(now) & np.isfinite(later) + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = both & (sex == s) & (age == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + overall = float(np.corrcoef(now[both], later[both])[0, 1]) + four_now = grid[:, :-4].ravel() + four_later = grid[:, 4:].ravel() + four = np.isfinite(four_now) & np.isfinite(four_later) + lag4 = float(np.corrcoef(four_now[four], four_later[four])[0, 1]) + return rho, { + "rho_persons": int(len(chosen)), + "rho_pairs": int(both.sum()), + "rho_overall": overall, + "lag4_normal_score_correlation": lag4, + "lag4_ar1_prediction": overall**2, + } + + # -- filling -------------------------------------------------------------- + def fill( + self, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + person_key: np.ndarray, + fill_mask: np.ndarray, + seed: int, + ) -> np.ndarray: + """Fill the masked cells; masked cells with no known neighbour stay NaN.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + parts = self._parts(context, self.share_edges, self.context_median) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + rho = self.rho[np.clip(parts["sex"], 0, 3), parts["age"]] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + valid = parts["valid"] + drawn = np.full(len(rows), np.nan) + if valid.any(): + sub = {k: v[valid] for k, v in parts.items()} + drawn[valid] = self._quantile(sub, ndtr(z[valid])) + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + # -- persistence ---------------------------------------------------------- + def to_bytes(self) -> bytes: + arrays = { + "kind": np.array(self.name), + "share_edges": self.share_edges, + "context_median": np.array(self.context_median), + "rho": self.rho, + } + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> OddQuantileFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + share_edges=arrays["share_edges"], + context_median=float(arrays["context_median"]), + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + rho=arrays["rho"], + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: a quantile regression forest (QRF) draw +# -------------------------------------------------------------------------- + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +#: The reference level of a unit with no positive share around it. +_DEFAULT_LEVEL = 0.3 + + +def reference_level(context: OddContext) -> np.ndarray: + """The level a unit's share is drawn relative to. + + The geometric mean of the positive shares at ``t-1`` and ``t+1``; else + of those at ``t-3`` and ``t+3``; else the mean positive share at + offsets 5-9; else 0.3. + """ + + def geometric(a, b): + a = np.nan_to_num(a, nan=0.0) + b = np.nan_to_num(b, nan=0.0) + both = (a > 0) & (b > 0) + one = np.where(a > 0, a, b) + value = np.where(both, np.sqrt(np.where(both, a * b, 1.0)), one) + return np.where((a > 0) | (b > 0), value, np.nan) + + level = geometric(context.left, context.right) + level = np.where( + np.isnan(level), geometric(context.left3, context.right3), level + ) + level = np.where( + np.isnan(level) & (context.wide_mean > 0), context.wide_mean, level + ) + return np.where(np.isnan(level), _DEFAULT_LEVEL, level) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.91, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + Two parts, both random forests (scikit-learn) on :func:`odd_features` + of TRAIN units inside the career, whose contexts see the career only: + + 1. a probability forest for a zero year: ``p0`` is the mean over trees + of the zero share of the unit's leaves; + 2. a quantile regression forest on positive shares (split target + ``log share``): every positive TRAIN unit used in the fit is passed + down every tree, and each leaf keeps the sorted true shares that + reach it (the cap included, stored as shares times 65,535). + + A draw maps the copula uniform ``u`` to zero below ``p0``; otherwise a + second seeded uniform picks a tree, and the share is that tree's leaf + value at the quantile ``(u - p0) / (1 - p0)``. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + zero_tree_offsets: np.ndarray + zero_node_left: np.ndarray + zero_node_right: np.ndarray + zero_node_feature: np.ndarray + zero_node_threshold: np.ndarray + zero_node_leaf: np.ndarray + zero_leaf_p: np.ndarray + stream: str = "epuf_fill.odd_forest.v4" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + # Units lie inside the career, and their contexts see the career + # only, as a fill's do (pre-career years are unknown to it). + inside = unit_year >= career_start(birth_year[rows]) + rows, unit_year = rows[inside], unit_year[inside] + pre_career = years[None, :] < career_start(birth_year)[:, None] + known = np.isfinite(shares) & ~pre_career + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x_all = odd_features(context) + y_all = target[chosen] + # Part one: the probability of a zero year, a probability forest. + from sklearn.ensemble import RandomForestClassifier + + zero_forest = RandomForestClassifier( + n_estimators=n_trees, + min_samples_leaf=4 * min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + zero_forest.fit(x_all, (y_all <= 0).astype(np.int8)) + zero_arrays = _forest_arrays( + zero_forest, x_all, (y_all <= 0).astype(np.float64) + ) + # Part two: the positive share, a quantile regression forest. + positive = y_all > 0 + x = x_all[positive] + y = y_all[positive] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + **zero_arrays, + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y_all)), + "n_positive_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _p_zero(self, x) -> np.ndarray: + """The probability forest's zero-year probability (mean over trees).""" + + x = np.asarray(x, dtype=np.float32) + n_trees = len(self.zero_tree_offsets) - 1 + total = np.zeros(len(x)) + for tree in range(n_trees): + start = self.zero_tree_offsets[tree] + stop = self.zero_tree_offsets[tree + 1] + node = _tree_leaves( + self.zero_node_left[start:stop].astype(np.int64), + self.zero_node_right[start:stop].astype(np.int64), + self.zero_node_feature[start:stop].astype(np.int64), + self.zero_node_threshold[start:stop], + x, + ) + total += self.zero_leaf_p[ + self.zero_node_leaf[start:stop][node].astype(np.int64) + ] + return total / n_trees + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + p_zero = np.ones(len(rows)) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + p_zero[valid] = self._p_zero(features) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "p_zero": p_zero, + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + u = ndtr(z) + p0 = unit["p_zero"] + valid = valid & (u >= p0) + v = (u[valid] - p0[valid]) / np.maximum(1.0 - p0[valid], 1e-12) + drawn[valid] = self._value(unit["leaves"][valid], v) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + pre_career = years[None, :] < start[:, None] + mask &= ~pre_career + given = np.where(mask | pre_career, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +def _forest_arrays(forest, x, y) -> dict[str, np.ndarray]: + """A fitted probability forest as arrays: nodes, and each leaf's mean y.""" + + offsets = [0] + lefts, rights, features, thresholds, leaf_index, means = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + n_leaves = int(is_leaf.sum()) + index = np.full(tree.node_count, -1, dtype=np.int64) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + total = np.bincount(local, minlength=n_leaves) + hits = np.bincount(local, weights=y, minlength=n_leaves) + means.append(hits / np.maximum(total, 1)) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + offsets.append(offsets[-1] + tree.node_count) + return { + "zero_tree_offsets": np.asarray(offsets, dtype=np.int64), + "zero_node_left": np.concatenate(lefts).astype(np.int32), + "zero_node_right": np.concatenate(rights).astype(np.int32), + "zero_node_feature": np.concatenate(features), + "zero_node_threshold": np.concatenate(thresholds), + "zero_node_leaf": np.concatenate(leaf_index).astype(np.int32), + "zero_leaf_p": np.concatenate(means).astype(np.float32), + } + + +_FOREST_ARRAYS = ( + "zero_tree_offsets", + "zero_node_left", + "zero_node_right", + "zero_node_feature", + "zero_node_threshold", + "zero_node_leaf", + "zero_leaf_p", + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + # Units inside the career; contexts see the career only. + start = career_start(np.asarray(birth_year)) + inside = unit_year >= start[rows] + rows, unit_year = rows[inside], unit_year[inside] + known = np.isfinite(shares) & ~( + np.asarray(years)[None, :] < start[:, None] + ) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum): + members = np.flatnonzero(stratum == value) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=context.left[keep].astype(np.float32), + bank_right=context.right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: Nearest-donor lists by input content, reused across draw seeds. +_NEAREST_CACHE: dict = {} +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:_DONOR_BANK] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def _nearest(self, match, birth_year, sex, targets): + """Each target's ``k`` nearest bank rows, and its group's bank rows. + + Seed-free, so it is computed once for a matrix and reused across + draw seeds (cached by the content of its inputs). + """ + + digest = hashlib.sha256( + np.ascontiguousarray(match[targets]).tobytes() + + np.ascontiguousarray(birth_year[targets]).tobytes() + + np.ascontiguousarray(sex[targets]).tobytes() + + np.ascontiguousarray(targets).tobytes() + + str((id(self), self.k)).encode() + ).hexdigest() + if digest in _NEAREST_CACHE: + return _NEAREST_CACHE[digest] + nearest = np.full((len(targets), self.k), -1, dtype=np.int64) + group_first = np.full(len(targets), -1, dtype=np.int64) + group_size = np.zeros(len(targets), dtype=np.int64) + no_match = np.zeros(len(targets), dtype=bool) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + local = np.flatnonzero( + (sex[targets] == s) & (birth_year[targets] == b) + ) + recipients = targets[local] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + group_first[local] = donors[0] + group_size[local] = len(donors) + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + finite = np.isfinite(donor_match[:, j]) + column = np.sort(donor_match[finite, j]) + ranks_donor[:, j] = np.where( + finite, + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + order = np.argpartition(distance, k - 1, axis=1)[:, :k] + near_distance = np.take_along_axis(distance, order, 1) + ranked = np.lexsort((order, near_distance), axis=1) + order = np.take_along_axis(order, ranked, 1) + rows = local[block] + nearest[rows, :k] = donors[order] + no_match[rows] = count.max(axis=1) == 0 + result = (nearest, group_first, group_size, no_match) + if len(_NEAREST_CACHE) >= 4: + _NEAREST_CACHE.pop(next(iter(_NEAREST_CACHE))) + _NEAREST_CACHE[digest] = result + return result + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + One of the ``k`` nearest bank donors of the recipient's sex and + birth year, chosen by the seeded uniform; a recipient with no + recorded match feature takes a random donor of its group. -1 for + rows with no masked cell or no bank donor of their group. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + nearest, group_first, group_size, no_match = self._nearest( + match, birth_year, sex, targets + ) + out = np.full(len(shares), -1, dtype=np.int64) + has_group = group_size > 0 + k_available = (nearest >= 0).sum(axis=1) + pick = np.minimum( + (u[targets] * np.maximum(k_available, 1)).astype(np.int64), + np.maximum(k_available - 1, 0), + ) + chosen = nearest[np.arange(len(targets)), pick] + random_donor = group_first + np.minimum( + (u[targets] * np.maximum(group_size, 1)).astype(np.int64), + np.maximum(group_size - 1, 0), + ) + chosen = np.where(no_match, random_donor, chosen) + out[targets[has_group]] = chosen[has_group] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_quantile": OddQuantileFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_c5e4a37417ec.py b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_c5e4a37417ec.py new file mode 100644 index 00000000..49420071 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_c5e4a37417ec.py @@ -0,0 +1,2131 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddQuantileFill` (odd years, primary): a two-part conditional + draw. The probability of a zero year and the conditional quantiles of a + positive share, relative to the neighbours' level, by sex, age, the + shares at ``t-1`` and ``t+1`` and the context at ``t-3``, ``t+3`` and + further; a Gaussian AR(1) copula correlates a person's draws across + masked years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "OddQuantileFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _invert(table: np.ndarray, rows: np.ndarray, value: np.ndarray): + """The level at which each row's quantile function reaches ``value``. + + Where the function is flat at ``value`` (a run of equal quantiles), the + middle of the run's levels. + """ + + points = table.shape[1] + levels = np.linspace(0.0, 1.0, points) + out = np.empty(len(rows)) + for start in range(0, len(rows), 200_000): + block = slice(start, start + 200_000) + curve = table[rows[block]].astype(np.float64) + target = np.asarray(value[block], dtype=np.float64)[:, None] + below = (curve < target).sum(axis=1) + above = (curve <= target).sum(axis=1) + flat = below < above + result = np.empty(len(curve)) + result[above == 0] = 0.0 + result[below >= points] = 1.0 + middle = flat & (above > 0) & (below < points) + result[middle] = 0.5 * ( + levels[below[middle]] + levels[above[middle] - 1] + ) + between = ~flat & (below > 0) & (below < points) + index = np.flatnonzero(between) + left = curve[index, below[index] - 1] + right = curve[index, below[index]] + share = np.where( + right > left, + (target[index, 0] - left) + / np.where(right > left, right - left, 1), + 0.5, + ) + result[index] = levels[below[index] - 1] + share * ( + levels[below[index]] - levels[below[index] - 1] + ) + out[block] = result + return out + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + """A compressed ``.npz`` whose bytes depend only on the arrays. + + ``numpy.savez_compressed`` stamps each member with the time of writing, + so two writes of the same fill differ. This writer fixes every member's + timestamp and order, so a fill's SHA-256 can be registered and refit. + """ + + import zipfile + + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive: + for name in sorted(arrays): + member = io.BytesIO() + np.lib.format.write_array( + member, np.asanyarray(arrays[name]), allow_pickle=False + ) + info = zipfile.ZipInfo( + f"{name}.npy", date_time=(1980, 1, 1, 0, 0, 0) + ) + info.compress_type = zipfile.ZIP_DEFLATED + archive.writestr(info, member.getvalue()) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: the two-part conditional draw +# -------------------------------------------------------------------------- +_N_SHARE_BINS = 20 + + +def _share_bin(value: np.ndarray, edges: np.ndarray) -> np.ndarray: + """0 zero, 1..20 quantile bins of a positive share below the cap, 21 cap.""" + + bins = np.searchsorted(edges, value, side="right") + 1 + bins = np.where(value <= 0, 0, bins) + return np.where(value >= 1.0, _N_SHARE_BINS + 1, bins) + + +def _coarse_age(band: np.ndarray) -> np.ndarray: + """Age bands grouped: under 30, 30-44, 45-59, 60 and over.""" + + return np.digitize(band, [4, 7, 10]) + + +@dataclass(frozen=True) +class OddQuantileFill: + """Two-part conditional draw for masked odd years, with an AR(1) copula. + + A unit's reference level ``m`` is the geometric mean of its positive + neighbours' shares (the one positive neighbour's share if only one is, + 1 if neither is). Its cell is the finest of seven nested keys with at + least ``MIN_CELL`` TRAIN units, built from sex, five-year age band, the + bins of the shares at ``t-1`` and ``t+1`` (zero, 20 quantile bins of a + positive share below the cap, at the cap), the context at ``t-3`` and + ``t+3`` (missing, zero, below or above the median positive share), and + whether any share at offsets 5, 7 or 9 is positive. In the cell: ``p0`` + the share of zero years, and 65 quantiles of ``log(x_t / m)`` among + positive years. A uniform ``u`` maps to zero if ``u < p0``, else to + ``min(m * exp(Q((u - p0) / (1 - p0))), 1)``. The uniforms of a person's + consecutive masked years (two years apart) are joined by a Gaussian + AR(1) copula with correlation ``rho`` by sex and age band, learned on + TRAIN from the probability integral transforms of consecutive units. + """ + + share_edges: np.ndarray + context_median: float + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + rho: np.ndarray + stream: str = "epuf_fill.odd_quantile.v1" + name: str = "odd_quantile" + + # -- keys --------------------------------------------------------------- + @staticmethod + def _parts(context: OddContext, share_edges, median): + left = np.nan_to_num(context.left, nan=-1.0) + right = np.nan_to_num(context.right, nan=-1.0) + # A missing neighbour takes the other's value (the PSID fallback). + left = np.where(left < 0, right, left) + right = np.where(right < 0, left, right) + positive_left = np.where(left > 0, left, 1.0) + positive_right = np.where(right > 0, right, 1.0) + level = np.where( + (left > 0) & (right > 0), + np.sqrt(positive_left * positive_right), + np.where(left > 0, positive_left, positive_right), + ) + + def context_code(value): + return np.where( + np.isnan(value), + 0, + np.where(value <= 0, 1, np.where(value < median, 2, 3)), + ) + + return { + "sex": context.sex, + "age": _age_band(context.age), + "left_bin": _share_bin(left, share_edges), + "right_bin": _share_bin(right, share_edges), + "context3": 4 * context_code(context.left3) + + context_code(context.right3), + "wide": context.wide, + "level": level, + "valid": ~(np.isnan(context.left) & np.isnan(context.right)), + } + + @staticmethod + def _keys(parts) -> list[np.ndarray]: + sex = parts["sex"] + age = parts["age"] + coarse = _coarse_age(age) + left = parts["left_bin"] + right = parts["right_bin"] + context3 = parts["context3"] + wide = parts["wide"] + + # Nested keys from finest to coarsest; a dropped component is held + # at a sentinel (age 16-20 marks the coarse bands, 21 none). + def key(s, a, lb, rb, c3, w): + return ((((s * 22 + a) * 23 + lb) * 23 + rb) * 17 + c3) * 3 + w + + return [ + key(sex, age, left, right, context3, wide), + key(sex, age, left, right, context3, 2), + key(sex, age, left, right, 16, 2), + key(sex, 16 + coarse, left, right, 16, 2), + key(sex, 21, left, right, 16, 2), + key(0 * sex, 21, left, right, 16, 2), + key(0 * sex, 21, np.minimum(left, 1), np.minimum(right, 1), 16, 2), + ] + + def _cells(self, parts) -> tuple[np.ndarray, np.ndarray]: + """(level, row) of each unit's finest populated cell.""" + + keys = self._keys(parts) + level = np.full(len(keys[0]), -1, dtype=np.int64) + row = np.full(len(keys[0]), -1, dtype=np.int64) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + if (level < 0).any(): + raise ValueError("a unit has no populated cell at any level") + return level, row + + def _quantile(self, parts, u: np.ndarray) -> np.ndarray: + level, row = self._cells(parts) + out = np.zeros(len(u)) + for index in np.unique(level): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate(self.level_quantiles[index], row[take], v) + share = np.minimum(parts["level"][take] * np.exp(residual), 1.0) + out[take] = np.where(positive, share, 0.0) + return out + + # -- fitting -------------------------------------------------------------- + @classmethod + def fit( + cls, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + unit_years: tuple[int, ...], + rho_seed: int = 0, + ) -> tuple[OddQuantileFill, dict[str, object]]: + """Fit on complete TRAIN shares; every year of ``unit_years`` a unit. + + Every person-year of ``unit_years`` whose two neighbours are inside + the matrix is a training unit (all years are recorded on TRAIN). + """ + + shares = np.asarray(shares, dtype=np.float64) + known = np.isfinite(shares) + n = len(shares) + rows_list, years_list = [], [] + for year in unit_years: + rows_list.append(np.arange(n)) + years_list.append(np.full(n, year)) + rows = np.concatenate(rows_list) + unit_year = np.concatenate(years_list) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + target = _take(shares, rows, _column_of(years, unit_year)) + neighbours = np.concatenate([context.left, context.right]) + inside = neighbours[(neighbours > 0) & (neighbours < 1.0)] + share_edges = np.quantile( + inside, np.linspace(0, 1, _N_SHARE_BINS + 1)[1:-1] + ) + median = float(np.median(inside)) + parts = cls._parts(context, share_edges, median) + keys = cls._keys(parts) + positive = target > 0 + residual = np.log(np.where(positive, target, 1.0)) - np.log( + parts["level"] + ) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zeros_unique, zeros = np.unique( + key[in_cells & ~positive], return_counts=True + ) + totals = count[count >= MIN_CELL] + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zeros_unique)] = zeros + p0 = p0 / totals + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + # A populated cell with no positive unit draws only zeros. + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0.astype(np.float64)) + level_quantiles.append(full) + provisional = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=np.zeros((4, 16)), + ) + rho, rho_diagnostics = provisional._fit_rho( + parts, target, rows, unit_year, rho_seed + ) + fill = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=rho, + ) + cells = [len(k) for k in level_keys] + return fill, { + "n_units": int(len(target)), + "cells_per_level": cells, + **rho_diagnostics, + } + + def _pit(self, parts, target, rows, unit_year, seed) -> np.ndarray: + """Randomised probability integral transforms of true shares.""" + + level, row = self._cells(parts) + jitter = hash_uniform( + "epuf_fill.odd_quantile.pit", seed, rows, unit_year + ) + out = np.empty(len(target)) + for index in np.unique(level): + take = np.flatnonzero(level == index) + p0 = self.level_p0[index][row[take]] + zero = target[take] <= 0 + out[take[zero]] = jitter[take[zero]] * p0[zero] + positive = take[~zero] + residual = np.log(target[positive]) - np.log( + parts["level"][positive] + ) + # At the cap the residual is censored: spread it over the mass + # the quantile function puts at or above the cap. + at_cap = target[positive] >= 1.0 + v = _invert(self.level_quantiles[index], row[positive], residual) + cap_v = v.copy() + cap_v[at_cap] = v[at_cap] + jitter[positive][at_cap] * ( + 1.0 - v[at_cap] + ) + p0_positive = p0[~zero] + out[positive] = p0_positive + (1.0 - p0_positive) * cap_v + return np.clip(out, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, parts, target, rows, unit_year, seed): + """AR(1) correlation of consecutive units' normal scores (t, t+2). + + On a 5 percent sample of persons (by seed): every unit's + probability integral transform under the fitted cells, its normal + score, and the correlation of the scores of ``t`` and ``t+2`` for + the same person, by sex and age band at ``t``. + """ + + persons = np.unique(rows) + rng = np.random.default_rng(seed) + chosen = persons[rng.random(len(persons)) < 0.05] + index = np.flatnonzero(np.isin(rows, chosen)) + sub = {k: v[index] for k, v in parts.items()} + z = ndtri( + self._pit(sub, target[index], rows[index], unit_year[index], seed) + ) + first_year = int(unit_year.min()) + n_years = int(unit_year.max()) - first_year + 1 + position = np.searchsorted(chosen, rows[index]) + grid = np.full((len(chosen), n_years), np.nan) + grid[position, unit_year[index] - first_year] = z + sex_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + sex_grid[position, unit_year[index] - first_year] = sub["sex"] + age_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + age_grid[position, unit_year[index] - first_year] = sub["age"] + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex = sex_grid[:, :-2].ravel() + age = age_grid[:, :-2].ravel() + both = np.isfinite(now) & np.isfinite(later) + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = both & (sex == s) & (age == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + overall = float(np.corrcoef(now[both], later[both])[0, 1]) + four_now = grid[:, :-4].ravel() + four_later = grid[:, 4:].ravel() + four = np.isfinite(four_now) & np.isfinite(four_later) + lag4 = float(np.corrcoef(four_now[four], four_later[four])[0, 1]) + return rho, { + "rho_persons": int(len(chosen)), + "rho_pairs": int(both.sum()), + "rho_overall": overall, + "lag4_normal_score_correlation": lag4, + "lag4_ar1_prediction": overall**2, + } + + # -- filling -------------------------------------------------------------- + def fill( + self, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + person_key: np.ndarray, + fill_mask: np.ndarray, + seed: int, + ) -> np.ndarray: + """Fill the masked cells; masked cells with no known neighbour stay NaN.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + parts = self._parts(context, self.share_edges, self.context_median) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + rho = self.rho[np.clip(parts["sex"], 0, 3), parts["age"]] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + valid = parts["valid"] + drawn = np.full(len(rows), np.nan) + if valid.any(): + sub = {k: v[valid] for k, v in parts.items()} + drawn[valid] = self._quantile(sub, ndtr(z[valid])) + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + # -- persistence ---------------------------------------------------------- + def to_bytes(self) -> bytes: + arrays = { + "kind": np.array(self.name), + "share_edges": self.share_edges, + "context_median": np.array(self.context_median), + "rho": self.rho, + } + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> OddQuantileFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + share_edges=arrays["share_edges"], + context_median=float(arrays["context_median"]), + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + rho=arrays["rho"], + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: a quantile regression forest (QRF) draw +# -------------------------------------------------------------------------- + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +#: The reference level of a unit with no positive share around it. +_DEFAULT_LEVEL = 0.3 + + +def reference_level(context: OddContext) -> np.ndarray: + """The level a unit's share is drawn relative to. + + The geometric mean of the positive shares at ``t-1`` and ``t+1``; else + of those at ``t-3`` and ``t+3``; else the mean positive share at + offsets 5-9; else 0.3. + """ + + def geometric(a, b): + a = np.nan_to_num(a, nan=0.0) + b = np.nan_to_num(b, nan=0.0) + both = (a > 0) & (b > 0) + one = np.where(a > 0, a, b) + value = np.where(both, np.sqrt(np.where(both, a * b, 1.0)), one) + return np.where((a > 0) | (b > 0), value, np.nan) + + level = geometric(context.left, context.right) + level = np.where( + np.isnan(level), geometric(context.left3, context.right3), level + ) + level = np.where( + np.isnan(level) & (context.wide_mean > 0), context.wide_mean, level + ) + return np.where(np.isnan(level), _DEFAULT_LEVEL, level) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.91, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + Two parts, both random forests (scikit-learn) on :func:`odd_features` + of TRAIN units inside the career, whose contexts see the career only: + + 1. a probability forest for a zero year: ``p0`` is the mean over trees + of the zero share of the unit's leaves; + 2. a quantile regression forest on positive shares (split target + ``log share``): every positive TRAIN unit used in the fit is passed + down every tree, and each leaf keeps the sorted true shares that + reach it (the cap included, stored as shares times 65,535). + + A draw maps the copula uniform ``u`` to zero below ``p0``; otherwise a + second seeded uniform picks a tree, and the share is that tree's leaf + value at the quantile ``(u - p0) / (1 - p0)``. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + zero_tree_offsets: np.ndarray + zero_node_left: np.ndarray + zero_node_right: np.ndarray + zero_node_feature: np.ndarray + zero_node_threshold: np.ndarray + zero_node_leaf: np.ndarray + zero_leaf_p: np.ndarray + stream: str = "epuf_fill.odd_forest.v4" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + # Units lie inside the career, and their contexts see the career + # only, as a fill's do (pre-career years are unknown to it). + inside = unit_year >= career_start(birth_year[rows]) + rows, unit_year = rows[inside], unit_year[inside] + pre_career = years[None, :] < career_start(birth_year)[:, None] + known = np.isfinite(shares) & ~pre_career + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x_all = odd_features(context) + y_all = target[chosen] + # Part one: the probability of a zero year, a probability forest. + from sklearn.ensemble import RandomForestClassifier + + zero_forest = RandomForestClassifier( + n_estimators=n_trees, + min_samples_leaf=4 * min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + zero_forest.fit(x_all, (y_all <= 0).astype(np.int8)) + zero_arrays = _forest_arrays( + zero_forest, x_all, (y_all <= 0).astype(np.float64) + ) + # Part two: the positive share, a quantile regression forest. + positive = y_all > 0 + x = x_all[positive] + y = y_all[positive] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + **zero_arrays, + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y_all)), + "n_positive_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _p_zero(self, x) -> np.ndarray: + """The probability forest's zero-year probability (mean over trees).""" + + x = np.asarray(x, dtype=np.float32) + n_trees = len(self.zero_tree_offsets) - 1 + total = np.zeros(len(x)) + for tree in range(n_trees): + start = self.zero_tree_offsets[tree] + stop = self.zero_tree_offsets[tree + 1] + node = _tree_leaves( + self.zero_node_left[start:stop].astype(np.int64), + self.zero_node_right[start:stop].astype(np.int64), + self.zero_node_feature[start:stop].astype(np.int64), + self.zero_node_threshold[start:stop], + x, + ) + total += self.zero_leaf_p[ + self.zero_node_leaf[start:stop][node].astype(np.int64) + ] + return total / n_trees + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + p_zero = np.ones(len(rows)) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + p_zero[valid] = self._p_zero(features) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "p_zero": p_zero, + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + u = ndtr(z) + p0 = unit["p_zero"] + valid = valid & (u >= p0) + v = (u[valid] - p0[valid]) / np.maximum(1.0 - p0[valid], 1e-12) + drawn[valid] = self._value(unit["leaves"][valid], v) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + pre_career = years[None, :] < start[:, None] + mask &= ~pre_career + given = np.where(mask | pre_career, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +def _forest_arrays(forest, x, y) -> dict[str, np.ndarray]: + """A fitted probability forest as arrays: nodes, and each leaf's mean y.""" + + offsets = [0] + lefts, rights, features, thresholds, leaf_index, means = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + n_leaves = int(is_leaf.sum()) + index = np.full(tree.node_count, -1, dtype=np.int64) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + total = np.bincount(local, minlength=n_leaves) + hits = np.bincount(local, weights=y, minlength=n_leaves) + means.append(hits / np.maximum(total, 1)) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + offsets.append(offsets[-1] + tree.node_count) + return { + "zero_tree_offsets": np.asarray(offsets, dtype=np.int64), + "zero_node_left": np.concatenate(lefts).astype(np.int32), + "zero_node_right": np.concatenate(rights).astype(np.int32), + "zero_node_feature": np.concatenate(features), + "zero_node_threshold": np.concatenate(thresholds), + "zero_node_leaf": np.concatenate(leaf_index).astype(np.int32), + "zero_leaf_p": np.concatenate(means).astype(np.float32), + } + + +_FOREST_ARRAYS = ( + "zero_tree_offsets", + "zero_node_left", + "zero_node_right", + "zero_node_feature", + "zero_node_threshold", + "zero_node_leaf", + "zero_leaf_p", + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + # Units inside the career; contexts see the career only. + start = career_start(np.asarray(birth_year)) + inside = unit_year >= start[rows] + rows, unit_year = rows[inside], unit_year[inside] + known = np.isfinite(shares) & ~( + np.asarray(years)[None, :] < start[:, None] + ) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + # A missing neighbour takes the other's value, as in the draw. + left = np.where(np.isnan(context.left), context.right, context.left) + right = np.where(np.isnan(context.right), context.left, context.right) + usable = np.isfinite(left) & np.isfinite(right) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum[usable]): + members = np.flatnonzero((stratum == value) & usable) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=left[keep].astype(np.float32), + bank_right=right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: Nearest-donor lists by input content, reused across draw seeds. +_NEAREST_CACHE: dict = {} +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:_DONOR_BANK] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def _nearest(self, match, birth_year, sex, targets): + """Each target's ``k`` nearest bank rows, and its group's bank rows. + + Seed-free, so it is computed once for a matrix and reused across + draw seeds (cached by the content of its inputs). + """ + + digest = hashlib.sha256( + np.ascontiguousarray(match[targets]).tobytes() + + np.ascontiguousarray(birth_year[targets]).tobytes() + + np.ascontiguousarray(sex[targets]).tobytes() + + np.ascontiguousarray(targets).tobytes() + + str((id(self), self.k)).encode() + ).hexdigest() + if digest in _NEAREST_CACHE: + return _NEAREST_CACHE[digest] + nearest = np.full((len(targets), self.k), -1, dtype=np.int64) + group_first = np.full(len(targets), -1, dtype=np.int64) + group_size = np.zeros(len(targets), dtype=np.int64) + no_match = np.zeros(len(targets), dtype=bool) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + local = np.flatnonzero( + (sex[targets] == s) & (birth_year[targets] == b) + ) + recipients = targets[local] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + group_first[local] = donors[0] + group_size[local] = len(donors) + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + finite = np.isfinite(donor_match[:, j]) + column = np.sort(donor_match[finite, j]) + ranks_donor[:, j] = np.where( + finite, + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + order = np.argpartition(distance, k - 1, axis=1)[:, :k] + near_distance = np.take_along_axis(distance, order, 1) + ranked = np.lexsort((order, near_distance), axis=1) + order = np.take_along_axis(order, ranked, 1) + rows = local[block] + nearest[rows, :k] = donors[order] + no_match[rows] = count.max(axis=1) == 0 + result = (nearest, group_first, group_size, no_match) + if len(_NEAREST_CACHE) >= 4: + _NEAREST_CACHE.pop(next(iter(_NEAREST_CACHE))) + _NEAREST_CACHE[digest] = result + return result + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + One of the ``k`` nearest bank donors of the recipient's sex and + birth year, chosen by the seeded uniform; a recipient with no + recorded match feature takes a random donor of its group. -1 for + rows with no masked cell or no bank donor of their group. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + nearest, group_first, group_size, no_match = self._nearest( + match, birth_year, sex, targets + ) + out = np.full(len(shares), -1, dtype=np.int64) + has_group = group_size > 0 + k_available = (nearest >= 0).sum(axis=1) + pick = np.minimum( + (u[targets] * np.maximum(k_available, 1)).astype(np.int64), + np.maximum(k_available - 1, 0), + ) + chosen = nearest[np.arange(len(targets)), pick] + random_donor = group_first + np.minimum( + (u[targets] * np.maximum(group_size, 1)).astype(np.int64), + np.maximum(group_size - 1, 0), + ) + chosen = np.where(no_match, random_donor, chosen) + out[targets[has_group]] = chosen[has_group] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_quantile": OddQuantileFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl index 13e1e11d..c9402386 100644 --- a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl +++ b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl @@ -3,3 +3,14 @@ {"utc": "2026-10-04T07:31:20+00:00", "candidate": "pre_chain2", "family": "pre", "artifact_sha256": "37c2d72e6a7700cf7ac9dfe705434152f71e708ed92adf0fbad78611328b5761", "code_sha256": "ff279cd69d1b94ec1447d9d14db0d893546142cd34b54fa8b23e22b3e76a0584", "seeds": [7100, 7101, 7102, 7103], "n_failing": 55, "n_gating": 136, "tier": "not_adopted", "worst": [["pre.women.b1966_1980.ylevel", 0.45917079627409274, 0.017769920520911725], ["pre.men.b1966_1980.ylevel", 0.4811824306056769, 0.021382204828984532], ["pre.women.b1966_1980.yzero", 0.2583487735491933, 0.014833297517636257], ["pre.women.b1946_1980.yzero", 0.1489263401252473, 0.008556191699814752], ["pre.women.b1946_1980.ylevel", 0.20714311009839204, 0.015450091379218515], ["pre.men.b1946_1980.yzero", 0.15148656243554004, 0.011384493587046594], ["pre.men.b1966_1980.yzero", 0.19620307014155636, 0.01675184764087755], ["pre.men.b1946_1980.ylevel", 0.1430531564860713, 0.01520534060461241], ["pre.women.b1956_1965.yzero", 0.15643826543566397, 0.017102901274216892], ["pre.men.b1956_1965.yzero", 0.16956100424184672, 0.022693736955041323]]} {"utc": "2026-10-04T07:42:02+00:00", "candidate": "odd_qrf_sex2", "family": "odd", "artifact_sha256": "be03f0f3c702e391008fb72eec16d799385be93fbd5de01ce78a79e1fe45330b", "code_sha256": "9b2c5dd94d3d51e5f53994b7aa707ce7955b99eeb1e6e8bc481e753d1260761a", "seeds": [7100, 7101, 7102, 7103], "n_failing": 23, "n_gating": 183, "tier": "not_adopted", "worst": [["odd.women.a22_29.r2", 0.02714856253605591, 0.013111971249341917], ["odd.men.a60_74.zint", 0.33247368988113246, 0.17758159695027936], ["odd.women.a60_74.r2", -0.03422723027556407, 0.022174026993056747], ["odd.men.a22_74.r2", -0.007062508310260118, 0.004576681103770273], ["odd.men.a22_29.r1", -0.010609081809470733, 0.007273490161268873], ["odd.women.a30_44.r4", -0.014271254170137415, 0.010187042430903364], ["odd.men.a30_44.zexit", -0.04086629032766431, 0.031953074094816486], ["odd.men.a60_74.r2", -0.02629022691682592, 0.020757725353477027], ["odd.men.a30_44.r4", -0.011478929773406477, 0.009218001294372115], ["odd.women.a45_59.r2", -0.009149018075044535, 0.0074184353060767995], ["odd.men.a22_74.zexit", -0.022985129792897574, 0.01916127382099423], ["odd.women.a60_74.r1", -0.011531037538756617, 0.009751654703991025], ["odd.men.a30_44.r2", -0.0075100613637079094, 0.006500650570758907], ["odd.women.a22_29.level", 0.020540551604554036, 0.018517427407013797], ["odd.men.a22_74.r4", -0.007762715111083396, 0.007000982741779954], ["odd.women.a60_74.r4", -0.03868148727219123, 0.036314307016843086], ["odd.women.a22_29.q50", 0.0232742152048111, 0.021849929026002995], ["odd.men.a45_59.r2", -0.007692352520425327, 0.0074262368998509335], ["odd.women.a45_59.r4", -0.012890705852779516, 0.012512985825002505], ["odd.women.a22_74.r4", -0.007885437242168614, 0.007666262733486471], ["odd.men.a22_29.r3", -0.01343644400734656, 0.013280307041936976], ["odd.men.a22_29.q50", 0.023239013149086052, 0.022980444394274636], ["odd.women.a30_44.r2", -0.007087772863819786, 0.007078708407960211], ["odd.men.a30_44.q10", -0.04751745643811356, 0.047669064141889934], ["odd.women.a22_74.r3", -0.005150669696892374, 0.005171487109756781], ["odd.men.a30_44.zint", -0.08884255086813031, 0.08989768311292742], ["odd.women.a22_29.r4", 0.01925238657360928, 0.01982762523559322], ["odd.men.a60_74.r1", -0.009265651157384092, 0.009750532797445479], ["odd.men.a22_29.level", 0.020253468381523643, 0.02171146931539837], ["odd.men.a22_74.r3", -0.004148209781093093, 0.004722018684826323]]} {"utc": "2026-10-04T07:44Z (approximate)", "candidate": "pre_chain3 (diagnostic check, not scored against tolerances)", "family": "pre", "description": "the chained alternative refitted with single-year ages 15-24 and unit years 1951-2005; DEV persons born 1966-1980 (men, 30,000) filled with seed 7100 and compared with the truth by age", "printed": {"15": {"truth_mean": 0.0028, "truth_zero": 0.857, "fill_mean": 0.0081, "fill_zero": 0.863}, "16": {"truth_mean": 0.0094, "truth_zero": 0.661, "fill_mean": 0.0189, "fill_zero": 0.677}, "17": {"truth_mean": 0.0222, "truth_zero": 0.488, "fill_mean": 0.0406, "fill_zero": 0.508}, "18": {"truth_mean": 0.0409, "truth_zero": 0.369, "fill_mean": 0.0737, "fill_zero": 0.387}, "19": {"truth_mean": 0.0697, "truth_zero": 0.304, "fill_mean": 0.1113, "fill_zero": 0.331}, "20": {"truth_mean": 0.0934, "truth_zero": 0.287, "fill_mean": 0.1285, "fill_zero": 0.311}, "21": {"truth_mean": 0.1116, "truth_zero": 0.276, "fill_mean": 0.134, "fill_zero": 0.294}}, "earlier_check": "the same comparison with pre_chain2 (five-year age bands) printed fill means 0.0236-0.1416 against truth 0.0028-0.1116 at ages 15-21", "code_sha256": "9b2c5dd94d3d51e5f53994b7aa707ce7955b99eeb1e6e8bc481e753d1260761a"} +{"utc": "2026-10-04T07:54:03+00:00", "candidate": "odd_qrf3", "family": "odd", "artifact_sha256": "28d52765c4f540909e9d5c1deb13724e7e222d6feac27c9aed797cb0516f0993", "code_sha256": "36e045bdc94681f5cc54fed30b0fd447fc4f87bacd4239c84ca4c1b6c03a027b", "seeds": [7100, 7101, 7102, 7103], "n_failing": 12, "n_gating": 183, "tier": "improves", "worst": [["odd.women.a22_29.r2", 0.027344747016985638, 0.013111971249341917], ["odd.men.a60_74.zint", 0.3415042291139394, 0.17758159695027936], ["odd.men.a30_44.zexit", -0.0511542734002437, 0.031953074094816486], ["odd.men.a22_29.r1", -0.011611591760373075, 0.007273490161268873], ["odd.men.a22_74.zexit", -0.027937271991644974, 0.01916127382099423], ["odd.women.a30_44.r4", -0.013133885403330492, 0.010187042430903364], ["odd.women.a60_74.r1", -0.011707903966676314, 0.009751654703991025], ["odd.women.a22_74.r3", -0.005771964915830763, 0.005171487109756781], ["odd.men.a22_29.q50", 0.024899453547185146, 0.022980444394274636], ["odd.men.a22_29.level", 0.02192345936691109, 0.02171146931539837], ["odd.women.a22_29.level", 0.018596125147572584, 0.018517427407013797], ["odd.men.a22_29.r3", -0.01329969151609589, 0.013280307041936976], ["odd.men.a45_59.zexit", -0.0351852359581335, 0.036314919811016276], ["odd.men.a30_44.zint", -0.08623838272974282, 0.08989768311292742], ["odd.men.a30_44.q10", -0.04558952377760228, 0.047669064141889934], ["odd.women.a22_29.q50", 0.020520979391902117, 0.021849929026002995], ["odd.women.a30_44.zint", -0.06508790701647582, 0.07029190354044747], ["odd.men.a22_74.r3", -0.004260068220721669, 0.004722018684826323], ["odd.men.a22_29.q10", 0.06445677303612518, 0.07329266358707544], ["odd.men.a60_74.r1", -0.008430272781665082, 0.009750532797445479], ["odd.men.a22_74.r1", -0.002051107586787837, 0.00258049589094515], ["odd.men.a22_29.q90", 0.01862070347052136, 0.02353059621893112], ["odd.women.a22_29.r4", 0.015341851710927168, 0.01982762523559322], ["odd.women.a22_74.r4", -0.0058897478528845415, 0.007666262733486471], ["odd.women.a22_74.zexit", -0.011878230212011065, 0.016039678883419197]]} +{"utc": "2026-10-04T07:59:25+00:00", "candidate": "odd_qrf4", "family": "odd", "artifact_sha256": "9b92fd98cd4021171c838bf4a8852ec25c2a396049eecbc1bca4ef9e8fcdbd2a", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101, 7102, 7103], "n_failing": 8, "n_gating": 183, "tier": "improves", "worst": [["odd.men.a22_29.r1", -0.01998280153218135, 0.007273490161268873], ["odd.women.a22_29.r1", -0.012394194276425408, 0.006445547562698701], ["odd.women.a22_29.zint", 0.14907864826487227, 0.09342121518593281], ["odd.men.a22_29.zint", 0.16973294481418666, 0.12140160693630424], ["odd.women.a22_74.zint", 0.055533732527217605, 0.04476234327525859], ["odd.women.a60_74.r1", -0.010935700430824924, 0.009751654703991025], ["odd.women.a22_74.r1", -0.0025564811431657564, 0.002475922564645967], ["odd.men.a22_74.r1", -0.002637012673456507, 0.00258049589094515], ["odd.men.a22_29.r3", -0.012301908709043796, 0.013280307041936976], ["odd.women.a22_74.r3", -0.004555637079359798, 0.005171487109756781], ["odd.men.a22_74.r3", -0.003944498885934511, 0.004722018684826323], ["odd.men.a60_74.r1", -0.008071695817202462, 0.009750532797445479], ["odd.women.a22_29.wint", 0.12118226962992029, 0.15037337545812676], ["odd.men.a22_74.zint", 0.043294991137900585, 0.05481211657295162], ["odd.women.a22_29.r2", 0.009530501163799499, 0.013111971249341917], ["odd.men.a30_44.r4", -0.006666088981186702, 0.009218001294372115], ["odd.women.a30_44.r4", -0.007245568446286876, 0.010187042430903364], ["odd.men.a22_74.r4", -0.0049113140287062595, 0.007000982741779954], ["odd.women.a22_29.r3", -0.00828787522771579, 0.012054115899361516], ["odd.men.a30_44.r3", -0.004138657644559562, 0.006087370619188444], ["odd.men.a60_74.zint", 0.11754876132006542, 0.17758159695027936], ["odd.women.a22_74.r2", 0.003035764997925239, 0.004772836686607259]]} +{"utc": "2026-10-04T08:01:26+00:00", "candidate": "pre_donor_k25", "family": "pre", "artifact_sha256": "eafa0ae5611817c69b800b42fa0798e71bc3fe34b2a946359eb89042b8879443", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 6, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1946_1980.yr_cross", 0.029133392188555096, 0.015684351176846988], ["pre.women.b1946_1980.yzero", 0.015405372793597327, 0.008556191699814752], ["pre.women.b1956_1965.yr_cross", 0.05360057865508633, 0.03047360671843986], ["pre.women.b1946_1955.yr_cross", 0.04836863108053191, 0.03669253843323441], ["pre.women.b1966_1980.yzero", 0.0182391249424394, 0.014833297517636257], ["pre.men.b1946_1980.yzero", 0.013244286631349356, 0.011384493587046594], ["pre.women.b1946_1955.yzero", 0.015014321048421264, 0.015053483015288157], ["pre.women.b1966_1980.yr_cross", 0.021044902116193698, 0.02158530789673962]]} +{"utc": "2026-10-04T08:02:45+00:00", "candidate": "pre_donor_k50", "family": "pre", "artifact_sha256": "d53cded2b59ec6452f52f1ca4b30a2dc97f1aa1ec9a989e03b32f201edf3e910", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 13, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1946_1980.yzero", 0.020559950615929745, 0.008556191699814752], ["pre.women.b1946_1980.yr_cross", 0.034508472442058846, 0.015684351176846988], ["pre.women.b1956_1965.yr_cross", 0.05776594239087035, 0.03047360671843986], ["pre.men.b1946_1980.yzero", 0.018480756025274547, 0.011384493587046594], ["pre.women.b1946_1955.yzero", 0.023269971969184455, 0.015053483015288157], ["pre.women.b1966_1980.yzero", 0.0200676390110347, 0.014833297517636257], ["pre.women.b1966_1980.yr_cross", 0.029076933330211885, 0.02158530789673962], ["pre.men.b1946_1980.yr_cross", 0.018382403592013652, 0.014632893176957621]]} +{"utc": "2026-10-04T08:04:25+00:00", "candidate": "pre_donor_k5", "family": "pre", "artifact_sha256": "6c739ba6f7437a1fc104a268ceb1f90d184fc1f60f3be811d4d35217e245d21e", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 4, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1956_1965.yr_cross", 0.04463879596543152, 0.03047360671843986], ["pre.women.b1946_1980.yr_cross", 0.02293633800904532, 0.015684351176846988], ["pre.women.b1946_1980.yzero", 0.011766480578584093, 0.008556191699814752], ["pre.women.b1946_1955.yr_cross", 0.03914897033531145, 0.03669253843323441], ["pre.men.b1946_1980.yzero", 0.010066340298164667, 0.011384493587046594], ["pre.women.b1966_1980.yzero", 0.012846678222160568, 0.014833297517636257], ["pre.women.b1966_1980.yr_cross", 0.01800530689095292, 0.02158530789673962], ["pre.men.b1946_1980.yr_cross", 0.012084512584237206, 0.014632893176957621]]} +{"utc": "2026-10-04T08:05:54+00:00", "candidate": "pre_donor_k3", "family": "pre", "artifact_sha256": "9902743dfd0f0b17d7ec063421d3a25da2a9de9987942b7508cbc0c4694eedfc", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 3, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1946_1980.yzero", 0.011163616040424484, 0.008556191699814752], ["pre.women.b1946_1980.yr_cross", 0.01978006406808891, 0.015684351176846988], ["pre.women.b1956_1965.yr_cross", 0.03759069775273621, 0.03047360671843986], ["pre.men.b1946_1980.yzero", 0.009549309004077244, 0.011384493587046594], ["pre.women.b1966_1980.yr_cross", 0.017873523200933716, 0.02158530789673962], ["pre.women.b1966_1980.yzero", 0.01194888116310322, 0.014833297517636257]]} +{"utc": "2026-10-04T08:07:00+00:00", "candidate": "pre_donor_k1", "family": "pre", "artifact_sha256": "bc7a7d28a503cfdfdba022ef71387d1420f7ad9da1471c2ba2dba58b775f5c6b", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 4, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1946_1980.yzero", 0.00991715298416973, 0.008556191699814752], ["pre.women.b1956_1965.yr_cross", 0.03522344626992119, 0.03047360671843986], ["pre.women.b1946_1980.yr_cross", 0.016289968599504656, 0.015684351176846988], ["pre.women.b1966_1980.yzero", 0.015207687342106424, 0.014833297517636257], ["pre.men.b1946_1980.yzero", 0.011275693354747762, 0.011384493587046594], ["pre.men.b1966_1980.yzero", 0.013394282698992344, 0.01675184764087755]]} +{"utc": "2026-10-04T08:10:45+00:00", "candidate": "pre_donor_b5k3", "family": "pre", "artifact_sha256": "47304b01a2c8a3ebdc8d9799efc7f86c0e501c925ba232191a420412980ea950", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 0, "n_gating": 136, "tier": "certified", "worst": [["pre.women.b1956_1965.yr_cross", 0.03042477794600501, 0.03047360671843986], ["pre.men.b1946_1980.yzero", 0.010177331108892296, 0.011384493587046594], ["pre.women.b1946_1980.yzero", 0.007455221837990744, 0.008556191699814752], ["pre.men.b1946_1980.ylevel", -0.011326381911699102, 0.01520534060461241], ["pre.women.b1946_1980.yr_cross", 0.011465522706747888, 0.015684351176846988], ["pre.men.b1946_1980.yr_cross", 0.010614564663255555, 0.014632893176957621]]} +{"utc": "2026-10-04T08:23:28+00:00", "candidate": "pre_donor_ball_k3", "family": "pre", "artifact_sha256": "c472c84325711c8856dc8067d02ef3c9c27cb8bce63befae22df2d18cab50062", "code_sha256": "4be7556f2263ef7890929ef27a92975d93938415e22153b38bf0afd01fef5983", "seeds": [7100, 7101, 7102, 7103], "n_failing": 0, "n_gating": 136, "tier": "certified", "worst": [["pre.men.b1946_1980.yzero", 0.00809038443776322, 0.011384493587046594], ["pre.women.b1946_1980.yzero", 0.005732645830089589, 0.008556191699814752], ["pre.men.b1956_1965.yzero", 0.015012948537747373, 0.022693736955041323], ["pre.women.b1956_1965.yr_cross", 0.019137592768893985, 0.03047360671843986], ["pre.women.b1946_1980.yr_cross", 0.00910100469507158, 0.015684351176846988], ["pre.men.b1946_1980.ylevel", -0.008408704694680136, 0.01520534060461241], ["pre.women.b1930_1945.pr_cross", 0.036302636511399144, 0.0677006808491353], ["pre.women.b1930_1945.pr_in", 0.03831971636456638, 0.07412154624977427]]} +{"utc": "2026-10-04T08:27:20+00:00", "candidate": "odd_knn2", "family": "odd", "artifact_sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 47, "n_gating": 183, "tier": "not_adopted", "worst": [["odd.men.a22_74.wint", -0.7579019412918284, 0.07337302886151653], ["odd.women.a22_74.wint", -0.6178668454511387, 0.07133033012317215], ["odd.men.a45_59.wint", -1.0410721850282765, 0.13344110142481172], ["odd.women.a45_59.wint", -0.9735852305534904, 0.1271041647107106], ["odd.men.a30_44.wint", -0.7720545590267562, 0.12001713872675526], ["odd.women.a30_44.wint", -0.5937411523100029, 0.10093703169083498]]} +{"utc": "2026-10-04T08:27:45+00:00", "candidate": "pre_chain3", "family": "pre", "artifact_sha256": "74c62142e1de4bfed543de3b1d8c88c0d70c674a8a2fa44129bfe5943d4ff0ea", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 50, "n_gating": 136, "tier": "not_adopted", "worst": [["pre.women.b1966_1980.ylevel", 0.35125489825189504, 0.017769920520911725], ["pre.men.b1966_1980.ylevel", 0.35211188653405623, 0.021382204828984532], ["pre.women.b1966_1980.yzero", 0.13965247274046733, 0.014833297517636257], ["pre.women.b1946_1980.ylevel", 0.1097696357515745, 0.015450091379218515], ["pre.men.b1930_1945.plevel", -0.18375090914832892, 0.029173903026433003], ["pre.women.b1930_1945.pzero", -0.18200182661313102, 0.02913422225552515]]} diff --git a/scripts/first_estimates_birth_evidence.py b/scripts/first_estimates_birth_evidence.py index 18b864e4..9108c560 100644 --- a/scripts/first_estimates_birth_evidence.py +++ b/scripts/first_estimates_birth_evidence.py @@ -355,6 +355,10 @@ # EPUF careers after the fact; nothing historical imports them. Path("src/populace_dynamics/harness/epuf_fill_gate.py"), Path("src/populace_dynamics/harness/epuf_fill_scoring.py"), + # The opt-in learned EPUF career fills and their application to a + # built PSID-2010 cohort; nothing historical imports them. + Path("src/populace_dynamics/estimates/epuf_fill.py"), + Path("src/populace_dynamics/cohorts/psid2010_epuf_fill.py"), ) POST_REVIEW_SHARED_SOURCE_BLOBS = { Path( diff --git a/scripts/fit_epuf_fills.py b/scripts/fit_epuf_fills.py new file mode 100644 index 00000000..b683459c --- /dev/null +++ b/scripts/fit_epuf_fills.py @@ -0,0 +1,225 @@ +"""Fit gate_epuf_fill's four registered candidate fills on EPUF TRAIN. + +Reads only the TRAIN part of the pinned EPUF +(``epuf_fill_gate.epuf_matrix(TRAIN)``) and fits, with the registered +parameters below: + +- ``odd_forest`` (odd years, primary): :class:`BySexFill` of + :class:`OddForestFill`; +- ``odd_knn`` (odd years, alternative): :class:`OddKnnFill`; +- ``pre_donor`` (pre-career years, primary): :class:`PreDonorFill`; +- ``pre_chain`` (pre-career years, alternative): :class:`PreChainFill`. + +Each fill is written as a byte-reproducible ``.npz`` outside the repository +(``~/PolicyEngine/epuf-data/fills``, or ``POPULACE_DYNAMICS_EPUF_FILLS_DIR``), +as EPUF itself is. The manifest records each file's SHA-256, size and +parameters, the code commit, whether the fill code was clean, and the +library versions; ``epuf_fill_scoring`` loads the candidates by that +SHA-256. Usage:: + + python scripts/fit_epuf_fills.py --manifest runs/epuf_fill_candidates_v1.json +""" + +from __future__ import annotations + +import argparse +import datetime as dt +import hashlib +import os +import subprocess +import time +from pathlib import Path + +import numpy as np + +from populace_dynamics.artifacts import write_new +from populace_dynamics.estimates import epuf_fill as F +from populace_dynamics.harness import epuf_fill_gate as g +from populace_dynamics.harness.epuf_operator import wage_base + +ROOT = Path(__file__).resolve().parents[1] +SCHEMA = "populace_dynamics.epuf_fill_candidates.v1" +DEFAULT_DIR = Path("~/PolicyEngine/epuf-data/fills").expanduser() +CODE_FILES = ( + "src/populace_dynamics/estimates/epuf_fill.py", + "scripts/fit_epuf_fills.py", +) +ODD_UNIT_YEARS = tuple(range(1991, 2006)) +PRE_UNIT_YEARS = tuple(range(1951, 2006)) +#: The registered parameters of each candidate. +REGISTERED = { + "odd_forest": { + "family": "odd", + "role": "primary", + "params": { + "unit_years": [ODD_UNIT_YEARS[0], ODD_UNIT_YEARS[-1]], + "n_units": 3_000_000, + "n_trees": 10, + "min_leaf": 15, + "max_features": 0.8, + "seed": 0, + }, + }, + "odd_knn": { + "family": "odd", + "role": "alternative", + "params": { + "unit_years": [ODD_UNIT_YEARS[0], ODD_UNIT_YEARS[-1]], + "k": 10, + "seed": 0, + }, + }, + "pre_donor": { + "family": "pre", + "role": "primary", + "params": {"k": 3, "bank_size": 100_000, "birth_years": [1905, 1985]}, + }, + "pre_chain": { + "family": "pre", + "role": "alternative", + "params": {"unit_years": [PRE_UNIT_YEARS[0], PRE_UNIT_YEARS[-1]]}, + }, +} + + +def _git(*args: str) -> str: + return subprocess.run( + ["git", "-C", str(ROOT), *args], + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +def _clean() -> bool: + return ( + subprocess.run( + ["git", "-C", str(ROOT), "diff", "--quiet", "HEAD", "--"] + + list(CODE_FILES), + check=False, + ).returncode + == 0 + ) + + +def fit(name: str, shares, birth, sex, person_id): + params = REGISTERED[name]["params"] + if name == "odd_forest": + return F.BySexFill.fit( + F.OddForestFill, + shares, + g.YEARS, + birth, + sex, + ODD_UNIT_YEARS, + person_id, + n_units=params["n_units"], + n_trees=params["n_trees"], + min_leaf=params["min_leaf"], + max_features=params["max_features"], + seed=params["seed"], + ) + if name == "odd_knn": + return F.OddKnnFill.fit( + shares, + np.asarray(g.YEARS), + birth, + sex, + ODD_UNIT_YEARS, + k=params["k"], + seed=params["seed"], + ) + if name == "pre_donor": + return F.PreDonorFill.fit( + shares, + np.asarray(g.YEARS), + birth, + sex, + person_id, + k=params["k"], + birth_years=tuple(params["birth_years"]), + bank_size=params["bank_size"], + ) + if name == "pre_chain": + return F.PreChainFill.fit( + shares, np.asarray(g.YEARS), birth, sex, PRE_UNIT_YEARS + ) + raise ValueError(name) + + +def main() -> None: + import scipy + import sklearn + + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--manifest", type=Path, required=True) + parser.add_argument( + "--out-dir", + type=Path, + default=Path( + os.environ.get("POPULACE_DYNAMICS_EPUF_FILLS_DIR", DEFAULT_DIR) + ), + ) + parser.add_argument("--only", nargs="*", default=sorted(REGISTERED)) + args = parser.parse_args() + started = time.time() + built_at = dt.datetime.now(dt.UTC).isoformat(timespec="seconds") + matrix = g.epuf_matrix(g.TRAIN) + caps = np.array([float(wage_base(year)) for year in g.YEARS]) + shares = matrix.earnings / caps[None, :] + args.out_dir.mkdir(parents=True, exist_ok=True) + fills = {} + for name in args.only: + t = time.time() + fill, diagnostics = fit( + name, shares, matrix.birth_year, matrix.sex, matrix.person_id + ) + blob = fill.to_bytes() + path = args.out_dir / f"{name}_v1.npz" + path.write_bytes(blob) + fills[name] = { + **REGISTERED[name], + "file": path.name, + "sha256": hashlib.sha256(blob).hexdigest(), + "bytes": len(blob), + "fit_seconds": round(time.time() - t, 1), + "diagnostics": _jsonable(diagnostics), + } + print(f"{name}: {fills[name]['sha256']} {len(blob)} bytes", flush=True) + document = { + "schema": SCHEMA, + "registration_id": g.REGISTRATION_ID, + "code_commit": _git("rev-parse", "HEAD"), + "code_files_clean": _clean(), + "built_at_utc": built_at, + "part": "train", + "n_persons": int(len(matrix.birth_year)), + "versions": { + "numpy": np.__version__, + "scipy": scipy.__version__, + "scikit_learn": sklearn.__version__, + }, + "staging": "files live outside the repository, like EPUF; refit " + "with this script at code_commit to reproduce their bytes", + "fills": fills, + "elapsed_seconds": round(time.time() - started, 1), + } + write_new(args.manifest, document, sidecar=True) + + +def _jsonable(value): + if isinstance(value, dict): + return {str(k): _jsonable(v) for k, v in value.items()} + if isinstance(value, list | tuple): + return [_jsonable(v) for v in value] + if isinstance(value, np.ndarray): + return _jsonable(value.tolist()) + if isinstance(value, float | np.floating): + return float(value) if np.isfinite(value) else str(float(value)) + if isinstance(value, np.integer): + return int(value) + return value + + +if __name__ == "__main__": + main() diff --git a/src/populace_dynamics/cohorts/psid2010_epuf_fill.py b/src/populace_dynamics/cohorts/psid2010_epuf_fill.py new file mode 100644 index 00000000..8a115945 --- /dev/null +++ b/src/populace_dynamics/cohorts/psid2010_epuf_fill.py @@ -0,0 +1,191 @@ +"""Learned EPUF career fills applied to a built PSID-2010 cohort (opt-in). + +The PSID-2010 cohort's careers (:func:`populace_dynamics.cohorts.psid2010. +build_psid2010_cohort`) come from ``career.build_career``, which fills each +odd income year from 1997 with its neighbours' mean (provenance +``gap_imputed``) and counts nothing before ``max(1968, birth_year + 22)``. +:func:`fill_careers` replaces either rule with a fill learned from SSA's +Earnings Public-Use File and registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``): + +- every ``gap_imputed`` year becomes the odd fill's draw, with provenance + :attr:`EPUFFillProvenance.GAP_EPUF_DRAWN`; +- every year from 1951 before the career start becomes the pre-career + fill's draw, with provenance + :attr:`EPUFFillProvenance.PRE_CAREER_EPUF_DONOR`. + +The fills see what the gate's scoring path gives them: +- capped shares of the wage base for the career years the PSID recorded; +- every other year unknown, including pre-career years the PSID happened to + record. + +A drawn share becomes capped earnings at that year's wage base. Every other +career row is left exactly as built. Nothing here changes +``career.build_career``, ``cohorts.psid2010`` or any registered run; using +learned fills in a registered comparison needs its own registration. +""" + +from __future__ import annotations + +import hashlib +from dataclasses import dataclass +from enum import Enum +from typing import Any + +import numpy as np +import pandas as pd + +__all__ = [ + "EPUFFillProvenance", + "FilledCareers", + "fill_careers", +] + +FIRST_YEAR = 1951 +_SEX_CODE = {"male": 1, "female": 2} + + +class EPUFFillProvenance(str, Enum): + """Provenance of a career year filled by a learned EPUF fill.""" + + GAP_EPUF_DRAWN = "gap_epuf_drawn" + PRE_CAREER_EPUF_DONOR = "pre_career_epuf_donor" + + +@dataclass(frozen=True) +class FilledCareers: + """Careers after the learned fills, with what produced them.""" + + careers: pd.DataFrame + fills: dict[str, str] + seed: int + content_sha256: str + + +def _wage_bases(years: np.ndarray) -> np.ndarray: + from populace_dynamics.cola_track_a.statutory import ( + captured_ssa_parameters, + ) + + params = captured_ssa_parameters() + return np.array([float(params.wage_base_for(int(y))) for y in years]) + + +def _content_sha256(careers: pd.DataFrame) -> str: + ordered = careers.sort_values(["person_id", "year"], kind="stable") + return hashlib.sha256( + ordered.to_csv(index=False, float_format="%.6f").encode() + ).hexdigest() + + +def fill_careers( + cohort: Any, + *, + odd_fill: Any | None = None, + pre_fill: Any | None = None, + seed: int, + start_year: int | None = None, +) -> FilledCareers: + """Replace the assembler's fill rules with learned fills. + + ``cohort`` needs ``persons`` (``person_id``, ``birth_year``, ``sex`` as + ``"male"`` / ``"female"``) and ``careers`` (``person_id``, ``year``, + ``earnings``, ``provenance``). Either fill may be None, which keeps that + rule. ``start_year`` defaults to the latest career year. + """ + + persons = cohort.persons[["person_id", "birth_year", "sex"]].copy() + careers = cohort.careers.copy() + last = int(careers["year"].max()) if start_year is None else start_year + years = np.arange(FIRST_YEAR, last + 1) + caps = _wage_bases(years) + person_ids = persons["person_id"].to_numpy(dtype=np.int64) + row_of = {int(pid): i for i, pid in enumerate(person_ids)} + birth = persons["birth_year"].to_numpy(dtype=np.int64) + sex = persons["sex"].map(_SEX_CODE).fillna(3).to_numpy(dtype=np.int64) + n = len(person_ids) + + shares = np.full((n, len(years)), np.nan) + odd_mask = np.zeros((n, len(years)), dtype=bool) + in_career = np.zeros((n, len(years)), dtype=bool) + rows = careers["person_id"].map(row_of).to_numpy() + if np.isnan(rows.astype(float)).any(): + raise ValueError("a career row's person is not in the cohort") + columns = careers["year"].to_numpy(dtype=np.int64) - FIRST_YEAR + inside = (columns >= 0) & (columns < len(years)) + rows, columns = rows[inside].astype(np.int64), columns[inside] + provenance = careers["provenance"].astype(str).to_numpy()[inside] + earnings = careers["earnings"].to_numpy(dtype=np.float64)[inside] + in_career[rows, columns] = True + observed = provenance == "observed" + shares[rows[observed], columns[observed]] = np.minimum( + np.maximum(earnings[observed], 0.0) / caps[columns[observed]], 1.0 + ) + gap = provenance == "gap_imputed" + odd_mask[rows[gap], columns[gap]] = True + start = np.maximum(1968, birth + 22) + pre_mask = years[None, :] < start[:, None] + given = np.where(odd_mask | pre_mask, np.nan, shares) + + def drawn(fill, mask): + out = np.asarray( + fill.fill( + given.copy(), years, birth, sex, person_ids, mask.copy(), seed + ), + dtype=np.float64, + ) + values = out[mask] + if ( + not np.isfinite(values).all() + or ((values < 0) | (values > 1)).any() + ): + raise ValueError("a learned fill returned an invalid share") + return out + + result = careers.copy() + names: dict[str, str] = {} + if odd_fill is not None: + out = drawn(odd_fill, odd_mask) + cells = out[rows[gap], columns[gap]] + values = np.where( + cells >= 1.0, caps[columns[gap]], cells * caps[columns[gap]] + ) + index = careers.index[inside][gap] + result.loc[index, "earnings"] = values + result.loc[index, "provenance"] = ( + EPUFFillProvenance.GAP_EPUF_DRAWN.value + ) + names["odd"] = getattr(odd_fill, "name", type(odd_fill).__name__) + if pre_fill is not None: + out = drawn(pre_fill, pre_mask) + pre_rows, pre_columns = np.nonzero(pre_mask & ~in_career) + cells = out[pre_rows, pre_columns] + added = pd.DataFrame( + { + "person_id": person_ids[pre_rows], + "year": years[pre_columns], + "earnings": np.where( + cells >= 1.0, + caps[pre_columns], + cells * caps[pre_columns], + ), + "provenance": EPUFFillProvenance.PRE_CAREER_EPUF_DONOR.value, + } + ) + result = pd.concat([added, result], ignore_index=True) + names["pre"] = getattr(pre_fill, "name", type(pre_fill).__name__) + result = result.sort_values(["person_id", "year"], kind="stable") + result = result.reset_index(drop=True).astype( + { + "person_id": "int64", + "year": "int64", + "earnings": "float64", + "provenance": "string", + } + ) + return FilledCareers( + careers=result, + fills=names, + seed=int(seed), + content_sha256=_content_sha256(result), + ) diff --git a/src/populace_dynamics/estimates/epuf_fill.py b/src/populace_dynamics/estimates/epuf_fill.py new file mode 100644 index 00000000..9b8a6861 --- /dev/null +++ b/src/populace_dynamics/estimates/epuf_fill.py @@ -0,0 +1,1665 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddForestFill` (odd years, primary; fitted per sex through + :class:`BySexFill`): a two-part draw from random forests. A probability + forest gives the chance of a zero year, and a quantile regression forest + gives the positive share. Both condition on the recorded shares around + ``t``, sex, age and year. A person-level Gaussian copula, calibrated on + held-out TRAIN persons, carries the persistence across a person's masked + years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years and two career-wide summaries. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + """A compressed ``.npz`` whose bytes depend only on the arrays. + + ``numpy.savez_compressed`` stamps each member with the time of writing, + so two writes of the same fill differ. This writer fixes every member's + timestamp and order, so a fill's SHA-256 can be registered and refit. + """ + + import zipfile + + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive: + for name in sorted(arrays): + member = io.BytesIO() + np.lib.format.write_array( + member, np.asanyarray(arrays[name]), allow_pickle=False + ) + info = zipfile.ZipInfo( + f"{name}.npy", date_time=(1980, 1, 1, 0, 0, 0) + ) + info.compress_type = zipfile.ZIP_DEFLATED + archive.writestr(info, member.getvalue()) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.91, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + Two parts, both random forests (scikit-learn) on :func:`odd_features` + of TRAIN units inside the career, whose contexts see the career only: + + 1. a probability forest for a zero year: ``p0`` is the mean over trees + of the zero share of the unit's leaves; + 2. a quantile regression forest on positive shares (split target + ``log share``): every positive TRAIN unit used in the fit is passed + down every tree, and each leaf keeps the sorted true shares that + reach it (the cap included, stored as shares times 65,535). + + A draw maps the copula uniform ``u`` to zero below ``p0``; otherwise a + second seeded uniform picks a tree, and the share is that tree's leaf + value at the quantile ``(u - p0) / (1 - p0)``. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + zero_tree_offsets: np.ndarray + zero_node_left: np.ndarray + zero_node_right: np.ndarray + zero_node_feature: np.ndarray + zero_node_threshold: np.ndarray + zero_node_leaf: np.ndarray + zero_leaf_p: np.ndarray + stream: str = "epuf_fill.odd_forest.v4" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + # Units lie inside the career, and their contexts see the career + # only, as a fill's do (pre-career years are unknown to it). + inside = unit_year >= career_start(birth_year[rows]) + rows, unit_year = rows[inside], unit_year[inside] + pre_career = years[None, :] < career_start(birth_year)[:, None] + known = np.isfinite(shares) & ~pre_career + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x_all = odd_features(context) + y_all = target[chosen] + # Part one: the probability of a zero year, a probability forest. + from sklearn.ensemble import RandomForestClassifier + + zero_forest = RandomForestClassifier( + n_estimators=n_trees, + min_samples_leaf=4 * min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + zero_forest.fit(x_all, (y_all <= 0).astype(np.int8)) + zero_arrays = _forest_arrays( + zero_forest, x_all, (y_all <= 0).astype(np.float64) + ) + # Part two: the positive share, a quantile regression forest. + positive = y_all > 0 + x = x_all[positive] + y = y_all[positive] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + **zero_arrays, + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y_all)), + "n_positive_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _p_zero(self, x) -> np.ndarray: + """The probability forest's zero-year probability (mean over trees).""" + + x = np.asarray(x, dtype=np.float32) + n_trees = len(self.zero_tree_offsets) - 1 + total = np.zeros(len(x)) + for tree in range(n_trees): + start = self.zero_tree_offsets[tree] + stop = self.zero_tree_offsets[tree + 1] + node = _tree_leaves( + self.zero_node_left[start:stop].astype(np.int64), + self.zero_node_right[start:stop].astype(np.int64), + self.zero_node_feature[start:stop].astype(np.int64), + self.zero_node_threshold[start:stop], + x, + ) + total += self.zero_leaf_p[ + self.zero_node_leaf[start:stop][node].astype(np.int64) + ] + return total / n_trees + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + p_zero = np.ones(len(rows)) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + p_zero[valid] = self._p_zero(features) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "p_zero": p_zero, + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + u = ndtr(z) + p0 = unit["p_zero"] + valid = valid & (u >= p0) + v = (u[valid] - p0[valid]) / np.maximum(1.0 - p0[valid], 1e-12) + drawn[valid] = self._value(unit["leaves"][valid], v) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + pre_career = years[None, :] < start[:, None] + mask &= ~pre_career + given = np.where(mask | pre_career, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +def _forest_arrays(forest, x, y) -> dict[str, np.ndarray]: + """A fitted probability forest as arrays: nodes, and each leaf's mean y.""" + + offsets = [0] + lefts, rights, features, thresholds, leaf_index, means = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + n_leaves = int(is_leaf.sum()) + index = np.full(tree.node_count, -1, dtype=np.int64) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + total = np.bincount(local, minlength=n_leaves) + hits = np.bincount(local, weights=y, minlength=n_leaves) + means.append(hits / np.maximum(total, 1)) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + offsets.append(offsets[-1] + tree.node_count) + return { + "zero_tree_offsets": np.asarray(offsets, dtype=np.int64), + "zero_node_left": np.concatenate(lefts).astype(np.int32), + "zero_node_right": np.concatenate(rights).astype(np.int32), + "zero_node_feature": np.concatenate(features), + "zero_node_threshold": np.concatenate(thresholds), + "zero_node_leaf": np.concatenate(leaf_index).astype(np.int32), + "zero_leaf_p": np.concatenate(means).astype(np.float32), + } + + +_FOREST_ARRAYS = ( + "zero_tree_offsets", + "zero_node_left", + "zero_node_right", + "zero_node_feature", + "zero_node_threshold", + "zero_node_leaf", + "zero_leaf_p", + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + # Units inside the career; contexts see the career only. + start = career_start(np.asarray(birth_year)) + inside = unit_year >= start[rows] + rows, unit_year = rows[inside], unit_year[inside] + known = np.isfinite(shares) & ~( + np.asarray(years)[None, :] < start[:, None] + ) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + # A missing neighbour takes the other's value, as in the draw. + left = np.where(np.isnan(context.left), context.right, context.left) + right = np.where(np.isnan(context.right), context.left, context.right) + usable = np.isfinite(left) & np.isfinite(right) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum[usable]): + members = np.flatnonzero((stratum == value) & usable) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=left[keep].astype(np.float32), + bank_right=right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: Nearest-donor lists by input content, reused across draw seeds. +_NEAREST_CACHE: dict = {} +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + if len(reference) == 0: + return np.full(len(values), np.nan) + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + bank_size=_DONOR_BANK, + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:bank_size] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def _nearest(self, match, birth_year, sex, targets): + """Each target's ``k`` nearest bank rows, and its group's bank rows. + + Seed-free, so it is computed once for a matrix and reused across + draw seeds (cached by the content of its inputs). + """ + + digest = hashlib.sha256( + np.ascontiguousarray(match[targets]).tobytes() + + np.ascontiguousarray(birth_year[targets]).tobytes() + + np.ascontiguousarray(sex[targets]).tobytes() + + np.ascontiguousarray(targets).tobytes() + + str((id(self), self.k)).encode() + ).hexdigest() + if digest in _NEAREST_CACHE: + return _NEAREST_CACHE[digest] + nearest = np.full((len(targets), self.k), -1, dtype=np.int64) + group_first = np.full(len(targets), -1, dtype=np.int64) + group_size = np.zeros(len(targets), dtype=np.int64) + no_match = np.zeros(len(targets), dtype=bool) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + local = np.flatnonzero( + (sex[targets] == s) & (birth_year[targets] == b) + ) + recipients = targets[local] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + group_first[local] = donors[0] + group_size[local] = len(donors) + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + finite = np.isfinite(donor_match[:, j]) + column = np.sort(donor_match[finite, j]) + ranks_donor[:, j] = np.where( + finite, + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + order = np.argpartition(distance, k - 1, axis=1)[:, :k] + near_distance = np.take_along_axis(distance, order, 1) + ranked = np.lexsort((order, near_distance), axis=1) + order = np.take_along_axis(order, ranked, 1) + rows = local[block] + nearest[rows, :k] = donors[order] + no_match[rows] = count.max(axis=1) == 0 + result = (nearest, group_first, group_size, no_match) + if len(_NEAREST_CACHE) >= 4: + _NEAREST_CACHE.pop(next(iter(_NEAREST_CACHE))) + _NEAREST_CACHE[digest] = result + return result + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + One of the ``k`` nearest bank donors of the recipient's sex and + birth year, chosen by the seeded uniform; a recipient with no + recorded match feature takes a random donor of its group. -1 for + rows with no masked cell or no bank donor of their group. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + nearest, group_first, group_size, no_match = self._nearest( + match, birth_year, sex, targets + ) + out = np.full(len(shares), -1, dtype=np.int64) + has_group = group_size > 0 + k_available = (nearest >= 0).sum(axis=1) + pick = np.minimum( + (u[targets] * np.maximum(k_available, 1)).astype(np.int64), + np.maximum(k_available - 1, 0), + ) + chosen = nearest[np.arange(len(targets)), pick] + random_donor = group_first + np.minimum( + (u[targets] * np.maximum(group_size, 1)).astype(np.int64), + np.maximum(group_size - 1, 0), + ) + chosen = np.where(no_match, random_donor, chosen) + out[targets[has_group]] = chosen[has_group] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/tests/cohorts/test_psid2010_epuf_fill.py b/tests/cohorts/test_psid2010_epuf_fill.py new file mode 100644 index 00000000..61acea29 --- /dev/null +++ b/tests/cohorts/test_psid2010_epuf_fill.py @@ -0,0 +1,158 @@ +"""Learned EPUF fills applied to a PSID-2010-shaped cohort (synthetic).""" + +from __future__ import annotations + +from types import SimpleNamespace + +import numpy as np +import pandas as pd +import pytest + +from populace_dynamics.cohorts import psid2010_epuf_fill as fill_module +from populace_dynamics.harness import epuf_fill_gate as g + +BIRTHS = {1: (1950, "male"), 2: (1960, "female"), 3: (1975, "male")} + + +def _cohort(): + caps = fill_module._wage_bases(np.arange(1951, 2011)) + rows = [] + for pid, (birth, _) in BIRTHS.items(): + start = max(1968, birth + 22) + for year in range(start, 2011): + cap = caps[year - 1951] + if year >= 1997 and year % 2 == 1: + continue + rows.append((pid, year, 0.3 * cap * (1 + 0.01 * pid), "observed")) + for year in range(max(start, 1997), 2011): + if year % 2 == 1: + left = [r for r in rows if r[0] == pid and r[1] == year - 1] + right = [r for r in rows if r[0] == pid and r[1] == year + 1] + if left and right: + mean = (left[0][2] + right[0][2]) / 2 + else: + mean = (left or right)[0][2] + rows.append((pid, year, mean, "gap_imputed")) + careers = pd.DataFrame( + rows, columns=["person_id", "year", "earnings", "provenance"] + ).sort_values(["person_id", "year"]) + persons = pd.DataFrame( + { + "person_id": list(BIRTHS), + "birth_year": [b for b, _ in BIRTHS.values()], + "sex": [s for _, s in BIRTHS.values()], + } + ) + return SimpleNamespace(persons=persons, careers=careers) + + +class _MeanFill: + """The assembler's neighbour mean, in dollars, on SSA's wage bases.""" + + name = "neighbour_mean" + + def fill(self, shares, years, birth, sex, key, mask, seed): + caps = fill_module._wage_bases(np.asarray(years)) + dollars = shares * caps[None, :] + out = shares.copy() + for column in np.flatnonzero(mask.any(axis=0)): + left = dollars[:, column - 1] + right = dollars[:, column + 1] + mean = np.where( + np.isnan(left), + right, + np.where(np.isnan(right), left, (left + right) / 2), + ) + rows = mask[:, column] + out[rows, column] = np.minimum( + np.nan_to_num(mean[rows]) / caps[column], 1.0 + ) + return out + + +def test_current_rule_fills_reproduce_the_assembler(): + cohort = _cohort() + result = fill_module.fill_careers( + cohort, + odd_fill=_MeanFill(), + pre_fill=g.CurrentPreFill(), + seed=7100, + ) + careers = result.careers + before = cohort.careers.set_index(["person_id", "year"]) + after = careers.set_index(["person_id", "year"]) + gap = before["provenance"] == "gap_imputed" + # Neighbours below the cap: the learned path's neighbour mean (in + # dollars, capped) is the assembler's mean. + np.testing.assert_allclose( + after.loc[before.index[gap], "earnings"], + before.loc[gap, "earnings"], + rtol=1e-9, + ) + assert ( + after.loc[before.index[gap], "provenance"] + == fill_module.EPUFFillProvenance.GAP_EPUF_DRAWN.value + ).all() + observed = before["provenance"] == "observed" + pd.testing.assert_series_equal( + after.loc[before.index[observed], "earnings"], + before.loc[observed, "earnings"], + check_names=False, + ) + pre = careers[ + careers["provenance"] + == fill_module.EPUFFillProvenance.PRE_CAREER_EPUF_DONOR.value + ] + for pid, (birth, _) in BIRTHS.items(): + years = pre.loc[pre["person_id"] == pid, "year"].tolist() + assert years == list(range(1951, max(1968, birth + 22))) + assert (pre["earnings"] == 0).all() + assert result.fills == { + "odd": "neighbour_mean", + "pre": "current_pre_career_rule", + } + + +def test_fills_see_no_pre_career_or_gap_year(): + cohort = _cohort() + seen = {} + + class Spy: + name = "spy" + + def fill(self, shares, years, birth, sex, key, mask, seed): + seen["shares"] = shares.copy() + seen["mask"] = mask.copy() + out = shares.copy() + out[mask] = 0.25 + return out + + result = fill_module.fill_careers(cohort, odd_fill=Spy(), seed=1) + years = np.arange(1951, 2011) + birth = np.array([b for b, _ in BIRTHS.values()]) + pre = years[None, :] < np.maximum(1968, birth + 22)[:, None] + assert np.isnan(seen["shares"][pre]).all() + assert np.isnan(seen["shares"][seen["mask"]]).all() + assert (seen["shares"][~pre & ~seen["mask"]] <= 1).all() + drawn = result.careers[result.careers["provenance"] == "gap_epuf_drawn"] + caps = fill_module._wage_bases(drawn["year"].to_numpy()) + np.testing.assert_allclose(drawn["earnings"], 0.25 * caps) + assert len(result.content_sha256) == 64 + + +def test_an_invalid_fill_is_refused(): + class Bad: + def fill(self, shares, years, birth, sex, key, mask, seed): + out = shares.copy() + out[mask] = 1.5 + return out + + with pytest.raises(ValueError, match="invalid share"): + fill_module.fill_careers(_cohort(), odd_fill=Bad(), seed=1) + + +def test_no_fill_keeps_the_careers(): + cohort = _cohort() + result = fill_module.fill_careers(cohort, seed=1) + assert len(result.careers) == len(cohort.careers) + assert result.fills == {} diff --git a/tests/estimates/test_birth_evidence_artifact.py b/tests/estimates/test_birth_evidence_artifact.py index 27151dff..fc475dd0 100644 --- a/tests/estimates/test_birth_evidence_artifact.py +++ b/tests/estimates/test_birth_evidence_artifact.py @@ -216,6 +216,8 @@ def test_post_review_sources_are_outside_historical_reducer_identity(): Path("src/populace_dynamics/harness/epuf_run.py"), Path("src/populace_dynamics/harness/epuf_fill_gate.py"), Path("src/populace_dynamics/harness/epuf_fill_scoring.py"), + Path("src/populace_dynamics/estimates/epuf_fill.py"), + Path("src/populace_dynamics/cohorts/psid2010_epuf_fill.py"), ) assert reducer.POST_REVIEW_SHARED_SOURCE_BLOBS == { Path( @@ -473,6 +475,8 @@ def test_post_review_exclusions_are_unreachable_from_birth_evidence(): "populace_dynamics.harness.epuf_run", "populace_dynamics.harness.epuf_fill_gate", "populace_dynamics.harness.epuf_fill_scoring", + "populace_dynamics.estimates.epuf_fill", + "populace_dynamics.cohorts.psid2010_epuf_fill", } assert epuf_modules.issubset(module_paths) assert epuf_modules.isdisjoint(reachable), ( diff --git a/tests/estimates/test_epuf_fill.py b/tests/estimates/test_epuf_fill.py new file mode 100644 index 00000000..d311554f --- /dev/null +++ b/tests/estimates/test_epuf_fill.py @@ -0,0 +1,159 @@ +"""The learned EPUF career fills, on synthetic careers.""" + +from __future__ import annotations + +import hashlib + +import numpy as np +import pytest + +from populace_dynamics.estimates import epuf_fill as F +from populace_dynamics.harness import epuf_fill_gate as g + +YEARS = np.asarray(g.YEARS) + + +def _shares(seed: int, n: int = 3_000): + rng = np.random.default_rng(seed) + birth = rng.integers(1925, 1981, size=n) + sex = rng.choice([1, 2], size=n) + level = rng.normal(-1.3, 0.7, size=n) + walk = rng.normal(0, 0.25, size=(n, len(YEARS))).cumsum(axis=1) * 0.3 + shares = np.minimum(np.exp(level[:, None] + walk), 1.0) + work = rng.random((n, len(YEARS))) < 0.85 + age = YEARS[None, :] - birth[:, None] + shares = np.where(work & (age >= 15) & (age <= 85), shares, 0.0) + return np.round(shares, 6), birth, sex, np.arange(n) + 1 + + +def test_hash_uniform_is_keyed_and_order_free(): + keys = np.arange(1, 10_001) + u = F.hash_uniform("s", 7, keys, 1999) + assert ((u > 0) & (u < 1)).all() + assert abs(u.mean() - 0.5) < 0.02 + np.testing.assert_array_equal( + F.hash_uniform("s", 7, keys[::-1], 1999)[::-1], u + ) + assert not np.array_equal(u, F.hash_uniform("s", 8, keys, 1999)) + assert not np.array_equal(u, F.hash_uniform("t", 7, keys, 1999)) + assert not np.array_equal(u, F.hash_uniform("s", 7, keys, 2001)) + + +@pytest.fixture(scope="module") +def fitted(): + shares, birth, sex, key = _shares(1) + unit_years = tuple(range(1991, 2006)) + return { + "odd_forest": F.BySexFill.fit( + F.OddForestFill, + shares, + YEARS, + birth, + sex, + unit_years, + key, + n_units=40_000, + n_trees=3, + min_leaf=10, + n_jobs=1, + )[0], + "odd_knn": F.OddKnnFill.fit(shares, YEARS, birth, sex, unit_years)[0], + "pre_donor": F.PreDonorFill.fit( + shares, YEARS, birth, sex, key, k=3, bank_size=500 + )[0], + "pre_chain": F.PreChainFill.fit( + shares, YEARS, birth, sex, tuple(range(1951, 2006)) + )[0], + } + + +def _given(seed=2, n=800): + shares, birth, sex, key = _shares(seed, n) + given = np.where(g.union_mask(birth), np.nan, shares) + return given, birth, sex, key + 10_000_000 + + +@pytest.mark.parametrize( + ("name", "family"), + [ + ("odd_forest", "odd"), + ("odd_knn", "odd"), + ("pre_donor", "pre"), + ("pre_chain", "pre"), + ], +) +def test_fills_obey_the_scoring_contract(fitted, name, family): + fill = fitted[name] + given, birth, sex, key = _given() + mask = g.family_mask(family, birth) + out = fill.fill(given.copy(), YEARS, birth, sex, key, mask.copy(), 7100) + assert np.isfinite(out[mask]).all() + assert ((out[mask] >= 0) & (out[mask] <= 1)).all() + same = (out[~mask] == given[~mask]) | ( + np.isnan(out[~mask]) & np.isnan(given[~mask]) + ) + assert same.all() + again = fill.fill(given.copy(), YEARS, birth, sex, key, mask.copy(), 7100) + np.testing.assert_array_equal(np.nan_to_num(out), np.nan_to_num(again)) + other = fill.fill(given.copy(), YEARS, birth, sex, key, mask.copy(), 7101) + assert not np.array_equal(np.nan_to_num(out), np.nan_to_num(other)) + # A person's draw does not depend on the other persons' order. + order = np.random.default_rng(0).permutation(len(birth)) + permuted = fill.fill( + given[order].copy(), + YEARS, + birth[order], + sex[order], + key[order], + mask[order].copy(), + 7100, + ) + np.testing.assert_allclose( + np.nan_to_num(permuted), np.nan_to_num(out[order]) + ) + + +@pytest.mark.parametrize( + "name", ["odd_forest", "odd_knn", "pre_donor", "pre_chain"] +) +def test_artifacts_round_trip_with_reproducible_bytes(fitted, name, tmp_path): + fill = fitted[name] + blob = fill.to_bytes() + assert blob == fill.to_bytes() + path = tmp_path / f"{name}.npz" + path.write_bytes(blob) + sha = hashlib.sha256(blob).hexdigest() + loaded = F.load_fill(path, sha256=sha) + assert type(loaded) is type(fill) + assert loaded.to_bytes() == blob + with pytest.raises(ValueError, match="SHA-256"): + F.load_fill(path, sha256="0" * 64) + + +def test_donor_blocks_come_from_the_bank(fitted): + fill = fitted["pre_donor"] + given, birth, sex, key = _given(3) + mask = g.family_mask("pre", birth) + out = fill.fill(given.copy(), YEARS, birth, sex, key, mask, 7100) + donors = fill.donors(given, YEARS, birth, sex, key, mask, 7100) + for row in np.flatnonzero(donors >= 0)[:50]: + donor = donors[row] + assert fill.bank_birth_year[donor] == birth[row] + assert fill.bank_sex[donor] == sex[row] + first = int(F.block_first_year(birth[row : row + 1])[0]) + block = fill.bank_block[donor].astype(float) / 65_535 + for offset in range(F.BLOCK_WIDTH): + year = first + offset + if year in YEARS and mask[row, year - YEARS[0]]: + assert out[row, year - YEARS[0]] == pytest.approx( + block[offset] + ) + early = (YEARS < first) & mask[row] + assert (out[row, early] == 0).all() + + +def test_the_forest_copula_is_calibrated_by_band(fitted): + for part in fitted["odd_forest"].parts.values(): + rho = part.rho + assert rho.shape == (4, 6) + assert ((rho >= 0) & (rho <= 0.9)).all() From ed6d07d2bc9f6047f52b47012378f5acec277556 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 04:52:39 -0400 Subject: [PATCH 02/13] gate_epuf_fill: register the four candidates (fitted on TRAIN, pinned by SHA-256) runs/epuf_fill_candidates_v1.json names the registered fills, fitted at 9f069477 by scripts/fit_epuf_fills.py; a second fit reproduced all four byte for byte. scripts/score_epuf_fill_test.py is the one TEST scoring, refused until gates.yaml locks the gate. A DEV dry run of the registered procedure (20 seeds) predicts: odd primary adopted as an uncertified improvement (7 of 183 cells failing, at ages 22-29 mostly), pre primary certified (0 of 136). The registration document records it before TEST. Co-Authored-By: Claude Opus 5.5 --- .../gate_epuf_fill_candidates_registration.md | 139 ++ ...e_epuf_fill_dev_scores_after_round_2.jsonl | 1 + runs/epuf_fill_candidates_v1.json | 1401 +++++++++++++++++ runs/epuf_fill_candidates_v1.json.env.json | 19 + scripts/score_epuf_fill_test.py | 85 + tests/README-tiers.md | 6 +- tests/test_epuf_fill_candidates_manifest.py | 98 ++ tests/tier_counts.json | 4 +- 8 files changed, 1748 insertions(+), 5 deletions(-) create mode 100644 docs/amendments/gate_epuf_fill_candidates_registration.md create mode 100644 runs/epuf_fill_candidates_v1.json create mode 100644 runs/epuf_fill_candidates_v1.json.env.json create mode 100644 scripts/score_epuf_fill_test.py create mode 100644 tests/test_epuf_fill_candidates_manifest.py diff --git a/docs/amendments/gate_epuf_fill_candidates_registration.md b/docs/amendments/gate_epuf_fill_candidates_registration.md new file mode 100644 index 00000000..3d20a536 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidates_registration.md @@ -0,0 +1,139 @@ +# gate_epuf_fill: registered candidates + +- **Registration id**: `2026-10-03-epuf-career-fill` +- **Gate**: `gate_epuf_fill` + (`docs/amendments/gate_epuf_fill_registration_proposal.md`). A gate here is + a pass-or-fail test whose rules and thresholds are fixed and published + before anything is scored against it. +- **Stage**: candidates registered before any TEST read. TEST is scored once, + by `scripts/score_epuf_fill_test.py`, after `gates.yaml` locks the gate on + Max's ratification (decision d927). +- **Code**: `src/populace_dynamics/estimates/epuf_fill.py` at the manifest's + `code_commit`. +- **Fitted artifacts**: `runs/epuf_fill_candidates_v1.json` records each + file's SHA-256, size, parameters and diagnostics, and the library versions. + - The files are fitted on EPUF TRAIN only, by `scripts/fit_epuf_fills.py`. + - They are byte-reproducible `.npz` files, staged outside the repository + as EPUF is (`~/PolicyEngine/epuf-data/fills`). + - `epuf_fill_scoring.score_registered` loads each one through `load_fill` + and refuses other bytes. + +## The registered artifacts + +- **Manifest**: `runs/epuf_fill_candidates_v1.json`, SHA-256 `8d421536351d884184af441889333db2361fb0f50326e3a4481e6fa4f07aeecf`. +- **Fitted at**: `9f069477` on TRAIN, with the code files clean. +- **Libraries**: numpy 2.5.1, scipy 1.18.0, scikit-learn 1.9.0. +- **Reproducibility**: a second fit at the same commit reproduced all four files byte for byte. + +| Name | Role | File | SHA-256 | Bytes | +|---|---|---|---|---:| +| `odd_forest` | odd primary | `odd_forest_v1.npz` | `37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa` | 44,836,853 | +| `odd_knn` | odd alternative | `odd_knn_v1.npz` | `8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291` | 4,076,580 | +| `pre_donor` | pre primary | `pre_donor_v1.npz` | `3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7` | 31,768,107 | +| `pre_chain` | pre alternative | `pre_chain_v1.npz` | `8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724` | 236,450 | + +## The four candidates + +| Family | Role | Name | What it is | +|---|---|---|---| +| odd | primary | `odd_forest` | Two-part random-forest draw, fitted per sex | +| odd | alternative | `odd_knn` | kNN triples | +| pre | primary | `pre_donor` | Rank-kNN donor careers | +| pre | alternative | `pre_chain` | Chained one-sided draw | + +**`odd_forest`** (odd primary) is fitted per sex. +- **Part one**: a probability forest gives the chance of a zero year. +- **Part two**: a quantile regression forest gives the positive share. Its + leaves keep the true TRAIN shares, so its draws are true values. +- **Features**: the recorded shares at `t-1`, `t+1`, `t-3` and `t+3`; their + mean, geometric mean and count positive; the mean positive share and the + share of positive years at offsets 5-9; sex; age; and the year. +- **Training units**: TRAIN units inside the career, with contexts that see + the career only. +- **Copula**: a person-level Gaussian copula with correlation by sex and + age band. It is calibrated on one TRAIN person in ten, held out of the + forests, to match two- and four-year persistence between masked years. + +**`odd_knn`** (odd alternative) draws the share at `t` from one of the 10 +nearest TRAIN career units in the shares at `t-1` and `t+1`, by sex and +five-year age band. + +**`pre_donor`** (pre primary) copies a whole masked block from one of the 3 +nearest TRAIN donors of the same sex and birth year. +- **Bank**: every TRAIN donor in the `pre` universe, up to 100,000 per sex + and birth year. +- **Distance**: percentile ranks over the first five recorded career years, + plus the career's mean share and share of positive years. + +**`pre_chain`** (pre alternative) draws year `y` from the next known later +year's share, sex and age, backward from the career start. Ages 15-24 are +single years. + +Each candidate's exact fit parameters are in the manifest and in +`scripts/fit_epuf_fills.py` (`REGISTERED`). + +## DEV development, disclosed + +Candidates were developed against DEV, as the registration allows. Every DEV +score is disclosed: +- `docs/amendments/gate_epuf_fill_dev_scores_before_amendment_1.json`; +- `docs/amendments/gate_epuf_fill_dev_scores_after_amendment_1.json`; +- `docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl`, which + logs every later score through the registered scoring path. + +The candidate code behind each later score is kept in +`docs/amendments/gate_epuf_fill_candidate_code/`. + +The last DEV scores of the four registered designs, against the v3 +tolerances, are below. They are scored on the registered artifacts with 20 +draw seeds; see the dry run below. + +| Candidate | Gating cells failed on DEV | Tier on DEV (fallback current rule) | +|---|---:|---| +| `odd_forest` | 7 of 183 (worst: `odd.men.a22_29.r1`, 2.69 tolerances) | improves | +| `odd_knn` | 46 of 183 (worst: `wint`, up to 10 tolerances) | not adopted | +| `pre_donor` | 0 of 136 (worst: 0.77 tolerances) | certified | +| `pre_chain` | 50 of 136 (worst: youth `ylevel`, 20 tolerances) | not adopted | + +**The registered procedure, dry-run on DEV.** On 2026-10-04, +`epuf_fill_scoring.score_registered` ran with the DEV matrix in place of +TEST, the registered artifacts loaded by SHA-256, and all 20 draw seeds. It +is logged in `docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl`. + +| Family | Current rule failing | Primary | Alternative | Adopted | +|---|---|---|---|---| +| odd | 100 (fallback reading), 99 (two-sided) | improves, 7 of 183 failing | not adopted, 46 failing | primary, uncertified | +| pre | 131 | certified, 0 of 136 failing | not adopted, 50 failing | primary | + +The odd primary's seven failing cells on DEV: + +| Cell | Tolerances | +|---|---:| +| `odd.men.a22_29.r1` | 2.69 | +| `odd.women.a22_29.r1` | 2.02 | +| `odd.women.a22_29.zint` | 1.58 | +| `odd.men.a22_29.zint` | 1.37 | +| `odd.women.a22_74.zint` | 1.23 | +| `odd.women.a60_74.r1` | 1.17 | +| `odd.women.a22_74.r1` | 1.09 | + +**What DEV predicts, stated before TEST.** DEV and TEST are disjoint random +fifths of EPUF of nearly equal size, so their scores should agree closely. +- The pre-career primary should certify. +- The odd primary should be adopted as an uncertified improvement, unless + its young-band cells move. +- Most of the odd primary's residual misses are at ages 22-29, where the year + before the career is hidden from every fill. The oracle O1 misses there + too. + +## Procedure on TEST (after lock) + +1. Confirm that `gates.yaml` locks `gate_epuf_fill` and that the staged files + match the manifest's SHA-256. +2. Run `scripts/score_epuf_fill_test.py --manifest + runs/epuf_fill_candidates_v1.json --output runs/epuf_fill_gate_test_v1.json` + once. +3. Publish the result whether it passes or fails. Record each family's tier + and adoption under the registered rule. +4. Record the adoption in this document. A family whose primary and + alternative both fail to be adopted keeps the current rule. diff --git a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl index c9402386..7ffa31bd 100644 --- a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl +++ b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl @@ -14,3 +14,4 @@ {"utc": "2026-10-04T08:23:28+00:00", "candidate": "pre_donor_ball_k3", "family": "pre", "artifact_sha256": "c472c84325711c8856dc8067d02ef3c9c27cb8bce63befae22df2d18cab50062", "code_sha256": "4be7556f2263ef7890929ef27a92975d93938415e22153b38bf0afd01fef5983", "seeds": [7100, 7101, 7102, 7103], "n_failing": 0, "n_gating": 136, "tier": "certified", "worst": [["pre.men.b1946_1980.yzero", 0.00809038443776322, 0.011384493587046594], ["pre.women.b1946_1980.yzero", 0.005732645830089589, 0.008556191699814752], ["pre.men.b1956_1965.yzero", 0.015012948537747373, 0.022693736955041323], ["pre.women.b1956_1965.yr_cross", 0.019137592768893985, 0.03047360671843986], ["pre.women.b1946_1980.yr_cross", 0.00910100469507158, 0.015684351176846988], ["pre.men.b1946_1980.ylevel", -0.008408704694680136, 0.01520534060461241], ["pre.women.b1930_1945.pr_cross", 0.036302636511399144, 0.0677006808491353], ["pre.women.b1930_1945.pr_in", 0.03831971636456638, 0.07412154624977427]]} {"utc": "2026-10-04T08:27:20+00:00", "candidate": "odd_knn2", "family": "odd", "artifact_sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 47, "n_gating": 183, "tier": "not_adopted", "worst": [["odd.men.a22_74.wint", -0.7579019412918284, 0.07337302886151653], ["odd.women.a22_74.wint", -0.6178668454511387, 0.07133033012317215], ["odd.men.a45_59.wint", -1.0410721850282765, 0.13344110142481172], ["odd.women.a45_59.wint", -0.9735852305534904, 0.1271041647107106], ["odd.men.a30_44.wint", -0.7720545590267562, 0.12001713872675526], ["odd.women.a30_44.wint", -0.5937411523100029, 0.10093703169083498]]} {"utc": "2026-10-04T08:27:45+00:00", "candidate": "pre_chain3", "family": "pre", "artifact_sha256": "74c62142e1de4bfed543de3b1d8c88c0d70c674a8a2fa44129bfe5943d4ff0ea", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 50, "n_gating": 136, "tier": "not_adopted", "worst": [["pre.women.b1966_1980.ylevel", 0.35125489825189504, 0.017769920520911725], ["pre.men.b1966_1980.ylevel", 0.35211188653405623, 0.021382204828984532], ["pre.women.b1966_1980.yzero", 0.13965247274046733, 0.014833297517636257], ["pre.women.b1946_1980.ylevel", 0.1097696357515745, 0.015450091379218515], ["pre.men.b1930_1945.plevel", -0.18375090914832892, 0.029173903026433003], ["pre.women.b1930_1945.pzero", -0.18200182661313102, 0.02913422225552515]]} +{"utc": "2026-10-04T08:52:00+00:00", "candidate": "registered set (manifest runs/epuf_fill_candidates_v1.json)", "family": "odd+pre", "description": "DEV dry run of the registered TEST procedure: epuf_fill_scoring.score_registered with the DEV matrix, 20 draw seeds, candidates loaded by SHA-256, both readings of the current odd rule", "manifest_sha256": "8d421536351d884184af441889333db2361fb0f50326e3a4481e6fa4f07aeecf", "summary": {"odd": {"adopted": "primary", "dropped": [], "primary": {"n_failing": 7, "n_gating": 183, "tier": "improves", "worst": [[2.690259040347596, "odd.men.a22_29.r1"], [2.0188009070760624, "odd.women.a22_29.r1"], [1.5849202031321585, "odd.women.a22_29.zint"], [1.3654563419043049, "odd.men.a22_29.zint"], [1.2322030452483468, "odd.women.a22_74.zint"]]}, "alternative": {"n_failing": 46, "n_gating": 183, "tier": "not_adopted", "worst": [[10.235181548470665, "odd.men.a22_74.wint"], [8.632139516178636, "odd.women.a22_74.wint"], [7.7476496021426895, "odd.women.a45_59.wint"], [7.571934298028397, "odd.men.a45_59.wint"], [6.518179638279406, "odd.men.a30_44.wint"]]}, "current": {"fallback": {"n_failing": 100}, "two_sided": {"n_failing": 99}}}, "pre": {"adopted": "primary", "dropped": [], "primary": {"n_failing": 0, "n_gating": 136, "tier": "certified", "worst": [[0.769263808372738, "pre.men.b1946_1980.yzero"], [0.7230242162372418, "pre.women.b1956_1965.yr_cross"], [0.7182944969684824, "pre.women.b1946_1980.yzero"], [0.6574070292194412, "pre.men.b1956_1965.yzero"], [0.6068639745544633, "pre.men.b1946_1980.ylevel"]]}, "alternative": {"n_failing": 50, "n_gating": 136, "tier": "not_adopted", "worst": [[19.56156909523513, "pre.women.b1966_1980.ylevel"], [16.437839846619262, "pre.men.b1966_1980.ylevel"], [9.486645296938711, "pre.women.b1966_1980.yzero"], [6.976223346512972, "pre.women.b1946_1980.ylevel"], [6.268232573425137, "pre.men.b1930_1945.plevel"]]}, "current": {"fallback": {"n_failing": 131}}}}} diff --git a/runs/epuf_fill_candidates_v1.json b/runs/epuf_fill_candidates_v1.json new file mode 100644 index 00000000..5cfc3a61 --- /dev/null +++ b/runs/epuf_fill_candidates_v1.json @@ -0,0 +1,1401 @@ +{ + "schema": "populace_dynamics.epuf_fill_candidates.v1", + "registration_id": "2026-10-03-epuf-career-fill", + "code_commit": "9f069477bf0119df8d388354a8c69ebac948466b", + "code_files_clean": true, + "built_at_utc": "2026-10-04T08:30:59+00:00", + "part": "train", + "n_persons": 2629944, + "versions": { + "numpy": "2.5.1", + "scipy": "1.18.0", + "scikit_learn": "1.9.0" + }, + "staging": "files live outside the repository, like EPUF; refit with this script at code_commit to reproduce their bytes", + "fills": { + "odd_forest": { + "family": "odd", + "role": "primary", + "params": { + "unit_years": [ + 1991, + 2005 + ], + "n_units": 3000000, + "n_trees": 10, + "min_leaf": 15, + "max_features": 0.8, + "seed": 0 + }, + "file": "odd_forest_v1.npz", + "sha256": "37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa", + "bytes": 44836853, + "fit_seconds": 73.4, + "diagnostics": { + "1": { + "n_units": 3000000, + "n_positive_units": 1267291, + "n_trees": 10, + "min_leaf": 15, + "n_leaves": 250386, + "n_nodes": 500762, + "calibration_persons": 136391, + "rho": [ + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.1, + 0.1, + 0.15, + 0.15, + 0.45, + 0.45 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + ], + "calibration": { + "1.1.rho_0.0": [ + -0.00937054964764883, + -0.006568577649590401 + ], + "1.2.rho_0.0": [ + -0.009185985434689847, + -0.017087922487923013 + ], + "1.3.rho_0.0": [ + -0.008760608986963514, + -0.005892891185897087 + ], + "1.4.rho_0.0": [ + -0.014822290100570679, + -0.0029128598220162782 + ], + "2.1.rho_0.0": [ + "nan", + "nan" + ], + "2.2.rho_0.0": [ + "nan", + "nan" + ], + "2.3.rho_0.0": [ + "nan", + "nan" + ], + "2.4.rho_0.0": [ + "nan", + "nan" + ], + "1.1.rho_0.05": [ + -0.0025221662657621824, + -0.0025405849711389594 + ], + "1.2.rho_0.05": [ + -0.005747487224877057, + -0.014480367835756236 + ], + "1.3.rho_0.05": [ + -0.006653418258712018, + -0.003659386315042923 + ], + "1.4.rho_0.05": [ + -0.013918461926527237, + -0.0014846818917918503 + ], + "2.1.rho_0.05": [ + "nan", + "nan" + ], + "2.2.rho_0.05": [ + "nan", + "nan" + ], + "2.3.rho_0.05": [ + "nan", + "nan" + ], + "2.4.rho_0.05": [ + "nan", + "nan" + ], + "1.1.rho_0.1": [ + 0.0022622655578613537, + 0.0017571621431731188 + ], + "1.2.rho_0.1": [ + -0.003463604263099551, + -0.013817366044770019 + ], + "1.3.rho_0.1": [ + -0.0043256050642532795, + -0.0020772149123101658 + ], + "1.4.rho_0.1": [ + -0.011090372488350764, + 0.0003620037980666124 + ], + "2.1.rho_0.1": [ + "nan", + "nan" + ], + "2.2.rho_0.1": [ + "nan", + "nan" + ], + "2.3.rho_0.1": [ + "nan", + "nan" + ], + "2.4.rho_0.1": [ + "nan", + "nan" + ], + "1.1.rho_0.15": [ + 0.00735065280082603, + 0.00547377504308344 + ], + "1.2.rho_0.15": [ + -0.00043591174742074745, + -0.011407436015852035 + ], + "1.3.rho_0.15": [ + -0.001469747919156772, + -6.807326989843876e-05 + ], + "1.4.rho_0.15": [ + -0.007842751587615049, + 0.0035570177498509548 + ], + "2.1.rho_0.15": [ + "nan", + "nan" + ], + "2.2.rho_0.15": [ + "nan", + "nan" + ], + "2.3.rho_0.15": [ + "nan", + "nan" + ], + "2.4.rho_0.15": [ + "nan", + "nan" + ], + "1.1.rho_0.2": [ + 0.01235436992080352, + 0.007278338434604126 + ], + "1.2.rho_0.2": [ + 0.0018633014396086667, + -0.00951255602154033 + ], + "1.3.rho_0.2": [ + 0.0014639774440903253, + 0.0017977312065096118 + ], + "1.4.rho_0.2": [ + -0.007672387545435644, + 0.005532053798539716 + ], + "2.1.rho_0.2": [ + "nan", + "nan" + ], + "2.2.rho_0.2": [ + "nan", + "nan" + ], + "2.3.rho_0.2": [ + "nan", + "nan" + ], + "2.4.rho_0.2": [ + "nan", + "nan" + ], + "1.1.rho_0.25": [ + 0.017532837584111616, + 0.012910343871726515 + ], + "1.2.rho_0.25": [ + 0.005432755081799634, + -0.0058967638868855365 + ], + "1.3.rho_0.25": [ + 0.004461578182951564, + 0.003245424028756938 + ], + "1.4.rho_0.25": [ + -0.006896856131266005, + 0.007402313522806625 + ], + "2.1.rho_0.25": [ + "nan", + "nan" + ], + "2.2.rho_0.25": [ + "nan", + "nan" + ], + "2.3.rho_0.25": [ + "nan", + "nan" + ], + "2.4.rho_0.25": [ + "nan", + "nan" + ], + "1.1.rho_0.3": [ + 0.02283322303498425, + 0.01782293559705972 + ], + "1.2.rho_0.3": [ + 0.007917659257822063, + -0.0042736292458098735 + ], + "1.3.rho_0.3": [ + 0.005912180793823385, + 0.004708972924462929 + ], + "1.4.rho_0.3": [ + -0.0037362954092261536, + 0.012169812222417309 + ], + "2.1.rho_0.3": [ + "nan", + "nan" + ], + "2.2.rho_0.3": [ + "nan", + "nan" + ], + "2.3.rho_0.3": [ + "nan", + "nan" + ], + "2.4.rho_0.3": [ + "nan", + "nan" + ], + "1.1.rho_0.35": [ + 0.02936744538086289, + 0.02269930584128066 + ], + "1.2.rho_0.35": [ + 0.010991221409968999, + -0.0020151785137914047 + ], + "1.3.rho_0.35": [ + 0.008154499506314195, + 0.006864335349464179 + ], + "1.4.rho_0.35": [ + -0.0033667339378640193, + 0.010716886425056416 + ], + "2.1.rho_0.35": [ + "nan", + "nan" + ], + "2.2.rho_0.35": [ + "nan", + "nan" + ], + "2.3.rho_0.35": [ + "nan", + "nan" + ], + "2.4.rho_0.35": [ + "nan", + "nan" + ], + "1.1.rho_0.4": [ + 0.033991025478408377, + 0.025266554928974005 + ], + "1.2.rho_0.4": [ + 0.01395682475791915, + 0.00013629350615307345 + ], + "1.3.rho_0.4": [ + 0.009812482470721196, + 0.008655649864332982 + ], + "1.4.rho_0.4": [ + -0.0018532086431369832, + 0.010892834474684587 + ], + "2.1.rho_0.4": [ + "nan", + "nan" + ], + "2.2.rho_0.4": [ + "nan", + "nan" + ], + "2.3.rho_0.4": [ + "nan", + "nan" + ], + "2.4.rho_0.4": [ + "nan", + "nan" + ], + "1.1.rho_0.45": [ + 0.04078840124406191, + 0.03125399303118248 + ], + "1.2.rho_0.45": [ + 0.016853239128019615, + 0.0023949314237127206 + ], + "1.3.rho_0.45": [ + 0.012218818061125014, + 0.0103161648513439 + ], + "1.4.rho_0.45": [ + 0.0007757976805343736, + 0.012061337583499476 + ], + "2.1.rho_0.45": [ + "nan", + "nan" + ], + "2.2.rho_0.45": [ + "nan", + "nan" + ], + "2.3.rho_0.45": [ + "nan", + "nan" + ], + "2.4.rho_0.45": [ + "nan", + "nan" + ], + "1.1.rho_0.5": [ + 0.04627958132301435, + 0.035137264874086305 + ], + "1.2.rho_0.5": [ + 0.020285078982263505, + 0.004963715627156029 + ], + "1.3.rho_0.5": [ + 0.014599779251665335, + 0.012138390969464785 + ], + "1.4.rho_0.5": [ + 0.004685831714231314, + 0.015445592554809928 + ], + "2.1.rho_0.5": [ + "nan", + "nan" + ], + "2.2.rho_0.5": [ + "nan", + "nan" + ], + "2.3.rho_0.5": [ + "nan", + "nan" + ], + "2.4.rho_0.5": [ + "nan", + "nan" + ], + "1.1.rho_0.55": [ + 0.0531305404913871, + 0.03958250535156038 + ], + "1.2.rho_0.55": [ + 0.023484969938003974, + 0.007224393712798816 + ], + "1.3.rho_0.55": [ + 0.016946622715777737, + 0.013757911329974615 + ], + "1.4.rho_0.55": [ + 0.005386466630163178, + 0.01558864988803843 + ], + "2.1.rho_0.55": [ + "nan", + "nan" + ], + "2.2.rho_0.55": [ + "nan", + "nan" + ], + "2.3.rho_0.55": [ + "nan", + "nan" + ], + "2.4.rho_0.55": [ + "nan", + "nan" + ], + "1.1.rho_0.6": [ + 0.0594398122408063, + 0.045308024035041305 + ], + "1.2.rho_0.6": [ + 0.02665635658052512, + 0.01009428430629955 + ], + "1.3.rho_0.6": [ + 0.020127026484403232, + 0.017144987639204246 + ], + "1.4.rho_0.6": [ + 0.0098509448536922, + 0.01789310382229714 + ], + "2.1.rho_0.6": [ + "nan", + "nan" + ], + "2.2.rho_0.6": [ + "nan", + "nan" + ], + "2.3.rho_0.6": [ + "nan", + "nan" + ], + "2.4.rho_0.6": [ + "nan", + "nan" + ], + "1.1.rho_0.65": [ + 0.06636452154965067, + 0.048690911284104854 + ], + "1.2.rho_0.65": [ + 0.02988279156747209, + 0.012593270355836461 + ], + "1.3.rho_0.65": [ + 0.023092885480960224, + 0.018772455583715764 + ], + "1.4.rho_0.65": [ + 0.01241442885124, + 0.02076244248410586 + ], + "2.1.rho_0.65": [ + "nan", + "nan" + ], + "2.2.rho_0.65": [ + "nan", + "nan" + ], + "2.3.rho_0.65": [ + "nan", + "nan" + ], + "2.4.rho_0.65": [ + "nan", + "nan" + ], + "1.1.rho_0.7": [ + 0.07254218457867379, + 0.0526910466051157 + ], + "1.2.rho_0.7": [ + 0.03329166905191294, + 0.015175105205589068 + ], + "1.3.rho_0.7": [ + 0.0258155301357077, + 0.020502762171670685 + ], + "1.4.rho_0.7": [ + 0.014453411163469543, + 0.020846671921446847 + ], + "2.1.rho_0.7": [ + "nan", + "nan" + ], + "2.2.rho_0.7": [ + "nan", + "nan" + ], + "2.3.rho_0.7": [ + "nan", + "nan" + ], + "2.4.rho_0.7": [ + "nan", + "nan" + ], + "1.1.rho_0.75": [ + 0.07916776188641705, + 0.057886291657737066 + ], + "1.2.rho_0.75": [ + 0.03725049246712253, + 0.018772837363772443 + ], + "1.3.rho_0.75": [ + 0.028609776086117367, + 0.02307257411267194 + ], + "1.4.rho_0.75": [ + 0.016869649042968726, + 0.02362923713906029 + ], + "2.1.rho_0.75": [ + "nan", + "nan" + ], + "2.2.rho_0.75": [ + "nan", + "nan" + ], + "2.3.rho_0.75": [ + "nan", + "nan" + ], + "2.4.rho_0.75": [ + "nan", + "nan" + ], + "1.1.rho_0.8": [ + 0.08680076817335436, + 0.06389736793273171 + ], + "1.2.rho_0.8": [ + 0.040963145105353815, + 0.021929454410539284 + ], + "1.3.rho_0.8": [ + 0.03147254961686874, + 0.02464394115084445 + ], + "1.4.rho_0.8": [ + 0.017611836623551924, + 0.02768136631318152 + ], + "2.1.rho_0.8": [ + "nan", + "nan" + ], + "2.2.rho_0.8": [ + "nan", + "nan" + ], + "2.3.rho_0.8": [ + "nan", + "nan" + ], + "2.4.rho_0.8": [ + "nan", + "nan" + ], + "1.1.rho_0.85": [ + 0.09307651453825694, + 0.06816742141440546 + ], + "1.2.rho_0.85": [ + 0.04423261009456436, + 0.024040513696466315 + ], + "1.3.rho_0.85": [ + 0.03373239871230438, + 0.02743872467269859 + ], + "1.4.rho_0.85": [ + 0.02157870721620181, + 0.02748553411253518 + ], + "2.1.rho_0.85": [ + "nan", + "nan" + ], + "2.2.rho_0.85": [ + "nan", + "nan" + ], + "2.3.rho_0.85": [ + "nan", + "nan" + ], + "2.4.rho_0.85": [ + "nan", + "nan" + ], + "1.1.rho_0.9": [ + 0.10107252485664509, + 0.07345061064036906 + ], + "1.2.rho_0.9": [ + 0.048178209491760327, + 0.026937093526634648 + ], + "1.3.rho_0.9": [ + 0.036002361151492135, + 0.02854935592244301 + ], + "1.4.rho_0.9": [ + 0.024308070051008657, + 0.02733002869966339 + ], + "2.1.rho_0.9": [ + "nan", + "nan" + ], + "2.2.rho_0.9": [ + "nan", + "nan" + ], + "2.3.rho_0.9": [ + "nan", + "nan" + ], + "2.4.rho_0.9": [ + "nan", + "nan" + ] + } + }, + "2": { + "n_units": 3000000, + "n_positive_units": 1215539, + "n_trees": 10, + "min_leaf": 15, + "n_leaves": 246831, + "n_nodes": 493652, + "calibration_persons": 126094, + "rho": [ + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.05, + 0.05, + 0.15, + 0.25, + 0.9, + 0.9 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + ], + "calibration": { + "1.1.rho_0.0": [ + "nan", + "nan" + ], + "1.2.rho_0.0": [ + "nan", + "nan" + ], + "1.3.rho_0.0": [ + "nan", + "nan" + ], + "1.4.rho_0.0": [ + "nan", + "nan" + ], + "2.1.rho_0.0": [ + -0.0007703488441346273, + -0.0010508879964993278 + ], + "2.2.rho_0.0": [ + -0.007493032484792828, + -0.013193098454176932 + ], + "2.3.rho_0.0": [ + -0.009429974516101503, + -0.01195790961172627 + ], + "2.4.rho_0.0": [ + -0.03692428228749378, + -0.046884412622328786 + ], + "1.1.rho_0.05": [ + "nan", + "nan" + ], + "1.2.rho_0.05": [ + "nan", + "nan" + ], + "1.3.rho_0.05": [ + "nan", + "nan" + ], + "1.4.rho_0.05": [ + "nan", + "nan" + ], + "2.1.rho_0.05": [ + 0.0011501760114371873, + 0.0002049050210158887 + ], + "2.2.rho_0.05": [ + -0.005131023108985944, + -0.01216928087935032 + ], + "2.3.rho_0.05": [ + -0.008520996492143773, + -0.010608354960477073 + ], + "2.4.rho_0.05": [ + -0.03473496584300617, + -0.05402554667776105 + ], + "1.1.rho_0.1": [ + "nan", + "nan" + ], + "1.2.rho_0.1": [ + "nan", + "nan" + ], + "1.3.rho_0.1": [ + "nan", + "nan" + ], + "1.4.rho_0.1": [ + "nan", + "nan" + ], + "2.1.rho_0.1": [ + 0.006927157675038598, + 0.004290292190204048 + ], + "2.2.rho_0.1": [ + -0.001703132719541478, + -0.010456836329726604 + ], + "2.3.rho_0.1": [ + -0.007321013146343036, + -0.009035871667865014 + ], + "2.4.rho_0.1": [ + -0.030882723546502233, + -0.048033202775634276 + ], + "1.1.rho_0.15": [ + "nan", + "nan" + ], + "1.2.rho_0.15": [ + "nan", + "nan" + ], + "1.3.rho_0.15": [ + "nan", + "nan" + ], + "1.4.rho_0.15": [ + "nan", + "nan" + ], + "2.1.rho_0.15": [ + 0.011751352091050937, + 0.008076268003208154 + ], + "2.2.rho_0.15": [ + 0.0020210219373677507, + -0.0070992294258820365 + ], + "2.3.rho_0.15": [ + -0.005528714504464016, + -0.0069653177500850205 + ], + "2.4.rho_0.15": [ + -0.029923398294732673, + -0.05002522943078225 + ], + "1.1.rho_0.2": [ + "nan", + "nan" + ], + "1.2.rho_0.2": [ + "nan", + "nan" + ], + "1.3.rho_0.2": [ + "nan", + "nan" + ], + "1.4.rho_0.2": [ + "nan", + "nan" + ], + "2.1.rho_0.2": [ + 0.016343948418381715, + 0.011187555046907938 + ], + "2.2.rho_0.2": [ + 0.0033473023560638415, + -0.0067290354973356115 + ], + "2.3.rho_0.2": [ + -0.002877760745891411, + -0.0033953050183980205 + ], + "2.4.rho_0.2": [ + -0.025096610145121545, + -0.05020372098724413 + ], + "1.1.rho_0.25": [ + "nan", + "nan" + ], + "1.2.rho_0.25": [ + "nan", + "nan" + ], + "1.3.rho_0.25": [ + "nan", + "nan" + ], + "1.4.rho_0.25": [ + "nan", + "nan" + ], + "2.1.rho_0.25": [ + 0.021425408842097315, + 0.01429463213074944 + ], + "2.2.rho_0.25": [ + 0.0065957040671158484, + -0.004705541238731792 + ], + "2.3.rho_0.25": [ + -1.7567476131130633e-05, + -0.0010814093051690898 + ], + "2.4.rho_0.25": [ + -0.0212615966241938, + -0.04842399617411386 + ], + "1.1.rho_0.3": [ + "nan", + "nan" + ], + "1.2.rho_0.3": [ + "nan", + "nan" + ], + "1.3.rho_0.3": [ + "nan", + "nan" + ], + "1.4.rho_0.3": [ + "nan", + "nan" + ], + "2.1.rho_0.3": [ + 0.02693682357503624, + 0.016009169813349877 + ], + "2.2.rho_0.3": [ + 0.009021593962986185, + -0.001825312400977941 + ], + "2.3.rho_0.3": [ + 0.002480943042680095, + -0.0003852668318783392 + ], + "2.4.rho_0.3": [ + -0.020630295377933372, + -0.05211604831280381 + ], + "1.1.rho_0.35": [ + "nan", + "nan" + ], + "1.2.rho_0.35": [ + "nan", + "nan" + ], + "1.3.rho_0.35": [ + "nan", + "nan" + ], + "1.4.rho_0.35": [ + "nan", + "nan" + ], + "2.1.rho_0.35": [ + 0.03296557024474411, + 0.021046225240840877 + ], + "2.2.rho_0.35": [ + 0.012335421323450668, + 0.0016580900689775468 + ], + "2.3.rho_0.35": [ + 0.004414993633125364, + 0.0012509622619388816 + ], + "2.4.rho_0.35": [ + -0.017200316022581097, + -0.04689517410403199 + ], + "1.1.rho_0.4": [ + "nan", + "nan" + ], + "1.2.rho_0.4": [ + "nan", + "nan" + ], + "1.3.rho_0.4": [ + "nan", + "nan" + ], + "1.4.rho_0.4": [ + "nan", + "nan" + ], + "2.1.rho_0.4": [ + 0.03881518696695174, + 0.02617586875204103 + ], + "2.2.rho_0.4": [ + 0.014916453205874203, + 0.0031898299704220534 + ], + "2.3.rho_0.4": [ + 0.006149846624828648, + 0.0020327028751055964 + ], + "2.4.rho_0.4": [ + -0.014400133887248034, + -0.04119061699894566 + ], + "1.1.rho_0.45": [ + "nan", + "nan" + ], + "1.2.rho_0.45": [ + "nan", + "nan" + ], + "1.3.rho_0.45": [ + "nan", + "nan" + ], + "1.4.rho_0.45": [ + "nan", + "nan" + ], + "2.1.rho_0.45": [ + 0.04408209328648782, + 0.03100375174528891 + ], + "2.2.rho_0.45": [ + 0.018276462063509857, + 0.0054696188221817765 + ], + "2.3.rho_0.45": [ + 0.0076902880826571485, + 0.00218766241834345 + ], + "2.4.rho_0.45": [ + -0.014908359815873351, + -0.0393019026587349 + ], + "1.1.rho_0.5": [ + "nan", + "nan" + ], + "1.2.rho_0.5": [ + "nan", + "nan" + ], + "1.3.rho_0.5": [ + "nan", + "nan" + ], + "1.4.rho_0.5": [ + "nan", + "nan" + ], + "2.1.rho_0.5": [ + 0.04970088094513325, + 0.034854865497334075 + ], + "2.2.rho_0.5": [ + 0.021724257795143087, + 0.008193822209299872 + ], + "2.3.rho_0.5": [ + 0.009745593525695817, + 0.003562541404541153 + ], + "2.4.rho_0.5": [ + -0.014383130743268246, + -0.04271338044982742 + ], + "1.1.rho_0.55": [ + "nan", + "nan" + ], + "1.2.rho_0.55": [ + "nan", + "nan" + ], + "1.3.rho_0.55": [ + "nan", + "nan" + ], + "1.4.rho_0.55": [ + "nan", + "nan" + ], + "2.1.rho_0.55": [ + 0.05502109102831798, + 0.03890967682059354 + ], + "2.2.rho_0.55": [ + 0.024839414066947896, + 0.010572170562098249 + ], + "2.3.rho_0.55": [ + 0.011772980929248611, + 0.003908378086523001 + ], + "2.4.rho_0.55": [ + -0.013131700928571188, + -0.04285851871646029 + ], + "1.1.rho_0.6": [ + "nan", + "nan" + ], + "1.2.rho_0.6": [ + "nan", + "nan" + ], + "1.3.rho_0.6": [ + "nan", + "nan" + ], + "1.4.rho_0.6": [ + "nan", + "nan" + ], + "2.1.rho_0.6": [ + 0.06088619981198029, + 0.0420376792891467 + ], + "2.2.rho_0.6": [ + 0.028468055929460223, + 0.012861481376605255 + ], + "2.3.rho_0.6": [ + 0.014264953161897465, + 0.006716731791151176 + ], + "2.4.rho_0.6": [ + -0.01411854107127919, + -0.04322951820182941 + ], + "1.1.rho_0.65": [ + "nan", + "nan" + ], + "1.2.rho_0.65": [ + "nan", + "nan" + ], + "1.3.rho_0.65": [ + "nan", + "nan" + ], + "1.4.rho_0.65": [ + "nan", + "nan" + ], + "2.1.rho_0.65": [ + 0.06682408268645768, + 0.04478633762350159 + ], + "2.2.rho_0.65": [ + 0.03150037859622068, + 0.014635098136048796 + ], + "2.3.rho_0.65": [ + 0.016061091785444348, + 0.008363861184702004 + ], + "2.4.rho_0.65": [ + -0.013183126314957772, + -0.04726560797289203 + ], + "1.1.rho_0.7": [ + "nan", + "nan" + ], + "1.2.rho_0.7": [ + "nan", + "nan" + ], + "1.3.rho_0.7": [ + "nan", + "nan" + ], + "1.4.rho_0.7": [ + "nan", + "nan" + ], + "2.1.rho_0.7": [ + 0.07410396740658753, + 0.05061327862615472 + ], + "2.2.rho_0.7": [ + 0.03457230831577063, + 0.01752794757556897 + ], + "2.3.rho_0.7": [ + 0.018780887727013362, + 0.009609081527960028 + ], + "2.4.rho_0.7": [ + -0.011927889063067187, + -0.04575175956846467 + ], + "1.1.rho_0.75": [ + "nan", + "nan" + ], + "1.2.rho_0.75": [ + "nan", + "nan" + ], + "1.3.rho_0.75": [ + "nan", + "nan" + ], + "1.4.rho_0.75": [ + "nan", + "nan" + ], + "2.1.rho_0.75": [ + 0.08092707240167418, + 0.05483242561629709 + ], + "2.2.rho_0.75": [ + 0.03658199157962372, + 0.018699837196579194 + ], + "2.3.rho_0.75": [ + 0.021946755597646472, + 0.011296140023272394 + ], + "2.4.rho_0.75": [ + -0.009788245012555041, + -0.045540955638491365 + ], + "1.1.rho_0.8": [ + "nan", + "nan" + ], + "1.2.rho_0.8": [ + "nan", + "nan" + ], + "1.3.rho_0.8": [ + "nan", + "nan" + ], + "1.4.rho_0.8": [ + "nan", + "nan" + ], + "2.1.rho_0.8": [ + 0.08743759706500098, + 0.060096830084552244 + ], + "2.2.rho_0.8": [ + 0.038972747910327676, + 0.01962918025075 + ], + "2.3.rho_0.8": [ + 0.02443160357398244, + 0.011703961899924842 + ], + "2.4.rho_0.8": [ + -0.008951498727275298, + -0.042132069352851076 + ], + "1.1.rho_0.85": [ + "nan", + "nan" + ], + "1.2.rho_0.85": [ + "nan", + "nan" + ], + "1.3.rho_0.85": [ + "nan", + "nan" + ], + "1.4.rho_0.85": [ + "nan", + "nan" + ], + "2.1.rho_0.85": [ + 0.09255511715216769, + 0.06287928620893612 + ], + "2.2.rho_0.85": [ + 0.04200109430794119, + 0.021762940153467025 + ], + "2.3.rho_0.85": [ + 0.027419910814330595, + 0.013140726589435658 + ], + "2.4.rho_0.85": [ + -0.0057698938150033685, + -0.03462993739531095 + ], + "1.1.rho_0.9": [ + "nan", + "nan" + ], + "1.2.rho_0.9": [ + "nan", + "nan" + ], + "1.3.rho_0.9": [ + "nan", + "nan" + ], + "1.4.rho_0.9": [ + "nan", + "nan" + ], + "2.1.rho_0.9": [ + 0.09982884654946289, + 0.06787532650906625 + ], + "2.2.rho_0.9": [ + 0.04549333428061564, + 0.02470270553781595 + ], + "2.3.rho_0.9": [ + 0.03002262611212725, + 0.014889240292517925 + ], + "2.4.rho_0.9": [ + -0.0024130894891942756, + -0.034578411611535076 + ] + } + } + } + }, + "odd_knn": { + "family": "odd", + "role": "alternative", + "params": { + "unit_years": [ + 1991, + 2005 + ], + "k": 10, + "seed": 0 + }, + "file": "odd_knn_v1.npz", + "sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", + "bytes": 4076580, + "fit_seconds": 6.1, + "diagnostics": { + "n_units": 27840831, + "bank": 1147199 + } + }, + "pre_chain": { + "family": "pre", + "role": "alternative", + "params": { + "unit_years": [ + 1951, + 2005 + ] + }, + "file": "pre_chain_v1.npz", + "sha256": "8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724", + "bytes": 236450, + "fit_seconds": 74.8, + "diagnostics": { + "n_units": 144646920 + } + }, + "pre_donor": { + "family": "pre", + "role": "primary", + "params": { + "k": 3, + "bank_size": 100000, + "birth_years": [ + 1905, + 1985 + ] + }, + "file": "pre_donor_v1.npz", + "sha256": "3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7", + "bytes": 31768107, + "fit_seconds": 2.8, + "diagnostics": { + "bank": 1518845 + } + } + }, + "elapsed_seconds": 188.1 +} diff --git a/runs/epuf_fill_candidates_v1.json.env.json b/runs/epuf_fill_candidates_v1.json.env.json new file mode 100644 index 00000000..652146fc --- /dev/null +++ b/runs/epuf_fill_candidates_v1.json.env.json @@ -0,0 +1,19 @@ +{ + "environment": { + "python": "3.14.4", + "numpy": "2.5.1", + "pandas": "3.0.3", + "sklearn": "1.9.0", + "scipy": "1.18.0", + "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O", + "fitting_stack": { + "populace_fit": "absent", + "populace_frame": "absent" + } + }, + "contract": { + "blob_sha": "b0c39af1e13a705f90b85d3e6b9a91e1d3c5485c", + "head_sha": "9f069477bf0119df8d388354a8c69ebac948466b", + "path": "gates.yaml" + } +} diff --git a/scripts/score_epuf_fill_test.py b/scripts/score_epuf_fill_test.py new file mode 100644 index 00000000..9cce7eec --- /dev/null +++ b/scripts/score_epuf_fill_test.py @@ -0,0 +1,85 @@ +"""gate_epuf_fill's one TEST scoring of the registered candidates. + +Runs only after ``gates.yaml`` locks the gate: it reads TEST through +``epuf_fill_gate.test_part`` (via ``epuf_fill_scoring.score_registered``), +which refuses otherwise. It loads the registered candidates named in the +manifest (``runs/epuf_fill_candidates_v1.json``) by their SHA-256 from the +staged fills folder, scores each family's current rule, primary and +alternative over the registered draw seeds against the registered v3 +floors, and writes the result whether the candidates pass or fail. It +refuses to overwrite an existing result. Usage:: + + python scripts/score_epuf_fill_test.py \ + --manifest runs/epuf_fill_candidates_v1.json \ + --output runs/epuf_fill_gate_test_v1.json +""" + +from __future__ import annotations + +import argparse +import datetime as dt +import hashlib +import json +import os +import subprocess +import time +from pathlib import Path + +from populace_dynamics.artifacts import write_new +from populace_dynamics.harness import epuf_fill_scoring as scoring + +ROOT = Path(__file__).resolve().parents[1] +DEFAULT_DIR = Path("~/PolicyEngine/epuf-data/fills").expanduser() + + +def candidates_from(manifest: dict, fills_dir: Path) -> dict: + """``{family: {role: (path, sha256)}}`` from a candidate manifest.""" + + spec: dict[str, dict[str, tuple[Path, str]]] = {} + for record in manifest["fills"].values(): + spec.setdefault(record["family"], {})[record["role"]] = ( + fills_dir / record["file"], + record["sha256"], + ) + return spec + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--manifest", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument( + "--fills-dir", + type=Path, + default=Path( + os.environ.get("POPULACE_DYNAMICS_EPUF_FILLS_DIR", DEFAULT_DIR) + ), + ) + args = parser.parse_args() + if args.output.exists(): + raise FileExistsError(f"{args.output} exists; TEST is scored once") + started = time.time() + manifest_bytes = args.manifest.read_bytes() + manifest = json.loads(manifest_bytes) + record = scoring.score_registered( + candidates_from(manifest, args.fills_dir) + ) + document = { + "schema": "populace_dynamics.epuf_fill_gate_test.v1", + "code_commit": subprocess.run( + ["git", "-C", str(ROOT), "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + ).stdout.strip(), + "scored_at_utc": dt.datetime.now(dt.UTC).isoformat(timespec="seconds"), + "manifest": str(args.manifest), + "manifest_sha256": hashlib.sha256(manifest_bytes).hexdigest(), + **record, + "elapsed_seconds": round(time.time() - started, 1), + } + write_new(args.output, document, sidecar=True) + + +if __name__ == "__main__": + main() diff --git a/tests/README-tiers.md b/tests/README-tiers.md index 40463151..13db9c18 100644 --- a/tests/README-tiers.md +++ b/tests/README-tiers.md @@ -39,9 +39,9 @@ pytest --collect-only -q -m oracle_policyengine | tail -1 | Tier | Tests at HEAD | |---|---:| -| `unit` | 6,005 | -| `artifact` | 3,388 | +| `unit` | 6,020 | +| `artifact` | 3,395 | | `integration_psid` | 1,341 | | `reproduction_legacy` | 520 | | `oracle_policyengine` | 220 | -| **Total** | **11,474** | +| **Total** | **11,496** | diff --git a/tests/test_epuf_fill_candidates_manifest.py b/tests/test_epuf_fill_candidates_manifest.py new file mode 100644 index 00000000..1046e41f --- /dev/null +++ b/tests/test_epuf_fill_candidates_manifest.py @@ -0,0 +1,98 @@ +"""The registered candidate manifest of gate_epuf_fill. + +``runs/epuf_fill_candidates_v1.json`` names the four registered fills by +SHA-256. Their bytes live outside the repository (``~/PolicyEngine/ +epuf-data/fills``), so the hash checks of the staged files skip where they +are not staged. +""" + +from __future__ import annotations + +import hashlib +import importlib.util +import json +import os +import subprocess +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +MANIFEST = ROOT / "runs" / "epuf_fill_candidates_v1.json" +FILLS_DIR = Path( + os.environ.get( + "POPULACE_DYNAMICS_EPUF_FILLS_DIR", "~/PolicyEngine/epuf-data/fills" + ) +).expanduser() + + +@pytest.fixture(scope="module") +def manifest(): + return json.loads(MANIFEST.read_text()) + + +def _script(): + spec = importlib.util.spec_from_file_location( + "fit_epuf_fills", ROOT / "scripts" / "fit_epuf_fills.py" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_the_manifest_registers_one_primary_and_one_alternative_per_family( + manifest, +): + assert manifest["schema"] == "populace_dynamics.epuf_fill_candidates.v1" + assert manifest["registration_id"] == "2026-10-03-epuf-career-fill" + assert manifest["part"] == "train" + assert manifest["code_files_clean"] is True + roles = sorted( + (record["family"], record["role"]) + for record in manifest["fills"].values() + ) + assert roles == [ + ("odd", "alternative"), + ("odd", "primary"), + ("pre", "alternative"), + ("pre", "primary"), + ] + for record in manifest["fills"].values(): + assert len(record["sha256"]) == 64 and record["bytes"] > 0 + + +def test_the_manifest_parameters_are_the_scripts(manifest): + registered = _script().REGISTERED + assert set(registered) == set(manifest["fills"]) + for name, record in manifest["fills"].items(): + assert record["params"] == registered[name]["params"] + assert record["family"] == registered[name]["family"] + assert record["role"] == registered[name]["role"] + + +def test_the_fill_code_is_unchanged_since_the_fit(manifest): + commit = manifest["code_commit"] + probe = subprocess.run( + ["git", "-C", str(ROOT), "cat-file", "-e", f"{commit}^{{commit}}"], + capture_output=True, + ) + if probe.returncode != 0: + pytest.skip("the fit's commit is not in this clone") + for path in _script().CODE_FILES: + built = subprocess.run( + ["git", "-C", str(ROOT), "show", f"{commit}:{path}"], + capture_output=True, + check=True, + ).stdout + assert built == (ROOT / path).read_bytes(), path + + +@pytest.mark.parametrize( + "name", ["odd_forest", "odd_knn", "pre_donor", "pre_chain"] +) +def test_staged_fills_hash_to_the_manifest(manifest, name): + record = manifest["fills"][name] + path = FILLS_DIR / record["file"] + if not path.is_file(): + pytest.skip(f"{path} is not staged") + assert hashlib.sha256(path.read_bytes()).hexdigest() == record["sha256"] diff --git a/tests/tier_counts.json b/tests/tier_counts.json index 6ac0d8a3..8067fc14 100644 --- a/tests/tier_counts.json +++ b/tests/tier_counts.json @@ -1,8 +1,8 @@ { "schema_version": 1, "counts": { - "unit": 6005, - "artifact": 3388, + "unit": 6020, + "artifact": 3395, "integration_psid": 1341, "reproduction_legacy": 520, "oracle_policyengine": 220 From ef4052250e783165c70d10e7b920cb1c3bf0408b Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 05:07:21 -0400 Subject: [PATCH 03/13] EPUF career fills: code review of #516 (REQUEST CHANGES) and its fixes fill_careers fills only the gap years the gate scored (1997-2005) by default and leaves a gap year with no visible neighbour at the assembler's value; the donor cache is keyed by the bank's content; persons of uncoded sex take the routed part's copula; a forest leaf can never be empty; the fit script refuses an existing manifest before fitting and never replaces a staged file with other bytes, and binds the gate files it depends on; the score script pins the registered manifest, records whether its code was clean, leaves a started marker and publishes no local paths. The manifest is refitted next at this commit. Co-Authored-By: Claude Opus 5.5 --- scripts/fit_epuf_fills.py | 27 ++++++- scripts/score_epuf_fill_test.py | 48 +++++++++++- .../cohorts/psid2010_epuf_fill.py | 41 ++++++++-- src/populace_dynamics/estimates/epuf_fill.py | 76 ++++++++++++++----- tests/cohorts/test_psid2010_epuf_fill.py | 68 ++++++++++++++++- tests/estimates/test_epuf_fill.py | 41 ++++++++++ tests/test_epuf_fill_candidates_manifest.py | 7 ++ 7 files changed, 276 insertions(+), 32 deletions(-) diff --git a/scripts/fit_epuf_fills.py b/scripts/fit_epuf_fills.py index b683459c..982fa3da 100644 --- a/scripts/fit_epuf_fills.py +++ b/scripts/fit_epuf_fills.py @@ -26,8 +26,10 @@ import datetime as dt import hashlib import os +import platform import subprocess import time +import zlib from pathlib import Path import numpy as np @@ -42,6 +44,8 @@ DEFAULT_DIR = Path("~/PolicyEngine/epuf-data/fills").expanduser() CODE_FILES = ( "src/populace_dynamics/estimates/epuf_fill.py", + "src/populace_dynamics/harness/epuf_fill_gate.py", + "src/populace_dynamics/harness/epuf_operator.py", "scripts/fit_epuf_fills.py", ) ODD_UNIT_YEARS = tuple(range(1991, 2006)) @@ -162,6 +166,11 @@ def main() -> None: ) parser.add_argument("--only", nargs="*", default=sorted(REGISTERED)) args = parser.parse_args() + if args.manifest.exists(): + raise FileExistsError( + f"{args.manifest} exists; a registered manifest is never refitted " + "in place" + ) started = time.time() built_at = dt.datetime.now(dt.UTC).isoformat(timespec="seconds") matrix = g.epuf_matrix(g.TRAIN) @@ -176,7 +185,15 @@ def main() -> None: ) blob = fill.to_bytes() path = args.out_dir / f"{name}_v1.npz" - path.write_bytes(blob) + if path.exists(): + # A staged file is never replaced: a refit must reproduce it. + if path.read_bytes() != blob: + raise FileExistsError( + f"{path} exists with other bytes; refusing to replace it" + ) + else: + with path.open("xb") as handle: + handle.write(blob) fills[name] = { **REGISTERED[name], "file": path.name, @@ -198,9 +215,13 @@ def main() -> None: "numpy": np.__version__, "scipy": scipy.__version__, "scikit_learn": sklearn.__version__, + "python": platform.python_version(), + "zlib_runtime": zlib.ZLIB_RUNTIME_VERSION, + "platform": platform.platform(), }, - "staging": "files live outside the repository, like EPUF; refit " - "with this script at code_commit to reproduce their bytes", + "staging": "files live outside the repository, like EPUF; a refit " + "with this script at code_commit, on the same library, zlib and " + "platform versions, reproduces their bytes", "fills": fills, "elapsed_seconds": round(time.time() - started, 1), } diff --git a/scripts/score_epuf_fill_test.py b/scripts/score_epuf_fill_test.py index 9cce7eec..3d595133 100644 --- a/scripts/score_epuf_fill_test.py +++ b/scripts/score_epuf_fill_test.py @@ -30,6 +30,16 @@ ROOT = Path(__file__).resolve().parents[1] DEFAULT_DIR = Path("~/PolicyEngine/epuf-data/fills").expanduser() +#: The registered manifest; any other manifest is refused. +REGISTERED_MANIFEST = "runs/epuf_fill_candidates_v1.json" +REGISTERED_MANIFEST_SHA256 = "PENDING_REFIT" +#: Files whose state the record reports. +CODE_FILES = ( + "src/populace_dynamics/harness/epuf_fill_gate.py", + "src/populace_dynamics/harness/epuf_fill_scoring.py", + "src/populace_dynamics/estimates/epuf_fill.py", + "scripts/score_epuf_fill_test.py", +) def candidates_from(manifest: dict, fills_dir: Path) -> dict: @@ -58,12 +68,43 @@ def main() -> None: args = parser.parse_args() if args.output.exists(): raise FileExistsError(f"{args.output} exists; TEST is scored once") - started = time.time() manifest_bytes = args.manifest.read_bytes() + manifest_sha256 = hashlib.sha256(manifest_bytes).hexdigest() + if manifest_sha256 != REGISTERED_MANIFEST_SHA256: + raise ValueError( + f"{args.manifest} has SHA-256 {manifest_sha256}, not the " + f"registered {REGISTERED_MANIFEST_SHA256}" + ) + started = time.time() + clean = ( + subprocess.run( + ["git", "-C", str(ROOT), "diff", "--quiet", "HEAD", "--"] + + list(CODE_FILES), + check=False, + ).returncode + == 0 + ) + # A started marker, so a run that fails after reading TEST leaves a + # trace of the read. + marker = Path(f"{args.output}.started.json") + with marker.open("x") as handle: + json.dump( + { + "started_at_utc": dt.datetime.now(dt.UTC).isoformat( + timespec="seconds" + ), + "manifest_sha256": manifest_sha256, + }, + handle, + ) manifest = json.loads(manifest_bytes) record = scoring.score_registered( candidates_from(manifest, args.fills_dir) ) + # Published paths are the registered file names, not local folders. + for family in record["candidates"].values(): + for role in family.values(): + role["path"] = Path(role["path"]).name document = { "schema": "populace_dynamics.epuf_fill_gate_test.v1", "code_commit": subprocess.run( @@ -73,8 +114,9 @@ def main() -> None: text=True, ).stdout.strip(), "scored_at_utc": dt.datetime.now(dt.UTC).isoformat(timespec="seconds"), - "manifest": str(args.manifest), - "manifest_sha256": hashlib.sha256(manifest_bytes).hexdigest(), + "code_files_clean": clean, + "manifest": REGISTERED_MANIFEST, + "manifest_sha256": manifest_sha256, **record, "elapsed_seconds": round(time.time() - started, 1), } diff --git a/src/populace_dynamics/cohorts/psid2010_epuf_fill.py b/src/populace_dynamics/cohorts/psid2010_epuf_fill.py index 8a115945..4937e672 100644 --- a/src/populace_dynamics/cohorts/psid2010_epuf_fill.py +++ b/src/populace_dynamics/cohorts/psid2010_epuf_fill.py @@ -8,8 +8,16 @@ Earnings Public-Use File and registered by ``gate_epuf_fill`` (``docs/amendments/gate_epuf_fill_registration_proposal.md``): -- every ``gap_imputed`` year becomes the odd fill's draw, with provenance +- every ``gap_imputed`` year in ``odd_years`` with a neighbour the fill + can see becomes the odd fill's draw, with provenance :attr:`EPUFFillProvenance.GAP_EPUF_DRAWN`; + - by default ``odd_years`` is the years the gate scored (1997-2005); + - the PSID's later gap years (2007-2011 and the 2013 seam) keep the + assembler's value unless the caller passes ``odd_years=None``, which + fills every gap year and is an extrapolation the gate does not certify; + - a gap year with no visible neighbour keeps the assembler's value, which + used a neighbour the fill does not see (an observed pre-career year, or + the 2014 boundary year); - every year from 1951 before the career start becomes the pre-career fill's draw, with provenance :attr:`EPUFFillProvenance.PRE_CAREER_EPUF_DONOR`. @@ -78,25 +86,32 @@ def _content_sha256(careers: pd.DataFrame) -> str: ).hexdigest() +#: The gap years the gate scored (``epuf_fill_gate.MASKED_ODD_YEARS``). +SCORED_ODD_YEARS: tuple[int, ...] = (1997, 1999, 2001, 2003, 2005) + + def fill_careers( cohort: Any, *, odd_fill: Any | None = None, pre_fill: Any | None = None, seed: int, - start_year: int | None = None, + last_year: int | None = None, + odd_years: tuple[int, ...] | None = SCORED_ODD_YEARS, ) -> FilledCareers: """Replace the assembler's fill rules with learned fills. ``cohort`` needs ``persons`` (``person_id``, ``birth_year``, ``sex`` as ``"male"`` / ``"female"``) and ``careers`` (``person_id``, ``year``, ``earnings``, ``provenance``). Either fill may be None, which keeps that - rule. ``start_year`` defaults to the latest career year. + rule. ``last_year`` is the last career year read (default: the latest + in ``careers``). ``odd_years`` limits the odd fill to those gap years; + None fills every gap year. """ persons = cohort.persons[["person_id", "birth_year", "sex"]].copy() careers = cohort.careers.copy() - last = int(careers["year"].max()) if start_year is None else start_year + last = int(careers["year"].max()) if last_year is None else last_year years = np.arange(FIRST_YEAR, last + 1) caps = _wage_bases(years) person_ids = persons["person_id"].to_numpy(dtype=np.int64) @@ -122,10 +137,24 @@ def fill_careers( np.maximum(earnings[observed], 0.0) / caps[columns[observed]], 1.0 ) gap = provenance == "gap_imputed" - odd_mask[rows[gap], columns[gap]] = True + if odd_years is not None: + gap &= np.isin(years[columns], np.asarray(odd_years)) start = np.maximum(1968, birth + 22) pre_mask = years[None, :] < start[:, None] - given = np.where(odd_mask | pre_mask, np.nan, shares) + # Every gap year is unknown to the fills, filled or not. + all_gaps = np.zeros_like(odd_mask) + is_gap = provenance == "gap_imputed" + all_gaps[rows[is_gap], columns[is_gap]] = True + given = np.where(all_gaps | pre_mask, np.nan, shares) + # A gap year is filled only if a neighbour is visible to the fill. + left = np.full(len(rows), np.nan) + right = np.full(len(rows), np.nan) + inner = columns > 0 + left[inner] = given[rows[inner], columns[inner] - 1] + outer = columns + 1 < len(years) + right[outer] = given[rows[outer], columns[outer] + 1] + gap &= np.isfinite(left) | np.isfinite(right) + odd_mask[rows[gap], columns[gap]] = True def drawn(fill, mask): out = np.asarray( diff --git a/src/populace_dynamics/estimates/epuf_fill.py b/src/populace_dynamics/estimates/epuf_fill.py index 9b8a6861..807932df 100644 --- a/src/populace_dynamics/estimates/epuf_fill.py +++ b/src/populace_dynamics/estimates/epuf_fill.py @@ -534,6 +534,11 @@ def fit( local = index[nodes] - leaf_count order = np.lexsort((stored, local)) counts = np.bincount(local, minlength=n_leaves) + if (counts == 0).any(): + raise ValueError( + "a forest leaf holds no TRAIN unit under the stored " + "traversal" + ) leaf_offsets.extend( (leaf_offsets[-1] + np.cumsum(counts)).tolist() ) @@ -1058,8 +1063,10 @@ def _first_recorded( ) -#: Nearest-donor lists by input content, reused across draw seeds. +#: Nearest-donor lists by input and bank content, reused across draw seeds. _NEAREST_CACHE: dict = {} +#: Bank digests by fill (the fill is held so its id cannot be reused). +_BANK_DIGESTS: dict = {} #: The match vector: the first MATCH_YEARS shares from the career start, #: then the mean share and the share of positive years over every known #: career year. @@ -1067,6 +1074,8 @@ def _first_recorded( #: Odd years the PSID never records (1997 on); hidden when a bank's match #: vectors are built, so they are built as a recipient's are. _UNRECORDED_ODD_FROM = 1997 +#: EPUF's last year, the last year a donor bank records. +_BANK_LAST_YEAR = 2006 def match_vector( @@ -1077,7 +1086,11 @@ def match_vector( shares = np.asarray(shares, dtype=np.float64) years = np.asarray(years, dtype=np.int64) first = _first_recorded(shares, years, birth_year) - career = years[None, :] >= career_start(birth_year)[:, None] + # The career summaries cover the bank's years (through 2006) only, so a + # PSID recipient's later years do not enter them. + career = (years[None, :] >= career_start(birth_year)[:, None]) & ( + years[None, :] <= _BANK_LAST_YEAR + ) known = career & np.isfinite(shares) count = known.sum(axis=1) values = np.where(known, shares, 0.0) @@ -1121,18 +1134,23 @@ def block_first_year(birth_year: np.ndarray) -> np.ndarray: class PreDonorFill: """Whole pre-career blocks copied from rank-matched TRAIN donors. - Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with - a positive share from their career start through 2006, chosen by the - lowest hash of their person id) holds each donor's shares in the years - from :func:`block_first_year` to the year before the career start (at - most the 17 years 1951-1967; stored as shares times 65,535, rounded), - and their shares in the first five years from the career start. A - recipient's match vector is its percentile mid-rank, within the bank, - in each of those five years it has recorded; distance is Euclidean over - the recorded years, scaled by five over their number. One of the ``k`` - nearest donors is chosen by the seeded uniform and its block copied; - masked years before :func:`block_first_year` are zero. A recipient with - no recorded match year takes a donor chosen at random from the bank. + Per sex and birth year, a bank of up to ``bank_size`` TRAIN donors + (those with a positive share from their career start through 2006, + chosen by the lowest hash of their person id; the registered fill keeps + them all) holds each donor's shares in the years from + :func:`block_first_year` to the year before the career start (at most + the 17 years 1951-1967, stored as shares times 65,535, rounded) and the + donor's match vector (:func:`match_vector`). The match vector has seven + features: the shares in the first five years from the career start, + and the mean share and share of positive years over the known career + years through 2006. + + A recipient's features are its percentile mid-ranks within the bank in + each feature it has. Distance is Euclidean over the features both have, + scaled by seven over their number. One of the ``k`` nearest donors is + chosen by the seeded uniform and its block copied; masked years before + :func:`block_first_year` are zero. A recipient with no feature takes a + donor chosen at random from its group. """ bank_sex: np.ndarray @@ -1197,11 +1215,30 @@ def fit( years, birth_year[chosen], ).astype(np.float32), - bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + bank_block=np.round( + np.clip(values, 0.0, 1.0) * _SHARE_SCALE + ).astype(np.uint16), k=k, ) return fill, {"bank": int(len(chosen))} + @property + def bank_digest(self) -> str: + """SHA-256 of the bank and k, the cache's key for this fill.""" + + cached = _BANK_DIGESTS.get(id(self)) + if cached is not None and cached[0] is self: + return cached[1] + digest = hashlib.sha256( + np.ascontiguousarray(self.bank_sex).tobytes() + + np.ascontiguousarray(self.bank_birth_year).tobytes() + + np.ascontiguousarray(self.bank_match).tobytes() + + np.ascontiguousarray(self.bank_block).tobytes() + + str(self.k).encode() + ).hexdigest() + _BANK_DIGESTS[id(self)] = (self, digest) + return digest + def _nearest(self, match, birth_year, sex, targets): """Each target's ``k`` nearest bank rows, and its group's bank rows. @@ -1214,7 +1251,7 @@ def _nearest(self, match, birth_year, sex, targets): + np.ascontiguousarray(birth_year[targets]).tobytes() + np.ascontiguousarray(sex[targets]).tobytes() + np.ascontiguousarray(targets).tobytes() - + str((id(self), self.k)).encode() + + self.bank_digest.encode() ).hexdigest() if digest in _NEAREST_CACHE: return _NEAREST_CACHE[digest] @@ -1494,7 +1531,6 @@ def fill( # With no known later year (a career starting after the file's # last year), the chain starts from a zero year. following = np.nan_to_num(following, nan=0.0) - ok = np.ones(len(rows), dtype=bool) age = year - birth_year[rows] keys = self._keys( sex[rows], age, np.nan_to_num(following), self.level_edges @@ -1509,7 +1545,7 @@ def fill( row[take] = found[take] drawn = np.full(len(rows), np.nan) for index in np.unique(level[level >= 0]): - take = (level == index) & ok + take = level == index p0 = self.level_p0[index][row[take]] positive = u[take] >= p0 v = np.where( @@ -1608,7 +1644,9 @@ def fill( shares[rows], years, np.asarray(birth_year)[rows], - sex[rows], + # Rows of uncoded sex take this part's sex, so its copula + # (calibrated for that sex) applies to them. + np.full(len(rows), value), np.asarray(person_key)[rows], fill_mask[rows], seed, diff --git a/tests/cohorts/test_psid2010_epuf_fill.py b/tests/cohorts/test_psid2010_epuf_fill.py index 61acea29..d8734376 100644 --- a/tests/cohorts/test_psid2010_epuf_fill.py +++ b/tests/cohorts/test_psid2010_epuf_fill.py @@ -77,6 +77,7 @@ def test_current_rule_fills_reproduce_the_assembler(): odd_fill=_MeanFill(), pre_fill=g.CurrentPreFill(), seed=7100, + odd_years=None, ) careers = result.careers before = cohort.careers.set_index(["person_id", "year"]) @@ -127,7 +128,9 @@ def fill(self, shares, years, birth, sex, key, mask, seed): out[mask] = 0.25 return out - result = fill_module.fill_careers(cohort, odd_fill=Spy(), seed=1) + result = fill_module.fill_careers( + cohort, odd_fill=Spy(), seed=1, odd_years=None + ) years = np.arange(1951, 2011) birth = np.array([b for b, _ in BIRTHS.values()]) pre = years[None, :] < np.maximum(1968, birth + 22)[:, None] @@ -156,3 +159,66 @@ def test_no_fill_keeps_the_careers(): result = fill_module.fill_careers(cohort, seed=1) assert len(result.careers) == len(cohort.careers) assert result.fills == {} + + +def test_by_default_only_the_scored_gap_years_are_filled(): + cohort = _cohort() + + class Constant: + name = "constant" + + def fill(self, shares, years, birth, sex, key, mask, seed): + out = shares.copy() + out[mask] = 0.25 + return out + + result = fill_module.fill_careers(cohort, odd_fill=Constant(), seed=1) + after = result.careers.set_index(["person_id", "year"]) + before = cohort.careers.set_index(["person_id", "year"]) + gaps = before.index[before["provenance"] == "gap_imputed"] + for key in gaps: + if key[1] in fill_module.SCORED_ODD_YEARS: + assert after.loc[key, "provenance"] == "gap_epuf_drawn" + else: + # 2007 and 2009 keep the assembler's value and provenance. + assert after.loc[key, "provenance"] == "gap_imputed" + assert after.loc[key, "earnings"] == before.loc[key, "earnings"] + + +def test_a_gap_with_no_visible_neighbour_keeps_the_assemblers_value(): + cohort = _cohort() + careers = cohort.careers + # Person 3 (born 1975) starts in 1997; drop 1998 so 1997's only + # neighbours are pre-career (1996) or missing. + careers = careers[~((careers.person_id == 3) & (careers.year == 1998))] + # Person 1 gets a 2013 seam filled from a 2014 boundary year, with no + # 2012 row: neither neighbour is visible to a fill. + extra = pd.DataFrame( + [ + (1, 2013, 30_000.0, "gap_imputed"), + (1, 2014, 30_000.0, "boundary_2014"), + ], + columns=careers.columns, + ) + cohort = SimpleNamespace( + persons=cohort.persons, + careers=pd.concat([careers, extra], ignore_index=True), + ) + + class Constant: + name = "constant" + + def fill(self, shares, years, birth, sex, key, mask, seed): + out = shares.copy() + out[mask] = 0.25 + return out + + result = fill_module.fill_careers( + cohort, odd_fill=Constant(), seed=1, odd_years=None + ) + after = result.careers.set_index(["person_id", "year"]) + assert after.loc[(3, 1997), "provenance"] == "gap_imputed" + assert after.loc[(1, 2013), "provenance"] == "gap_imputed" + assert after.loc[(1, 2013), "earnings"] == 30_000.0 + assert after.loc[(1, 2014), "provenance"] == "boundary_2014" + assert after.loc[(1, 1999), "provenance"] == "gap_epuf_drawn" diff --git a/tests/estimates/test_epuf_fill.py b/tests/estimates/test_epuf_fill.py index d311554f..747953bd 100644 --- a/tests/estimates/test_epuf_fill.py +++ b/tests/estimates/test_epuf_fill.py @@ -157,3 +157,44 @@ def test_the_forest_copula_is_calibrated_by_band(fitted): rho = part.rho assert rho.shape == (4, 6) assert ((rho >= 0) & (rho <= 0.9)).all() + + +def test_the_donor_cache_is_keyed_by_the_bank_not_the_object(): + shares, birth, sex, key = _shares(4, n=2_000) + first = F.PreDonorFill.fit( + shares, YEARS, birth, sex, key, k=3, bank_size=5 + )[0] + second = F.PreDonorFill.fit( + shares, YEARS, birth, sex, key + 1, k=3, bank_size=5 + )[0] + assert first.bank_digest != second.bank_digest + given, b, s, k = _given(5) + mask = g.family_mask("pre", b) + a = first.donors(given, YEARS, b, s, k, mask, 7100) + c = second.donors(given, YEARS, b, s, k, mask, 7100) + # Each fill's donors index its own bank and match its recipients' group. + for fill, chosen in ((first, a), (second, c)): + rows = np.flatnonzero(chosen >= 0) + assert (chosen[rows] < len(fill.bank_sex)).all() + assert (fill.bank_birth_year[chosen[rows]] == b[rows]).all() + + +def test_uncoded_sex_takes_the_routed_parts_sex(fitted): + fill = fitted["odd_forest"] + given, birth, sex, key = _given(6, n=400) + mask = g.family_mask("odd", birth) + uncoded = sex.copy() + uncoded[::3] = 3 + as_men = sex.copy() + as_men[::3] = 1 + out = fill.fill(given.copy(), YEARS, birth, uncoded, key, mask, 7100) + expected = fill.fill(given.copy(), YEARS, birth, as_men, key, mask, 7100) + rows = np.arange(len(sex))[::3] + np.testing.assert_array_equal( + np.nan_to_num(out[rows]), np.nan_to_num(expected[rows]) + ) + + +def test_fitted_forests_have_no_empty_leaf(fitted): + for part in fitted["odd_forest"].parts.values(): + assert (np.diff(part.leaf_offsets) > 0).all() diff --git a/tests/test_epuf_fill_candidates_manifest.py b/tests/test_epuf_fill_candidates_manifest.py index 1046e41f..c4b2dae6 100644 --- a/tests/test_epuf_fill_candidates_manifest.py +++ b/tests/test_epuf_fill_candidates_manifest.py @@ -96,3 +96,10 @@ def test_staged_fills_hash_to_the_manifest(manifest, name): if not path.is_file(): pytest.skip(f"{path} is not staged") assert hashlib.sha256(path.read_bytes()).hexdigest() == record["sha256"] + + +def test_the_registered_copula_binds_for_each_coded_sex(manifest): + diagnostics = manifest["fills"]["odd_forest"]["diagnostics"] + for sex in ("1", "2"): + rho = diagnostics[sex]["rho"] + assert any(value > 0 for value in rho[int(sex)]) From 15c61ee2c265f87a0131b36166849bc50ecbba83 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 05:07:22 -0400 Subject: [PATCH 04/13] gate_epuf_fill: withdraw the first candidate manifest for refit at the fixed code The first manifest (SHA-256 8d42153...) recorded the pre-review code commit. It was never used to read TEST. It is refitted at the reviewed code next; the artifacts' bytes are expected to be identical. Co-Authored-By: Claude Opus 5.5 --- runs/epuf_fill_candidates_v1.json | 1401 -------------------- runs/epuf_fill_candidates_v1.json.env.json | 19 - 2 files changed, 1420 deletions(-) delete mode 100644 runs/epuf_fill_candidates_v1.json delete mode 100644 runs/epuf_fill_candidates_v1.json.env.json diff --git a/runs/epuf_fill_candidates_v1.json b/runs/epuf_fill_candidates_v1.json deleted file mode 100644 index 5cfc3a61..00000000 --- a/runs/epuf_fill_candidates_v1.json +++ /dev/null @@ -1,1401 +0,0 @@ -{ - "schema": "populace_dynamics.epuf_fill_candidates.v1", - "registration_id": "2026-10-03-epuf-career-fill", - "code_commit": "9f069477bf0119df8d388354a8c69ebac948466b", - "code_files_clean": true, - "built_at_utc": "2026-10-04T08:30:59+00:00", - "part": "train", - "n_persons": 2629944, - "versions": { - "numpy": "2.5.1", - "scipy": "1.18.0", - "scikit_learn": "1.9.0" - }, - "staging": "files live outside the repository, like EPUF; refit with this script at code_commit to reproduce their bytes", - "fills": { - "odd_forest": { - "family": "odd", - "role": "primary", - "params": { - "unit_years": [ - 1991, - 2005 - ], - "n_units": 3000000, - "n_trees": 10, - "min_leaf": 15, - "max_features": 0.8, - "seed": 0 - }, - "file": "odd_forest_v1.npz", - "sha256": "37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa", - "bytes": 44836853, - "fit_seconds": 73.4, - "diagnostics": { - "1": { - "n_units": 3000000, - "n_positive_units": 1267291, - "n_trees": 10, - "min_leaf": 15, - "n_leaves": 250386, - "n_nodes": 500762, - "calibration_persons": 136391, - "rho": [ - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ], - [ - 0.1, - 0.1, - 0.15, - 0.15, - 0.45, - 0.45 - ], - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ], - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ] - ], - "calibration": { - "1.1.rho_0.0": [ - -0.00937054964764883, - -0.006568577649590401 - ], - "1.2.rho_0.0": [ - -0.009185985434689847, - -0.017087922487923013 - ], - "1.3.rho_0.0": [ - -0.008760608986963514, - -0.005892891185897087 - ], - "1.4.rho_0.0": [ - -0.014822290100570679, - -0.0029128598220162782 - ], - "2.1.rho_0.0": [ - "nan", - "nan" - ], - "2.2.rho_0.0": [ - "nan", - "nan" - ], - "2.3.rho_0.0": [ - "nan", - "nan" - ], - "2.4.rho_0.0": [ - "nan", - "nan" - ], - "1.1.rho_0.05": [ - -0.0025221662657621824, - -0.0025405849711389594 - ], - "1.2.rho_0.05": [ - -0.005747487224877057, - -0.014480367835756236 - ], - "1.3.rho_0.05": [ - -0.006653418258712018, - -0.003659386315042923 - ], - "1.4.rho_0.05": [ - -0.013918461926527237, - -0.0014846818917918503 - ], - "2.1.rho_0.05": [ - "nan", - "nan" - ], - "2.2.rho_0.05": [ - "nan", - "nan" - ], - "2.3.rho_0.05": [ - "nan", - "nan" - ], - "2.4.rho_0.05": [ - "nan", - "nan" - ], - "1.1.rho_0.1": [ - 0.0022622655578613537, - 0.0017571621431731188 - ], - "1.2.rho_0.1": [ - -0.003463604263099551, - -0.013817366044770019 - ], - "1.3.rho_0.1": [ - -0.0043256050642532795, - -0.0020772149123101658 - ], - "1.4.rho_0.1": [ - -0.011090372488350764, - 0.0003620037980666124 - ], - "2.1.rho_0.1": [ - "nan", - "nan" - ], - "2.2.rho_0.1": [ - "nan", - "nan" - ], - "2.3.rho_0.1": [ - "nan", - "nan" - ], - "2.4.rho_0.1": [ - "nan", - "nan" - ], - "1.1.rho_0.15": [ - 0.00735065280082603, - 0.00547377504308344 - ], - "1.2.rho_0.15": [ - -0.00043591174742074745, - -0.011407436015852035 - ], - "1.3.rho_0.15": [ - -0.001469747919156772, - -6.807326989843876e-05 - ], - "1.4.rho_0.15": [ - -0.007842751587615049, - 0.0035570177498509548 - ], - "2.1.rho_0.15": [ - "nan", - "nan" - ], - "2.2.rho_0.15": [ - "nan", - "nan" - ], - "2.3.rho_0.15": [ - "nan", - "nan" - ], - "2.4.rho_0.15": [ - "nan", - "nan" - ], - "1.1.rho_0.2": [ - 0.01235436992080352, - 0.007278338434604126 - ], - "1.2.rho_0.2": [ - 0.0018633014396086667, - -0.00951255602154033 - ], - "1.3.rho_0.2": [ - 0.0014639774440903253, - 0.0017977312065096118 - ], - "1.4.rho_0.2": [ - -0.007672387545435644, - 0.005532053798539716 - ], - "2.1.rho_0.2": [ - "nan", - "nan" - ], - "2.2.rho_0.2": [ - "nan", - "nan" - ], - "2.3.rho_0.2": [ - "nan", - "nan" - ], - "2.4.rho_0.2": [ - "nan", - "nan" - ], - "1.1.rho_0.25": [ - 0.017532837584111616, - 0.012910343871726515 - ], - "1.2.rho_0.25": [ - 0.005432755081799634, - -0.0058967638868855365 - ], - "1.3.rho_0.25": [ - 0.004461578182951564, - 0.003245424028756938 - ], - "1.4.rho_0.25": [ - -0.006896856131266005, - 0.007402313522806625 - ], - "2.1.rho_0.25": [ - "nan", - "nan" - ], - "2.2.rho_0.25": [ - "nan", - "nan" - ], - "2.3.rho_0.25": [ - "nan", - "nan" - ], - "2.4.rho_0.25": [ - "nan", - "nan" - ], - "1.1.rho_0.3": [ - 0.02283322303498425, - 0.01782293559705972 - ], - "1.2.rho_0.3": [ - 0.007917659257822063, - -0.0042736292458098735 - ], - "1.3.rho_0.3": [ - 0.005912180793823385, - 0.004708972924462929 - ], - "1.4.rho_0.3": [ - -0.0037362954092261536, - 0.012169812222417309 - ], - "2.1.rho_0.3": [ - "nan", - "nan" - ], - "2.2.rho_0.3": [ - "nan", - "nan" - ], - "2.3.rho_0.3": [ - "nan", - "nan" - ], - "2.4.rho_0.3": [ - "nan", - "nan" - ], - "1.1.rho_0.35": [ - 0.02936744538086289, - 0.02269930584128066 - ], - "1.2.rho_0.35": [ - 0.010991221409968999, - -0.0020151785137914047 - ], - "1.3.rho_0.35": [ - 0.008154499506314195, - 0.006864335349464179 - ], - "1.4.rho_0.35": [ - -0.0033667339378640193, - 0.010716886425056416 - ], - "2.1.rho_0.35": [ - "nan", - "nan" - ], - "2.2.rho_0.35": [ - "nan", - "nan" - ], - "2.3.rho_0.35": [ - "nan", - "nan" - ], - "2.4.rho_0.35": [ - "nan", - "nan" - ], - "1.1.rho_0.4": [ - 0.033991025478408377, - 0.025266554928974005 - ], - "1.2.rho_0.4": [ - 0.01395682475791915, - 0.00013629350615307345 - ], - "1.3.rho_0.4": [ - 0.009812482470721196, - 0.008655649864332982 - ], - "1.4.rho_0.4": [ - -0.0018532086431369832, - 0.010892834474684587 - ], - "2.1.rho_0.4": [ - "nan", - "nan" - ], - "2.2.rho_0.4": [ - "nan", - "nan" - ], - "2.3.rho_0.4": [ - "nan", - "nan" - ], - "2.4.rho_0.4": [ - "nan", - "nan" - ], - "1.1.rho_0.45": [ - 0.04078840124406191, - 0.03125399303118248 - ], - "1.2.rho_0.45": [ - 0.016853239128019615, - 0.0023949314237127206 - ], - "1.3.rho_0.45": [ - 0.012218818061125014, - 0.0103161648513439 - ], - "1.4.rho_0.45": [ - 0.0007757976805343736, - 0.012061337583499476 - ], - "2.1.rho_0.45": [ - "nan", - "nan" - ], - "2.2.rho_0.45": [ - "nan", - "nan" - ], - "2.3.rho_0.45": [ - "nan", - "nan" - ], - "2.4.rho_0.45": [ - "nan", - "nan" - ], - "1.1.rho_0.5": [ - 0.04627958132301435, - 0.035137264874086305 - ], - "1.2.rho_0.5": [ - 0.020285078982263505, - 0.004963715627156029 - ], - "1.3.rho_0.5": [ - 0.014599779251665335, - 0.012138390969464785 - ], - "1.4.rho_0.5": [ - 0.004685831714231314, - 0.015445592554809928 - ], - "2.1.rho_0.5": [ - "nan", - "nan" - ], - "2.2.rho_0.5": [ - "nan", - "nan" - ], - "2.3.rho_0.5": [ - "nan", - "nan" - ], - "2.4.rho_0.5": [ - "nan", - "nan" - ], - "1.1.rho_0.55": [ - 0.0531305404913871, - 0.03958250535156038 - ], - "1.2.rho_0.55": [ - 0.023484969938003974, - 0.007224393712798816 - ], - "1.3.rho_0.55": [ - 0.016946622715777737, - 0.013757911329974615 - ], - "1.4.rho_0.55": [ - 0.005386466630163178, - 0.01558864988803843 - ], - "2.1.rho_0.55": [ - "nan", - "nan" - ], - "2.2.rho_0.55": [ - "nan", - "nan" - ], - "2.3.rho_0.55": [ - "nan", - "nan" - ], - "2.4.rho_0.55": [ - "nan", - "nan" - ], - "1.1.rho_0.6": [ - 0.0594398122408063, - 0.045308024035041305 - ], - "1.2.rho_0.6": [ - 0.02665635658052512, - 0.01009428430629955 - ], - "1.3.rho_0.6": [ - 0.020127026484403232, - 0.017144987639204246 - ], - "1.4.rho_0.6": [ - 0.0098509448536922, - 0.01789310382229714 - ], - "2.1.rho_0.6": [ - "nan", - "nan" - ], - "2.2.rho_0.6": [ - "nan", - "nan" - ], - "2.3.rho_0.6": [ - "nan", - "nan" - ], - "2.4.rho_0.6": [ - "nan", - "nan" - ], - "1.1.rho_0.65": [ - 0.06636452154965067, - 0.048690911284104854 - ], - "1.2.rho_0.65": [ - 0.02988279156747209, - 0.012593270355836461 - ], - "1.3.rho_0.65": [ - 0.023092885480960224, - 0.018772455583715764 - ], - "1.4.rho_0.65": [ - 0.01241442885124, - 0.02076244248410586 - ], - "2.1.rho_0.65": [ - "nan", - "nan" - ], - "2.2.rho_0.65": [ - "nan", - "nan" - ], - "2.3.rho_0.65": [ - "nan", - "nan" - ], - "2.4.rho_0.65": [ - "nan", - "nan" - ], - "1.1.rho_0.7": [ - 0.07254218457867379, - 0.0526910466051157 - ], - "1.2.rho_0.7": [ - 0.03329166905191294, - 0.015175105205589068 - ], - "1.3.rho_0.7": [ - 0.0258155301357077, - 0.020502762171670685 - ], - "1.4.rho_0.7": [ - 0.014453411163469543, - 0.020846671921446847 - ], - "2.1.rho_0.7": [ - "nan", - "nan" - ], - "2.2.rho_0.7": [ - "nan", - "nan" - ], - "2.3.rho_0.7": [ - "nan", - "nan" - ], - "2.4.rho_0.7": [ - "nan", - "nan" - ], - "1.1.rho_0.75": [ - 0.07916776188641705, - 0.057886291657737066 - ], - "1.2.rho_0.75": [ - 0.03725049246712253, - 0.018772837363772443 - ], - "1.3.rho_0.75": [ - 0.028609776086117367, - 0.02307257411267194 - ], - "1.4.rho_0.75": [ - 0.016869649042968726, - 0.02362923713906029 - ], - "2.1.rho_0.75": [ - "nan", - "nan" - ], - "2.2.rho_0.75": [ - "nan", - "nan" - ], - "2.3.rho_0.75": [ - "nan", - "nan" - ], - "2.4.rho_0.75": [ - "nan", - "nan" - ], - "1.1.rho_0.8": [ - 0.08680076817335436, - 0.06389736793273171 - ], - "1.2.rho_0.8": [ - 0.040963145105353815, - 0.021929454410539284 - ], - "1.3.rho_0.8": [ - 0.03147254961686874, - 0.02464394115084445 - ], - "1.4.rho_0.8": [ - 0.017611836623551924, - 0.02768136631318152 - ], - "2.1.rho_0.8": [ - "nan", - "nan" - ], - "2.2.rho_0.8": [ - "nan", - "nan" - ], - "2.3.rho_0.8": [ - "nan", - "nan" - ], - "2.4.rho_0.8": [ - "nan", - "nan" - ], - "1.1.rho_0.85": [ - 0.09307651453825694, - 0.06816742141440546 - ], - "1.2.rho_0.85": [ - 0.04423261009456436, - 0.024040513696466315 - ], - "1.3.rho_0.85": [ - 0.03373239871230438, - 0.02743872467269859 - ], - "1.4.rho_0.85": [ - 0.02157870721620181, - 0.02748553411253518 - ], - "2.1.rho_0.85": [ - "nan", - "nan" - ], - "2.2.rho_0.85": [ - "nan", - "nan" - ], - "2.3.rho_0.85": [ - "nan", - "nan" - ], - "2.4.rho_0.85": [ - "nan", - "nan" - ], - "1.1.rho_0.9": [ - 0.10107252485664509, - 0.07345061064036906 - ], - "1.2.rho_0.9": [ - 0.048178209491760327, - 0.026937093526634648 - ], - "1.3.rho_0.9": [ - 0.036002361151492135, - 0.02854935592244301 - ], - "1.4.rho_0.9": [ - 0.024308070051008657, - 0.02733002869966339 - ], - "2.1.rho_0.9": [ - "nan", - "nan" - ], - "2.2.rho_0.9": [ - "nan", - "nan" - ], - "2.3.rho_0.9": [ - "nan", - "nan" - ], - "2.4.rho_0.9": [ - "nan", - "nan" - ] - } - }, - "2": { - "n_units": 3000000, - "n_positive_units": 1215539, - "n_trees": 10, - "min_leaf": 15, - "n_leaves": 246831, - "n_nodes": 493652, - "calibration_persons": 126094, - "rho": [ - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ], - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ], - [ - 0.05, - 0.05, - 0.15, - 0.25, - 0.9, - 0.9 - ], - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ] - ], - "calibration": { - "1.1.rho_0.0": [ - "nan", - "nan" - ], - "1.2.rho_0.0": [ - "nan", - "nan" - ], - "1.3.rho_0.0": [ - "nan", - "nan" - ], - "1.4.rho_0.0": [ - "nan", - "nan" - ], - "2.1.rho_0.0": [ - -0.0007703488441346273, - -0.0010508879964993278 - ], - "2.2.rho_0.0": [ - -0.007493032484792828, - -0.013193098454176932 - ], - "2.3.rho_0.0": [ - -0.009429974516101503, - -0.01195790961172627 - ], - "2.4.rho_0.0": [ - -0.03692428228749378, - -0.046884412622328786 - ], - "1.1.rho_0.05": [ - "nan", - "nan" - ], - "1.2.rho_0.05": [ - "nan", - "nan" - ], - "1.3.rho_0.05": [ - "nan", - "nan" - ], - "1.4.rho_0.05": [ - "nan", - "nan" - ], - "2.1.rho_0.05": [ - 0.0011501760114371873, - 0.0002049050210158887 - ], - "2.2.rho_0.05": [ - -0.005131023108985944, - -0.01216928087935032 - ], - "2.3.rho_0.05": [ - -0.008520996492143773, - -0.010608354960477073 - ], - "2.4.rho_0.05": [ - -0.03473496584300617, - -0.05402554667776105 - ], - "1.1.rho_0.1": [ - "nan", - "nan" - ], - "1.2.rho_0.1": [ - "nan", - "nan" - ], - "1.3.rho_0.1": [ - "nan", - "nan" - ], - "1.4.rho_0.1": [ - "nan", - "nan" - ], - "2.1.rho_0.1": [ - 0.006927157675038598, - 0.004290292190204048 - ], - "2.2.rho_0.1": [ - -0.001703132719541478, - -0.010456836329726604 - ], - "2.3.rho_0.1": [ - -0.007321013146343036, - -0.009035871667865014 - ], - "2.4.rho_0.1": [ - -0.030882723546502233, - -0.048033202775634276 - ], - "1.1.rho_0.15": [ - "nan", - "nan" - ], - "1.2.rho_0.15": [ - "nan", - "nan" - ], - "1.3.rho_0.15": [ - "nan", - "nan" - ], - "1.4.rho_0.15": [ - "nan", - "nan" - ], - "2.1.rho_0.15": [ - 0.011751352091050937, - 0.008076268003208154 - ], - "2.2.rho_0.15": [ - 0.0020210219373677507, - -0.0070992294258820365 - ], - "2.3.rho_0.15": [ - -0.005528714504464016, - -0.0069653177500850205 - ], - "2.4.rho_0.15": [ - -0.029923398294732673, - -0.05002522943078225 - ], - "1.1.rho_0.2": [ - "nan", - "nan" - ], - "1.2.rho_0.2": [ - "nan", - "nan" - ], - "1.3.rho_0.2": [ - "nan", - "nan" - ], - "1.4.rho_0.2": [ - "nan", - "nan" - ], - "2.1.rho_0.2": [ - 0.016343948418381715, - 0.011187555046907938 - ], - "2.2.rho_0.2": [ - 0.0033473023560638415, - -0.0067290354973356115 - ], - "2.3.rho_0.2": [ - -0.002877760745891411, - -0.0033953050183980205 - ], - "2.4.rho_0.2": [ - -0.025096610145121545, - -0.05020372098724413 - ], - "1.1.rho_0.25": [ - "nan", - "nan" - ], - "1.2.rho_0.25": [ - "nan", - "nan" - ], - "1.3.rho_0.25": [ - "nan", - "nan" - ], - "1.4.rho_0.25": [ - "nan", - "nan" - ], - "2.1.rho_0.25": [ - 0.021425408842097315, - 0.01429463213074944 - ], - "2.2.rho_0.25": [ - 0.0065957040671158484, - -0.004705541238731792 - ], - "2.3.rho_0.25": [ - -1.7567476131130633e-05, - -0.0010814093051690898 - ], - "2.4.rho_0.25": [ - -0.0212615966241938, - -0.04842399617411386 - ], - "1.1.rho_0.3": [ - "nan", - "nan" - ], - "1.2.rho_0.3": [ - "nan", - "nan" - ], - "1.3.rho_0.3": [ - "nan", - "nan" - ], - "1.4.rho_0.3": [ - "nan", - "nan" - ], - "2.1.rho_0.3": [ - 0.02693682357503624, - 0.016009169813349877 - ], - "2.2.rho_0.3": [ - 0.009021593962986185, - -0.001825312400977941 - ], - "2.3.rho_0.3": [ - 0.002480943042680095, - -0.0003852668318783392 - ], - "2.4.rho_0.3": [ - -0.020630295377933372, - -0.05211604831280381 - ], - "1.1.rho_0.35": [ - "nan", - "nan" - ], - "1.2.rho_0.35": [ - "nan", - "nan" - ], - "1.3.rho_0.35": [ - "nan", - "nan" - ], - "1.4.rho_0.35": [ - "nan", - "nan" - ], - "2.1.rho_0.35": [ - 0.03296557024474411, - 0.021046225240840877 - ], - "2.2.rho_0.35": [ - 0.012335421323450668, - 0.0016580900689775468 - ], - "2.3.rho_0.35": [ - 0.004414993633125364, - 0.0012509622619388816 - ], - "2.4.rho_0.35": [ - -0.017200316022581097, - -0.04689517410403199 - ], - "1.1.rho_0.4": [ - "nan", - "nan" - ], - "1.2.rho_0.4": [ - "nan", - "nan" - ], - "1.3.rho_0.4": [ - "nan", - "nan" - ], - "1.4.rho_0.4": [ - "nan", - "nan" - ], - "2.1.rho_0.4": [ - 0.03881518696695174, - 0.02617586875204103 - ], - "2.2.rho_0.4": [ - 0.014916453205874203, - 0.0031898299704220534 - ], - "2.3.rho_0.4": [ - 0.006149846624828648, - 0.0020327028751055964 - ], - "2.4.rho_0.4": [ - -0.014400133887248034, - -0.04119061699894566 - ], - "1.1.rho_0.45": [ - "nan", - "nan" - ], - "1.2.rho_0.45": [ - "nan", - "nan" - ], - "1.3.rho_0.45": [ - "nan", - "nan" - ], - "1.4.rho_0.45": [ - "nan", - "nan" - ], - "2.1.rho_0.45": [ - 0.04408209328648782, - 0.03100375174528891 - ], - "2.2.rho_0.45": [ - 0.018276462063509857, - 0.0054696188221817765 - ], - "2.3.rho_0.45": [ - 0.0076902880826571485, - 0.00218766241834345 - ], - "2.4.rho_0.45": [ - -0.014908359815873351, - -0.0393019026587349 - ], - "1.1.rho_0.5": [ - "nan", - "nan" - ], - "1.2.rho_0.5": [ - "nan", - "nan" - ], - "1.3.rho_0.5": [ - "nan", - "nan" - ], - "1.4.rho_0.5": [ - "nan", - "nan" - ], - "2.1.rho_0.5": [ - 0.04970088094513325, - 0.034854865497334075 - ], - "2.2.rho_0.5": [ - 0.021724257795143087, - 0.008193822209299872 - ], - "2.3.rho_0.5": [ - 0.009745593525695817, - 0.003562541404541153 - ], - "2.4.rho_0.5": [ - -0.014383130743268246, - -0.04271338044982742 - ], - "1.1.rho_0.55": [ - "nan", - "nan" - ], - "1.2.rho_0.55": [ - "nan", - "nan" - ], - "1.3.rho_0.55": [ - "nan", - "nan" - ], - "1.4.rho_0.55": [ - "nan", - "nan" - ], - "2.1.rho_0.55": [ - 0.05502109102831798, - 0.03890967682059354 - ], - "2.2.rho_0.55": [ - 0.024839414066947896, - 0.010572170562098249 - ], - "2.3.rho_0.55": [ - 0.011772980929248611, - 0.003908378086523001 - ], - "2.4.rho_0.55": [ - -0.013131700928571188, - -0.04285851871646029 - ], - "1.1.rho_0.6": [ - "nan", - "nan" - ], - "1.2.rho_0.6": [ - "nan", - "nan" - ], - "1.3.rho_0.6": [ - "nan", - "nan" - ], - "1.4.rho_0.6": [ - "nan", - "nan" - ], - "2.1.rho_0.6": [ - 0.06088619981198029, - 0.0420376792891467 - ], - "2.2.rho_0.6": [ - 0.028468055929460223, - 0.012861481376605255 - ], - "2.3.rho_0.6": [ - 0.014264953161897465, - 0.006716731791151176 - ], - "2.4.rho_0.6": [ - -0.01411854107127919, - -0.04322951820182941 - ], - "1.1.rho_0.65": [ - "nan", - "nan" - ], - "1.2.rho_0.65": [ - "nan", - "nan" - ], - "1.3.rho_0.65": [ - "nan", - "nan" - ], - "1.4.rho_0.65": [ - "nan", - "nan" - ], - "2.1.rho_0.65": [ - 0.06682408268645768, - 0.04478633762350159 - ], - "2.2.rho_0.65": [ - 0.03150037859622068, - 0.014635098136048796 - ], - "2.3.rho_0.65": [ - 0.016061091785444348, - 0.008363861184702004 - ], - "2.4.rho_0.65": [ - -0.013183126314957772, - -0.04726560797289203 - ], - "1.1.rho_0.7": [ - "nan", - "nan" - ], - "1.2.rho_0.7": [ - "nan", - "nan" - ], - "1.3.rho_0.7": [ - "nan", - "nan" - ], - "1.4.rho_0.7": [ - "nan", - "nan" - ], - "2.1.rho_0.7": [ - 0.07410396740658753, - 0.05061327862615472 - ], - "2.2.rho_0.7": [ - 0.03457230831577063, - 0.01752794757556897 - ], - "2.3.rho_0.7": [ - 0.018780887727013362, - 0.009609081527960028 - ], - "2.4.rho_0.7": [ - -0.011927889063067187, - -0.04575175956846467 - ], - "1.1.rho_0.75": [ - "nan", - "nan" - ], - "1.2.rho_0.75": [ - "nan", - "nan" - ], - "1.3.rho_0.75": [ - "nan", - "nan" - ], - "1.4.rho_0.75": [ - "nan", - "nan" - ], - "2.1.rho_0.75": [ - 0.08092707240167418, - 0.05483242561629709 - ], - "2.2.rho_0.75": [ - 0.03658199157962372, - 0.018699837196579194 - ], - "2.3.rho_0.75": [ - 0.021946755597646472, - 0.011296140023272394 - ], - "2.4.rho_0.75": [ - -0.009788245012555041, - -0.045540955638491365 - ], - "1.1.rho_0.8": [ - "nan", - "nan" - ], - "1.2.rho_0.8": [ - "nan", - "nan" - ], - "1.3.rho_0.8": [ - "nan", - "nan" - ], - "1.4.rho_0.8": [ - "nan", - "nan" - ], - "2.1.rho_0.8": [ - 0.08743759706500098, - 0.060096830084552244 - ], - "2.2.rho_0.8": [ - 0.038972747910327676, - 0.01962918025075 - ], - "2.3.rho_0.8": [ - 0.02443160357398244, - 0.011703961899924842 - ], - "2.4.rho_0.8": [ - -0.008951498727275298, - -0.042132069352851076 - ], - "1.1.rho_0.85": [ - "nan", - "nan" - ], - "1.2.rho_0.85": [ - "nan", - "nan" - ], - "1.3.rho_0.85": [ - "nan", - "nan" - ], - "1.4.rho_0.85": [ - "nan", - "nan" - ], - "2.1.rho_0.85": [ - 0.09255511715216769, - 0.06287928620893612 - ], - "2.2.rho_0.85": [ - 0.04200109430794119, - 0.021762940153467025 - ], - "2.3.rho_0.85": [ - 0.027419910814330595, - 0.013140726589435658 - ], - "2.4.rho_0.85": [ - -0.0057698938150033685, - -0.03462993739531095 - ], - "1.1.rho_0.9": [ - "nan", - "nan" - ], - "1.2.rho_0.9": [ - "nan", - "nan" - ], - "1.3.rho_0.9": [ - "nan", - "nan" - ], - "1.4.rho_0.9": [ - "nan", - "nan" - ], - "2.1.rho_0.9": [ - 0.09982884654946289, - 0.06787532650906625 - ], - "2.2.rho_0.9": [ - 0.04549333428061564, - 0.02470270553781595 - ], - "2.3.rho_0.9": [ - 0.03002262611212725, - 0.014889240292517925 - ], - "2.4.rho_0.9": [ - -0.0024130894891942756, - -0.034578411611535076 - ] - } - } - } - }, - "odd_knn": { - "family": "odd", - "role": "alternative", - "params": { - "unit_years": [ - 1991, - 2005 - ], - "k": 10, - "seed": 0 - }, - "file": "odd_knn_v1.npz", - "sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", - "bytes": 4076580, - "fit_seconds": 6.1, - "diagnostics": { - "n_units": 27840831, - "bank": 1147199 - } - }, - "pre_chain": { - "family": "pre", - "role": "alternative", - "params": { - "unit_years": [ - 1951, - 2005 - ] - }, - "file": "pre_chain_v1.npz", - "sha256": "8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724", - "bytes": 236450, - "fit_seconds": 74.8, - "diagnostics": { - "n_units": 144646920 - } - }, - "pre_donor": { - "family": "pre", - "role": "primary", - "params": { - "k": 3, - "bank_size": 100000, - "birth_years": [ - 1905, - 1985 - ] - }, - "file": "pre_donor_v1.npz", - "sha256": "3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7", - "bytes": 31768107, - "fit_seconds": 2.8, - "diagnostics": { - "bank": 1518845 - } - } - }, - "elapsed_seconds": 188.1 -} diff --git a/runs/epuf_fill_candidates_v1.json.env.json b/runs/epuf_fill_candidates_v1.json.env.json deleted file mode 100644 index 652146fc..00000000 --- a/runs/epuf_fill_candidates_v1.json.env.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "environment": { - "python": "3.14.4", - "numpy": "2.5.1", - "pandas": "3.0.3", - "sklearn": "1.9.0", - "scipy": "1.18.0", - "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O", - "fitting_stack": { - "populace_fit": "absent", - "populace_frame": "absent" - } - }, - "contract": { - "blob_sha": "b0c39af1e13a705f90b85d3e6b9a91e1d3c5485c", - "head_sha": "9f069477bf0119df8d388354a8c69ebac948466b", - "path": "gates.yaml" - } -} From ab25c0e5659a245283adca9cd07170427a55058a Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 05:12:43 -0400 Subject: [PATCH 05/13] gate_epuf_fill: candidates refitted at the reviewed code; manifest pinned runs/epuf_fill_candidates_v1.json is refitted at 598e4436 with the code files clean; the four artifacts' SHA-256 values are unchanged from the withdrawn manifest. The score script pins this manifest and checks the lock before it leaves any marker. The registration document records the code review, the refit and why the DEV dry run still stands. Co-Authored-By: Claude Opus 5.5 --- .../gate_epuf_fill_candidates_registration.md | 49 +- ...te_epuf_fill_pr516_code_review_20261004.md | 90 ++ runs/epuf_fill_candidates_v1.json | 1404 +++++++++++++++++ runs/epuf_fill_candidates_v1.json.env.json | 19 + scripts/score_epuf_fill_test.py | 12 +- tests/README-tiers.md | 6 +- tests/tier_counts.json | 4 +- 7 files changed, 1573 insertions(+), 11 deletions(-) create mode 100644 reviews/gate_epuf_fill_pr516_code_review_20261004.md create mode 100644 runs/epuf_fill_candidates_v1.json create mode 100644 runs/epuf_fill_candidates_v1.json.env.json diff --git a/docs/amendments/gate_epuf_fill_candidates_registration.md b/docs/amendments/gate_epuf_fill_candidates_registration.md index 3d20a536..4efb7efe 100644 --- a/docs/amendments/gate_epuf_fill_candidates_registration.md +++ b/docs/amendments/gate_epuf_fill_candidates_registration.md @@ -20,10 +20,15 @@ ## The registered artifacts -- **Manifest**: `runs/epuf_fill_candidates_v1.json`, SHA-256 `8d421536351d884184af441889333db2361fb0f50326e3a4481e6fa4f07aeecf`. -- **Fitted at**: `9f069477` on TRAIN, with the code files clean. -- **Libraries**: numpy 2.5.1, scipy 1.18.0, scikit-learn 1.9.0. -- **Reproducibility**: a second fit at the same commit reproduced all four files byte for byte. +- **Manifest**: `runs/epuf_fill_candidates_v1.json`, SHA-256 `a304311343f3c78f7702ec6918b991b6bea30dad2a23529df6f0f975ce7f62d0`. +- **Fitted at**: `598e4436` on TRAIN, with the code files clean. +- **Environment**: numpy 2.5.1, scipy 1.18.0, scikit-learn 1.9.0, Python 3.14.4, zlib 1.2.12, on macOS-26.6.2-arm64-arm-64bit-Mach-O. +- **Reproducibility**: + - A second fit at the same commit reproduced all four files byte for + byte. + - The refit after code review (below) did too. + - Reproducing the bytes needs the same library, zlib and platform + versions. The deflate output can depend on the zlib build. | Name | Role | File | SHA-256 | Bytes | |---|---|---|---|---:| @@ -49,7 +54,7 @@ mean, geometric mean and count positive; the mean positive share and the share of positive years at offsets 5-9; sex; age; and the year. - **Training units**: TRAIN units inside the career, with contexts that see - the career only. + the career only. `n_units`, 3,000,000, applies to each sex's forest. - **Copula**: a person-level Gaussian copula with correlation by sex and age band. It is calibrated on one TRAIN person in ten, held out of the forests, to match two- and four-year persistence between masked years. @@ -126,6 +131,40 @@ fifths of EPUF of nearly equal size, so their scores should agree closely. before the career is hidden from every fill. The oracle O1 misses there too. +## Code review and refit + +An independent code review of PR #516 returned REQUEST CHANGES +(`reviews/gate_epuf_fill_pr516_code_review_20261004.md`). The fixes: + +**`fill_careers` (PSID-2010).** +- It fills only the gap years the gate scored (1997-2005) unless the + caller opts into every gap year, which is an uncertified extrapolation. +- A gap year with no neighbour the fill can see keeps the assembler's + value. + +**`epuf_fill.py`.** +- The donor cache is keyed by the bank's content. +- Persons of uncoded sex take the routed part's copula. +- A forest leaf can never be empty: none is, and the smallest holds 16 + values. +- The career summaries in the donor match stop at 2006. + +**The scripts.** +- The fit script refuses an existing manifest before fitting and never + replaces a staged file with other bytes. +- The score script pins this manifest's SHA-256, records whether its code + was clean, and publishes no local paths. + +**The manifest.** The first manifest (SHA-256 `8d42153...`) was withdrawn +and refitted at the reviewed code (`598e4436`). The four artifacts' +SHA-256 values are unchanged. + +**The DEV dry run still stands.** It scored the same artifacts, and none of +the fixes changes a draw for a person the gate scores. They touch the PSID +application, which EPUF scoring does not use; the cache's key; the draws +for persons of uncoded sex, who are never scored; and the bank summaries' +years, which on EPUF end in 2006 anyway. + ## Procedure on TEST (after lock) 1. Confirm that `gates.yaml` locks `gate_epuf_fill` and that the staged files diff --git a/reviews/gate_epuf_fill_pr516_code_review_20261004.md b/reviews/gate_epuf_fill_pr516_code_review_20261004.md new file mode 100644 index 00000000..48e0af5c --- /dev/null +++ b/reviews/gate_epuf_fill_pr516_code_review_20261004.md @@ -0,0 +1,90 @@ +**Verdict: REQUEST CHANGES** + +The registration itself holds up. The candidates are fitted on TRAIN only, nothing reads a masked cell, the files are pinned by SHA-256, and the DEV numbers in the document match the log. But the opt-in PSID path has a real correctness bug (finding 1), and two robustness hazards (findings 2 and 3) should be fixed before lock. + +Most fixes touch `epuf_fill.py`. That file is pinned by `test_the_fill_code_is_unchanged_since_the_fit`, so changing it means re-fitting at a new commit and updating `code_commit`. None of the fixes below changes the bytes `to_bytes` writes (finding 4 is the one exception, flagged there). + +**I could not run git or pytest in this session**: only read and search tools were available. So I have not checked `git diff 70553e70 940ad6f1`, whether `career.py`, `cohorts/psid2010.py` and the registered runs are untouched, the test results, or the manifest's SHA-256. Everything below comes from reading the files at the PR head. + +## Findings + +**1. Medium — learned odd fills break or diverge on real PSID gap years** (`psid2010_epuf_fill.py:120-128`, `epuf_fill.py:983-1003`, `epuf_fill.py:652-698`) +- **What goes wrong:** `fill_careers` builds the shares the fills see from `observed` rows only. But `career._impute_gap` (`career.py:964-978`) also fills a gap from other neighbours: a pre-career observed year, or `boundary_2014` for the 2013 seam (`career.py:1059-1062`). So a `gap_imputed` year can have no neighbour the fill can see. Two examples: + - the 2013 seam when 2012 is unknown; + - a person whose career starts in 1997, whose 1996 is pre-career, and whose 1998 is missing. +- **What each fill does with such a year:** + - `OddKnnFill` leaves it NaN, because `take` requires a finite `left`. `drawn()` then raises "a learned fill returned an invalid share", so the registered alternative crashes on real data. + - `OddForestFill` silently writes a zero where the assembler wrote the neighbour's value. +- **Out-of-range years:** the module docstring (`:11-12`) says *every* `gap_imputed` year gets a draw. That includes 2007-2013, which lie outside the fills' training years (1991-2005) and outside anything the gate scores (`epuf_fill_gate.py:141-143`). +- **Tests miss it:** neither test's data has the seam, `boundary_2014`, or a gap with no visible neighbour. +- **Fix (in `psid2010_epuf_fill.py`, which is not pinned):** + - leave gap rows with no visible neighbour as `gap_imputed`, or route them through an explicit fallback; + - then either limit the learned odd fill to the scored years (≤2005) or state the extrapolation in the docstring and the registration; + - add a seam case to the test. + +**2. Medium — the nearest-donor cache is keyed by `id(self)`, not by content** (`epuf_fill.py:1212-1220`) +- **What goes wrong:** the cache key includes `str((id(self), self.k))`. CPython reuses addresses, so if one `PreDonorFill` is garbage-collected and another with a different bank but the same `k` lands at the same address, the same inputs hit the cache. That returns the old bank's donor indices for the new bank: wrong blocks, or an index out of range. +- **Registered path:** not affected, because the fills are loaded once and stay alive. +- **Fix:** key on a digest of `bank_match`, `bank_sex`, `bank_birth_year` and `k`, computed once. Add a test with two fills fitted on different banks. + +**3. Medium-low — the fit script overwrites the registered files before it refuses** (`scripts/fit_epuf_fills.py:178-179, 207`) +- **What goes wrong:** `path.write_bytes(blob)` replaces `~/…/fills/{name}_v1.npz` unconditionally. `write_new` only refuses an existing manifest at the very end. The manifest's note invites a refit to reproduce the bytes; doing that at a different commit or in a different environment would destroy the staged registered files, and only then fail. +- **Fix:** check that the manifest does not exist before fitting. Write each file with `open(path, "xb")`, or write to a temporary path and compare against the existing file. + +**4. Low — tree thresholds are stored as float32, so traversal can differ from scikit-learn's** (`epuf_fill.py:531, 544, 822, 840`) +- **What goes wrong:** scikit-learn's threshold is a float64 midpoint of two float32 values, compared as `float64(x) <= thr`. Casting that midpoint to float32 rounds half-to-even when the two values are one ulp apart, which can give `Xf[p]`. Then `x == Xf[p]` goes left here and right in scikit-learn. +- **Effect:** the stored model is self-consistent, since the fit and the draws use the same traversal. But leaf membership can differ from `estimator.apply`, and an empty leaf makes `_value` (`:636`) read `leaf_values[start-1]`, the neighbouring leaf's value. +- **Tests:** no test compares the traversal with `apply`. +- **Fix:** add a test asserting every leaf has a count of at least 1 and that `_tree_leaves` matches `apply`. Keeping the thresholds in float64 would change the artifacts' bytes, so do that only if you re-register. + +**5. Low — persons of uncoded sex get no copula** (`epuf_fill.py:678`, `:1601-1604`) +- **What goes wrong:** `BySexFill` sends sex 3 to the men's part. That part looks up `rho[3, band]`, which is 0.0 because calibration only sets the row of its own sex (manifest: the men's part has `rho` rows 0, 2 and 3 all zero). So these persons' masked years are drawn independently. +- **Fix:** in `BySexFill.fill`, pass `sex = value` for the rows it routes there. + +**6. Low — "byte-reproducible" depends on the environment** (`epuf_fill.py:184-206`, `fit_epuf_fills.py:197-201`) +- **What goes wrong:** the deflate output depends on the zlib build (for example zlib against zlib-ng), and `ZipInfo.create_system` depends on the platform. The manifest records the numpy, scipy and scikit-learn versions, but not the Python version, the zlib runtime version or the platform. +- **Fix:** record `zlib.ZLIB_RUNTIME_VERSION`, the Python version and the platform, or use `ZIP_STORED`. Qualify the reproducibility claim in the registration document. + +**7. Low — provenance gaps in what the scripts record** (`scripts/score_epuf_fill_test.py:59-81`, `fit_epuf_fills.py:43-46`) +- **Score script:** + - It refuses before lock (through `test_part`, `epuf_fill_gate.py:1450-1453`) and refuses an existing output — both fine. + - But it accepts any `--manifest` and only records that manifest's SHA. Pin the registered `8d42153…` and refuse others. + - It does not record whether the working tree was clean. Record that, as the fit script does. + - It writes an absolute local path (including the home directory) into a JSON file that will be published. + - If the run crashes after the TEST read, nothing is written. Consider writing a "started" record first. +- **Fit script:** `CODE_FILES` leaves out `harness/epuf_fill_gate.py` and `harness/epuf_operator.py`, which the fit depends on (TRAIN split, `YEARS`, `wage_base`). So the "code unchanged" test does not cover them. + +**8. Nits** +- `PreDonorFill`'s class docstring is stale (`epuf_fill.py:1124-1136`). It says 2,000 donors, `k` of 10, five match years and "scaled by five". The code matches on `MATCH_DIMS = 7`, and the registered fill uses a bank of 100,000 and `k` of 3. +- `start_year` in `fill_careers` is really the *last* year (`psid2010_epuf_fill.py:87, 99`). If it is set below 2013, later gap rows silently keep the assembler's value. +- `bank_block` is cast to `uint16` without clipping (`:1200`). A share above 1 would wrap. EPUF is capped so it is safe today, but the forest clips and this should too. +- `ok` in `PreChainFill.fill` is always True (`:1497`). +- For PSID recipients, `match_vector`'s career summaries (`:1080`) cover years past 2006, while the bank's stop at 2006. Consider limiting them to ≤2006. +- `test_the_forest_copula_is_calibrated_by_band` only checks shape and range. Assert that the coded-sex row of each part is nonzero. +- The `n_units` registered as 3,000,000 applies per sex: the diagnostics show 3M for each part. Say so in the registration document. + +## What I verified (by reading) + +- **Leakage:** + - `fit_epuf_fills` reads only `epuf_matrix(TRAIN)`. + - The copula's calibration persons are TRAIN persons held out of the forests. + - Every fill reads only cells it is allowed to read: `known = isfinite & ~mask`, or the masked cells set to NaN. + - The scoring path hides the union mask (`epuf_fill_gate.py:1151-1153`). + - The forest's training contexts exclude pre-career years. +- **`hash_uniform`:** keyed by stream, seed, person and year; values strictly inside (0, 1); broadcasting is correct. +- **The draws:** + - The forest: zero below `p0`, otherwise the rescaled quantile; one `eta` per person; one `epsilon` per unit; a separate uniform picks the tree. + - kNN and the chain: draws do not depend on the persons' order. + - The donor draw: each group's donors are stored contiguously, so the fallback random donor is correct. +- **Loading:** `load_fill` checks the SHA-256 before parsing, and nested `BySexFill` files round-trip. +- **`fill_careers`:** + - observed rows are untouched; + - gap rows are relabelled `gap_epuf_drawn`; + - pre-career rows are only added (`pre_mask & ~in_career`) as `pre_career_epuf_donor`; + - shares are converted to dollars at each year's base; + - wage bases past 2006 come from the captured step function (`ss/params.py:151-160`). +- **Birth-evidence reducer:** both new modules are excluded in `scripts/first_estimates_birth_evidence.py:358-361`, with the reachability assertions in `tests/estimates/test_birth_evidence_artifact.py:217-220, 476-482`. +- **Claims:** + - The registration document's SHAs, sizes, library versions and `code_commit` match `runs/epuf_fill_candidates_v1.json`. + - The manifest SHA `8d42153…` matches the one in the DEV dry-run log (line 17). + - The dry run's tier counts match the document's tables: odd primary 7 of 183 failing ("improves"), odd alternative 46, current odd rule 100 and 99; pre primary 0 of 136 ("certified"), pre alternative 50, current pre rule 131. \ No newline at end of file diff --git a/runs/epuf_fill_candidates_v1.json b/runs/epuf_fill_candidates_v1.json new file mode 100644 index 00000000..ce2e3f49 --- /dev/null +++ b/runs/epuf_fill_candidates_v1.json @@ -0,0 +1,1404 @@ +{ + "schema": "populace_dynamics.epuf_fill_candidates.v1", + "registration_id": "2026-10-03-epuf-career-fill", + "code_commit": "598e44366621fff26469ec87f23ff4871215efdc", + "code_files_clean": true, + "built_at_utc": "2026-10-04T09:07:47+00:00", + "part": "train", + "n_persons": 2629944, + "versions": { + "numpy": "2.5.1", + "scipy": "1.18.0", + "scikit_learn": "1.9.0", + "python": "3.14.4", + "zlib_runtime": "1.2.12", + "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O" + }, + "staging": "files live outside the repository, like EPUF; a refit with this script at code_commit, on the same library, zlib and platform versions, reproduces their bytes", + "fills": { + "odd_forest": { + "family": "odd", + "role": "primary", + "params": { + "unit_years": [ + 1991, + 2005 + ], + "n_units": 3000000, + "n_trees": 10, + "min_leaf": 15, + "max_features": 0.8, + "seed": 0 + }, + "file": "odd_forest_v1.npz", + "sha256": "37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa", + "bytes": 44836853, + "fit_seconds": 67.5, + "diagnostics": { + "1": { + "n_units": 3000000, + "n_positive_units": 1267291, + "n_trees": 10, + "min_leaf": 15, + "n_leaves": 250386, + "n_nodes": 500762, + "calibration_persons": 136391, + "rho": [ + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.1, + 0.1, + 0.15, + 0.15, + 0.45, + 0.45 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + ], + "calibration": { + "1.1.rho_0.0": [ + -0.00937054964764883, + -0.006568577649590401 + ], + "1.2.rho_0.0": [ + -0.009185985434689847, + -0.017087922487923013 + ], + "1.3.rho_0.0": [ + -0.008760608986963514, + -0.005892891185897087 + ], + "1.4.rho_0.0": [ + -0.014822290100570679, + -0.0029128598220162782 + ], + "2.1.rho_0.0": [ + "nan", + "nan" + ], + "2.2.rho_0.0": [ + "nan", + "nan" + ], + "2.3.rho_0.0": [ + "nan", + "nan" + ], + "2.4.rho_0.0": [ + "nan", + "nan" + ], + "1.1.rho_0.05": [ + -0.0025221662657621824, + -0.0025405849711389594 + ], + "1.2.rho_0.05": [ + -0.005747487224877057, + -0.014480367835756236 + ], + "1.3.rho_0.05": [ + -0.006653418258712018, + -0.003659386315042923 + ], + "1.4.rho_0.05": [ + -0.013918461926527237, + -0.0014846818917918503 + ], + "2.1.rho_0.05": [ + "nan", + "nan" + ], + "2.2.rho_0.05": [ + "nan", + "nan" + ], + "2.3.rho_0.05": [ + "nan", + "nan" + ], + "2.4.rho_0.05": [ + "nan", + "nan" + ], + "1.1.rho_0.1": [ + 0.0022622655578613537, + 0.0017571621431731188 + ], + "1.2.rho_0.1": [ + -0.003463604263099551, + -0.013817366044770019 + ], + "1.3.rho_0.1": [ + -0.0043256050642532795, + -0.0020772149123101658 + ], + "1.4.rho_0.1": [ + -0.011090372488350764, + 0.0003620037980666124 + ], + "2.1.rho_0.1": [ + "nan", + "nan" + ], + "2.2.rho_0.1": [ + "nan", + "nan" + ], + "2.3.rho_0.1": [ + "nan", + "nan" + ], + "2.4.rho_0.1": [ + "nan", + "nan" + ], + "1.1.rho_0.15": [ + 0.00735065280082603, + 0.00547377504308344 + ], + "1.2.rho_0.15": [ + -0.00043591174742074745, + -0.011407436015852035 + ], + "1.3.rho_0.15": [ + -0.001469747919156772, + -6.807326989843876e-05 + ], + "1.4.rho_0.15": [ + -0.007842751587615049, + 0.0035570177498509548 + ], + "2.1.rho_0.15": [ + "nan", + "nan" + ], + "2.2.rho_0.15": [ + "nan", + "nan" + ], + "2.3.rho_0.15": [ + "nan", + "nan" + ], + "2.4.rho_0.15": [ + "nan", + "nan" + ], + "1.1.rho_0.2": [ + 0.01235436992080352, + 0.007278338434604126 + ], + "1.2.rho_0.2": [ + 0.0018633014396086667, + -0.00951255602154033 + ], + "1.3.rho_0.2": [ + 0.0014639774440903253, + 0.0017977312065096118 + ], + "1.4.rho_0.2": [ + -0.007672387545435644, + 0.005532053798539716 + ], + "2.1.rho_0.2": [ + "nan", + "nan" + ], + "2.2.rho_0.2": [ + "nan", + "nan" + ], + "2.3.rho_0.2": [ + "nan", + "nan" + ], + "2.4.rho_0.2": [ + "nan", + "nan" + ], + "1.1.rho_0.25": [ + 0.017532837584111616, + 0.012910343871726515 + ], + "1.2.rho_0.25": [ + 0.005432755081799634, + -0.0058967638868855365 + ], + "1.3.rho_0.25": [ + 0.004461578182951564, + 0.003245424028756938 + ], + "1.4.rho_0.25": [ + -0.006896856131266005, + 0.007402313522806625 + ], + "2.1.rho_0.25": [ + "nan", + "nan" + ], + "2.2.rho_0.25": [ + "nan", + "nan" + ], + "2.3.rho_0.25": [ + "nan", + "nan" + ], + "2.4.rho_0.25": [ + "nan", + "nan" + ], + "1.1.rho_0.3": [ + 0.02283322303498425, + 0.01782293559705972 + ], + "1.2.rho_0.3": [ + 0.007917659257822063, + -0.0042736292458098735 + ], + "1.3.rho_0.3": [ + 0.005912180793823385, + 0.004708972924462929 + ], + "1.4.rho_0.3": [ + -0.0037362954092261536, + 0.012169812222417309 + ], + "2.1.rho_0.3": [ + "nan", + "nan" + ], + "2.2.rho_0.3": [ + "nan", + "nan" + ], + "2.3.rho_0.3": [ + "nan", + "nan" + ], + "2.4.rho_0.3": [ + "nan", + "nan" + ], + "1.1.rho_0.35": [ + 0.02936744538086289, + 0.02269930584128066 + ], + "1.2.rho_0.35": [ + 0.010991221409968999, + -0.0020151785137914047 + ], + "1.3.rho_0.35": [ + 0.008154499506314195, + 0.006864335349464179 + ], + "1.4.rho_0.35": [ + -0.0033667339378640193, + 0.010716886425056416 + ], + "2.1.rho_0.35": [ + "nan", + "nan" + ], + "2.2.rho_0.35": [ + "nan", + "nan" + ], + "2.3.rho_0.35": [ + "nan", + "nan" + ], + "2.4.rho_0.35": [ + "nan", + "nan" + ], + "1.1.rho_0.4": [ + 0.033991025478408377, + 0.025266554928974005 + ], + "1.2.rho_0.4": [ + 0.01395682475791915, + 0.00013629350615307345 + ], + "1.3.rho_0.4": [ + 0.009812482470721196, + 0.008655649864332982 + ], + "1.4.rho_0.4": [ + -0.0018532086431369832, + 0.010892834474684587 + ], + "2.1.rho_0.4": [ + "nan", + "nan" + ], + "2.2.rho_0.4": [ + "nan", + "nan" + ], + "2.3.rho_0.4": [ + "nan", + "nan" + ], + "2.4.rho_0.4": [ + "nan", + "nan" + ], + "1.1.rho_0.45": [ + 0.04078840124406191, + 0.03125399303118248 + ], + "1.2.rho_0.45": [ + 0.016853239128019615, + 0.0023949314237127206 + ], + "1.3.rho_0.45": [ + 0.012218818061125014, + 0.0103161648513439 + ], + "1.4.rho_0.45": [ + 0.0007757976805343736, + 0.012061337583499476 + ], + "2.1.rho_0.45": [ + "nan", + "nan" + ], + "2.2.rho_0.45": [ + "nan", + "nan" + ], + "2.3.rho_0.45": [ + "nan", + "nan" + ], + "2.4.rho_0.45": [ + "nan", + "nan" + ], + "1.1.rho_0.5": [ + 0.04627958132301435, + 0.035137264874086305 + ], + "1.2.rho_0.5": [ + 0.020285078982263505, + 0.004963715627156029 + ], + "1.3.rho_0.5": [ + 0.014599779251665335, + 0.012138390969464785 + ], + "1.4.rho_0.5": [ + 0.004685831714231314, + 0.015445592554809928 + ], + "2.1.rho_0.5": [ + "nan", + "nan" + ], + "2.2.rho_0.5": [ + "nan", + "nan" + ], + "2.3.rho_0.5": [ + "nan", + "nan" + ], + "2.4.rho_0.5": [ + "nan", + "nan" + ], + "1.1.rho_0.55": [ + 0.0531305404913871, + 0.03958250535156038 + ], + "1.2.rho_0.55": [ + 0.023484969938003974, + 0.007224393712798816 + ], + "1.3.rho_0.55": [ + 0.016946622715777737, + 0.013757911329974615 + ], + "1.4.rho_0.55": [ + 0.005386466630163178, + 0.01558864988803843 + ], + "2.1.rho_0.55": [ + "nan", + "nan" + ], + "2.2.rho_0.55": [ + "nan", + "nan" + ], + "2.3.rho_0.55": [ + "nan", + "nan" + ], + "2.4.rho_0.55": [ + "nan", + "nan" + ], + "1.1.rho_0.6": [ + 0.0594398122408063, + 0.045308024035041305 + ], + "1.2.rho_0.6": [ + 0.02665635658052512, + 0.01009428430629955 + ], + "1.3.rho_0.6": [ + 0.020127026484403232, + 0.017144987639204246 + ], + "1.4.rho_0.6": [ + 0.0098509448536922, + 0.01789310382229714 + ], + "2.1.rho_0.6": [ + "nan", + "nan" + ], + "2.2.rho_0.6": [ + "nan", + "nan" + ], + "2.3.rho_0.6": [ + "nan", + "nan" + ], + "2.4.rho_0.6": [ + "nan", + "nan" + ], + "1.1.rho_0.65": [ + 0.06636452154965067, + 0.048690911284104854 + ], + "1.2.rho_0.65": [ + 0.02988279156747209, + 0.012593270355836461 + ], + "1.3.rho_0.65": [ + 0.023092885480960224, + 0.018772455583715764 + ], + "1.4.rho_0.65": [ + 0.01241442885124, + 0.02076244248410586 + ], + "2.1.rho_0.65": [ + "nan", + "nan" + ], + "2.2.rho_0.65": [ + "nan", + "nan" + ], + "2.3.rho_0.65": [ + "nan", + "nan" + ], + "2.4.rho_0.65": [ + "nan", + "nan" + ], + "1.1.rho_0.7": [ + 0.07254218457867379, + 0.0526910466051157 + ], + "1.2.rho_0.7": [ + 0.03329166905191294, + 0.015175105205589068 + ], + "1.3.rho_0.7": [ + 0.0258155301357077, + 0.020502762171670685 + ], + "1.4.rho_0.7": [ + 0.014453411163469543, + 0.020846671921446847 + ], + "2.1.rho_0.7": [ + "nan", + "nan" + ], + "2.2.rho_0.7": [ + "nan", + "nan" + ], + "2.3.rho_0.7": [ + "nan", + "nan" + ], + "2.4.rho_0.7": [ + "nan", + "nan" + ], + "1.1.rho_0.75": [ + 0.07916776188641705, + 0.057886291657737066 + ], + "1.2.rho_0.75": [ + 0.03725049246712253, + 0.018772837363772443 + ], + "1.3.rho_0.75": [ + 0.028609776086117367, + 0.02307257411267194 + ], + "1.4.rho_0.75": [ + 0.016869649042968726, + 0.02362923713906029 + ], + "2.1.rho_0.75": [ + "nan", + "nan" + ], + "2.2.rho_0.75": [ + "nan", + "nan" + ], + "2.3.rho_0.75": [ + "nan", + "nan" + ], + "2.4.rho_0.75": [ + "nan", + "nan" + ], + "1.1.rho_0.8": [ + 0.08680076817335436, + 0.06389736793273171 + ], + "1.2.rho_0.8": [ + 0.040963145105353815, + 0.021929454410539284 + ], + "1.3.rho_0.8": [ + 0.03147254961686874, + 0.02464394115084445 + ], + "1.4.rho_0.8": [ + 0.017611836623551924, + 0.02768136631318152 + ], + "2.1.rho_0.8": [ + "nan", + "nan" + ], + "2.2.rho_0.8": [ + "nan", + "nan" + ], + "2.3.rho_0.8": [ + "nan", + "nan" + ], + "2.4.rho_0.8": [ + "nan", + "nan" + ], + "1.1.rho_0.85": [ + 0.09307651453825694, + 0.06816742141440546 + ], + "1.2.rho_0.85": [ + 0.04423261009456436, + 0.024040513696466315 + ], + "1.3.rho_0.85": [ + 0.03373239871230438, + 0.02743872467269859 + ], + "1.4.rho_0.85": [ + 0.02157870721620181, + 0.02748553411253518 + ], + "2.1.rho_0.85": [ + "nan", + "nan" + ], + "2.2.rho_0.85": [ + "nan", + "nan" + ], + "2.3.rho_0.85": [ + "nan", + "nan" + ], + "2.4.rho_0.85": [ + "nan", + "nan" + ], + "1.1.rho_0.9": [ + 0.10107252485664509, + 0.07345061064036906 + ], + "1.2.rho_0.9": [ + 0.048178209491760327, + 0.026937093526634648 + ], + "1.3.rho_0.9": [ + 0.036002361151492135, + 0.02854935592244301 + ], + "1.4.rho_0.9": [ + 0.024308070051008657, + 0.02733002869966339 + ], + "2.1.rho_0.9": [ + "nan", + "nan" + ], + "2.2.rho_0.9": [ + "nan", + "nan" + ], + "2.3.rho_0.9": [ + "nan", + "nan" + ], + "2.4.rho_0.9": [ + "nan", + "nan" + ] + } + }, + "2": { + "n_units": 3000000, + "n_positive_units": 1215539, + "n_trees": 10, + "min_leaf": 15, + "n_leaves": 246831, + "n_nodes": 493652, + "calibration_persons": 126094, + "rho": [ + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.05, + 0.05, + 0.15, + 0.25, + 0.9, + 0.9 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + ], + "calibration": { + "1.1.rho_0.0": [ + "nan", + "nan" + ], + "1.2.rho_0.0": [ + "nan", + "nan" + ], + "1.3.rho_0.0": [ + "nan", + "nan" + ], + "1.4.rho_0.0": [ + "nan", + "nan" + ], + "2.1.rho_0.0": [ + -0.0007703488441346273, + -0.0010508879964993278 + ], + "2.2.rho_0.0": [ + -0.007493032484792828, + -0.013193098454176932 + ], + "2.3.rho_0.0": [ + -0.009429974516101503, + -0.01195790961172627 + ], + "2.4.rho_0.0": [ + -0.03692428228749378, + -0.046884412622328786 + ], + "1.1.rho_0.05": [ + "nan", + "nan" + ], + "1.2.rho_0.05": [ + "nan", + "nan" + ], + "1.3.rho_0.05": [ + "nan", + "nan" + ], + "1.4.rho_0.05": [ + "nan", + "nan" + ], + "2.1.rho_0.05": [ + 0.0011501760114371873, + 0.0002049050210158887 + ], + "2.2.rho_0.05": [ + -0.005131023108985944, + -0.01216928087935032 + ], + "2.3.rho_0.05": [ + -0.008520996492143773, + -0.010608354960477073 + ], + "2.4.rho_0.05": [ + -0.03473496584300617, + -0.05402554667776105 + ], + "1.1.rho_0.1": [ + "nan", + "nan" + ], + "1.2.rho_0.1": [ + "nan", + "nan" + ], + "1.3.rho_0.1": [ + "nan", + "nan" + ], + "1.4.rho_0.1": [ + "nan", + "nan" + ], + "2.1.rho_0.1": [ + 0.006927157675038598, + 0.004290292190204048 + ], + "2.2.rho_0.1": [ + -0.001703132719541478, + -0.010456836329726604 + ], + "2.3.rho_0.1": [ + -0.007321013146343036, + -0.009035871667865014 + ], + "2.4.rho_0.1": [ + -0.030882723546502233, + -0.048033202775634276 + ], + "1.1.rho_0.15": [ + "nan", + "nan" + ], + "1.2.rho_0.15": [ + "nan", + "nan" + ], + "1.3.rho_0.15": [ + "nan", + "nan" + ], + "1.4.rho_0.15": [ + "nan", + "nan" + ], + "2.1.rho_0.15": [ + 0.011751352091050937, + 0.008076268003208154 + ], + "2.2.rho_0.15": [ + 0.0020210219373677507, + -0.0070992294258820365 + ], + "2.3.rho_0.15": [ + -0.005528714504464016, + -0.0069653177500850205 + ], + "2.4.rho_0.15": [ + -0.029923398294732673, + -0.05002522943078225 + ], + "1.1.rho_0.2": [ + "nan", + "nan" + ], + "1.2.rho_0.2": [ + "nan", + "nan" + ], + "1.3.rho_0.2": [ + "nan", + "nan" + ], + "1.4.rho_0.2": [ + "nan", + "nan" + ], + "2.1.rho_0.2": [ + 0.016343948418381715, + 0.011187555046907938 + ], + "2.2.rho_0.2": [ + 0.0033473023560638415, + -0.0067290354973356115 + ], + "2.3.rho_0.2": [ + -0.002877760745891411, + -0.0033953050183980205 + ], + "2.4.rho_0.2": [ + -0.025096610145121545, + -0.05020372098724413 + ], + "1.1.rho_0.25": [ + "nan", + "nan" + ], + "1.2.rho_0.25": [ + "nan", + "nan" + ], + "1.3.rho_0.25": [ + "nan", + "nan" + ], + "1.4.rho_0.25": [ + "nan", + "nan" + ], + "2.1.rho_0.25": [ + 0.021425408842097315, + 0.01429463213074944 + ], + "2.2.rho_0.25": [ + 0.0065957040671158484, + -0.004705541238731792 + ], + "2.3.rho_0.25": [ + -1.7567476131130633e-05, + -0.0010814093051690898 + ], + "2.4.rho_0.25": [ + -0.0212615966241938, + -0.04842399617411386 + ], + "1.1.rho_0.3": [ + "nan", + "nan" + ], + "1.2.rho_0.3": [ + "nan", + "nan" + ], + "1.3.rho_0.3": [ + "nan", + "nan" + ], + "1.4.rho_0.3": [ + "nan", + "nan" + ], + "2.1.rho_0.3": [ + 0.02693682357503624, + 0.016009169813349877 + ], + "2.2.rho_0.3": [ + 0.009021593962986185, + -0.001825312400977941 + ], + "2.3.rho_0.3": [ + 0.002480943042680095, + -0.0003852668318783392 + ], + "2.4.rho_0.3": [ + -0.020630295377933372, + -0.05211604831280381 + ], + "1.1.rho_0.35": [ + "nan", + "nan" + ], + "1.2.rho_0.35": [ + "nan", + "nan" + ], + "1.3.rho_0.35": [ + "nan", + "nan" + ], + "1.4.rho_0.35": [ + "nan", + "nan" + ], + "2.1.rho_0.35": [ + 0.03296557024474411, + 0.021046225240840877 + ], + "2.2.rho_0.35": [ + 0.012335421323450668, + 0.0016580900689775468 + ], + "2.3.rho_0.35": [ + 0.004414993633125364, + 0.0012509622619388816 + ], + "2.4.rho_0.35": [ + -0.017200316022581097, + -0.04689517410403199 + ], + "1.1.rho_0.4": [ + "nan", + "nan" + ], + "1.2.rho_0.4": [ + "nan", + "nan" + ], + "1.3.rho_0.4": [ + "nan", + "nan" + ], + "1.4.rho_0.4": [ + "nan", + "nan" + ], + "2.1.rho_0.4": [ + 0.03881518696695174, + 0.02617586875204103 + ], + "2.2.rho_0.4": [ + 0.014916453205874203, + 0.0031898299704220534 + ], + "2.3.rho_0.4": [ + 0.006149846624828648, + 0.0020327028751055964 + ], + "2.4.rho_0.4": [ + -0.014400133887248034, + -0.04119061699894566 + ], + "1.1.rho_0.45": [ + "nan", + "nan" + ], + "1.2.rho_0.45": [ + "nan", + "nan" + ], + "1.3.rho_0.45": [ + "nan", + "nan" + ], + "1.4.rho_0.45": [ + "nan", + "nan" + ], + "2.1.rho_0.45": [ + 0.04408209328648782, + 0.03100375174528891 + ], + "2.2.rho_0.45": [ + 0.018276462063509857, + 0.0054696188221817765 + ], + "2.3.rho_0.45": [ + 0.0076902880826571485, + 0.00218766241834345 + ], + "2.4.rho_0.45": [ + -0.014908359815873351, + -0.0393019026587349 + ], + "1.1.rho_0.5": [ + "nan", + "nan" + ], + "1.2.rho_0.5": [ + "nan", + "nan" + ], + "1.3.rho_0.5": [ + "nan", + "nan" + ], + "1.4.rho_0.5": [ + "nan", + "nan" + ], + "2.1.rho_0.5": [ + 0.04970088094513325, + 0.034854865497334075 + ], + "2.2.rho_0.5": [ + 0.021724257795143087, + 0.008193822209299872 + ], + "2.3.rho_0.5": [ + 0.009745593525695817, + 0.003562541404541153 + ], + "2.4.rho_0.5": [ + -0.014383130743268246, + -0.04271338044982742 + ], + "1.1.rho_0.55": [ + "nan", + "nan" + ], + "1.2.rho_0.55": [ + "nan", + "nan" + ], + "1.3.rho_0.55": [ + "nan", + "nan" + ], + "1.4.rho_0.55": [ + "nan", + "nan" + ], + "2.1.rho_0.55": [ + 0.05502109102831798, + 0.03890967682059354 + ], + "2.2.rho_0.55": [ + 0.024839414066947896, + 0.010572170562098249 + ], + "2.3.rho_0.55": [ + 0.011772980929248611, + 0.003908378086523001 + ], + "2.4.rho_0.55": [ + -0.013131700928571188, + -0.04285851871646029 + ], + "1.1.rho_0.6": [ + "nan", + "nan" + ], + "1.2.rho_0.6": [ + "nan", + "nan" + ], + "1.3.rho_0.6": [ + "nan", + "nan" + ], + "1.4.rho_0.6": [ + "nan", + "nan" + ], + "2.1.rho_0.6": [ + 0.06088619981198029, + 0.0420376792891467 + ], + "2.2.rho_0.6": [ + 0.028468055929460223, + 0.012861481376605255 + ], + "2.3.rho_0.6": [ + 0.014264953161897465, + 0.006716731791151176 + ], + "2.4.rho_0.6": [ + -0.01411854107127919, + -0.04322951820182941 + ], + "1.1.rho_0.65": [ + "nan", + "nan" + ], + "1.2.rho_0.65": [ + "nan", + "nan" + ], + "1.3.rho_0.65": [ + "nan", + "nan" + ], + "1.4.rho_0.65": [ + "nan", + "nan" + ], + "2.1.rho_0.65": [ + 0.06682408268645768, + 0.04478633762350159 + ], + "2.2.rho_0.65": [ + 0.03150037859622068, + 0.014635098136048796 + ], + "2.3.rho_0.65": [ + 0.016061091785444348, + 0.008363861184702004 + ], + "2.4.rho_0.65": [ + -0.013183126314957772, + -0.04726560797289203 + ], + "1.1.rho_0.7": [ + "nan", + "nan" + ], + "1.2.rho_0.7": [ + "nan", + "nan" + ], + "1.3.rho_0.7": [ + "nan", + "nan" + ], + "1.4.rho_0.7": [ + "nan", + "nan" + ], + "2.1.rho_0.7": [ + 0.07410396740658753, + 0.05061327862615472 + ], + "2.2.rho_0.7": [ + 0.03457230831577063, + 0.01752794757556897 + ], + "2.3.rho_0.7": [ + 0.018780887727013362, + 0.009609081527960028 + ], + "2.4.rho_0.7": [ + -0.011927889063067187, + -0.04575175956846467 + ], + "1.1.rho_0.75": [ + "nan", + "nan" + ], + "1.2.rho_0.75": [ + "nan", + "nan" + ], + "1.3.rho_0.75": [ + "nan", + "nan" + ], + "1.4.rho_0.75": [ + "nan", + "nan" + ], + "2.1.rho_0.75": [ + 0.08092707240167418, + 0.05483242561629709 + ], + "2.2.rho_0.75": [ + 0.03658199157962372, + 0.018699837196579194 + ], + "2.3.rho_0.75": [ + 0.021946755597646472, + 0.011296140023272394 + ], + "2.4.rho_0.75": [ + -0.009788245012555041, + -0.045540955638491365 + ], + "1.1.rho_0.8": [ + "nan", + "nan" + ], + "1.2.rho_0.8": [ + "nan", + "nan" + ], + "1.3.rho_0.8": [ + "nan", + "nan" + ], + "1.4.rho_0.8": [ + "nan", + "nan" + ], + "2.1.rho_0.8": [ + 0.08743759706500098, + 0.060096830084552244 + ], + "2.2.rho_0.8": [ + 0.038972747910327676, + 0.01962918025075 + ], + "2.3.rho_0.8": [ + 0.02443160357398244, + 0.011703961899924842 + ], + "2.4.rho_0.8": [ + -0.008951498727275298, + -0.042132069352851076 + ], + "1.1.rho_0.85": [ + "nan", + "nan" + ], + "1.2.rho_0.85": [ + "nan", + "nan" + ], + "1.3.rho_0.85": [ + "nan", + "nan" + ], + "1.4.rho_0.85": [ + "nan", + "nan" + ], + "2.1.rho_0.85": [ + 0.09255511715216769, + 0.06287928620893612 + ], + "2.2.rho_0.85": [ + 0.04200109430794119, + 0.021762940153467025 + ], + "2.3.rho_0.85": [ + 0.027419910814330595, + 0.013140726589435658 + ], + "2.4.rho_0.85": [ + -0.0057698938150033685, + -0.03462993739531095 + ], + "1.1.rho_0.9": [ + "nan", + "nan" + ], + "1.2.rho_0.9": [ + "nan", + "nan" + ], + "1.3.rho_0.9": [ + "nan", + "nan" + ], + "1.4.rho_0.9": [ + "nan", + "nan" + ], + "2.1.rho_0.9": [ + 0.09982884654946289, + 0.06787532650906625 + ], + "2.2.rho_0.9": [ + 0.04549333428061564, + 0.02470270553781595 + ], + "2.3.rho_0.9": [ + 0.03002262611212725, + 0.014889240292517925 + ], + "2.4.rho_0.9": [ + -0.0024130894891942756, + -0.034578411611535076 + ] + } + } + } + }, + "odd_knn": { + "family": "odd", + "role": "alternative", + "params": { + "unit_years": [ + 1991, + 2005 + ], + "k": 10, + "seed": 0 + }, + "file": "odd_knn_v1.npz", + "sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", + "bytes": 4076580, + "fit_seconds": 6.2, + "diagnostics": { + "n_units": 27840831, + "bank": 1147199 + } + }, + "pre_chain": { + "family": "pre", + "role": "alternative", + "params": { + "unit_years": [ + 1951, + 2005 + ] + }, + "file": "pre_chain_v1.npz", + "sha256": "8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724", + "bytes": 236450, + "fit_seconds": 74.0, + "diagnostics": { + "n_units": 144646920 + } + }, + "pre_donor": { + "family": "pre", + "role": "primary", + "params": { + "k": 3, + "bank_size": 100000, + "birth_years": [ + 1905, + 1985 + ] + }, + "file": "pre_donor_v1.npz", + "sha256": "3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7", + "bytes": 31768107, + "fit_seconds": 2.9, + "diagnostics": { + "bank": 1518845 + } + } + }, + "elapsed_seconds": 182.1 +} diff --git a/runs/epuf_fill_candidates_v1.json.env.json b/runs/epuf_fill_candidates_v1.json.env.json new file mode 100644 index 00000000..7e9ba5df --- /dev/null +++ b/runs/epuf_fill_candidates_v1.json.env.json @@ -0,0 +1,19 @@ +{ + "environment": { + "python": "3.14.4", + "numpy": "2.5.1", + "pandas": "3.0.3", + "sklearn": "1.9.0", + "scipy": "1.18.0", + "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O", + "fitting_stack": { + "populace_fit": "absent", + "populace_frame": "absent" + } + }, + "contract": { + "blob_sha": "b0c39af1e13a705f90b85d3e6b9a91e1d3c5485c", + "head_sha": "598e44366621fff26469ec87f23ff4871215efdc", + "path": "gates.yaml" + } +} diff --git a/scripts/score_epuf_fill_test.py b/scripts/score_epuf_fill_test.py index 3d595133..79594ebc 100644 --- a/scripts/score_epuf_fill_test.py +++ b/scripts/score_epuf_fill_test.py @@ -26,13 +26,16 @@ from pathlib import Path from populace_dynamics.artifacts import write_new +from populace_dynamics.harness import epuf_fill_gate as g from populace_dynamics.harness import epuf_fill_scoring as scoring ROOT = Path(__file__).resolve().parents[1] DEFAULT_DIR = Path("~/PolicyEngine/epuf-data/fills").expanduser() #: The registered manifest; any other manifest is refused. REGISTERED_MANIFEST = "runs/epuf_fill_candidates_v1.json" -REGISTERED_MANIFEST_SHA256 = "PENDING_REFIT" +REGISTERED_MANIFEST_SHA256 = ( + "a304311343f3c78f7702ec6918b991b6bea30dad2a23529df6f0f975ce7f62d0" +) #: Files whose state the record reports. CODE_FILES = ( "src/populace_dynamics/harness/epuf_fill_gate.py", @@ -84,6 +87,13 @@ def main() -> None: ).returncode == 0 ) + # Refuse before any marker unless the gate is locked (test_part checks + # again before it reads TEST). + status = g._gate_lock_status(ROOT / "gates.yaml") + if not status["locked"] or status["registration_id"] != g.REGISTRATION_ID: + raise g.TestPartLocked( + "gate_epuf_fill is not locked; TEST stays unread" + ) # A started marker, so a run that fails after reading TEST leaves a # trace of the read. marker = Path(f"{args.output}.started.json") diff --git a/tests/README-tiers.md b/tests/README-tiers.md index 13db9c18..ee7bf28c 100644 --- a/tests/README-tiers.md +++ b/tests/README-tiers.md @@ -39,9 +39,9 @@ pytest --collect-only -q -m oracle_policyengine | tail -1 | Tier | Tests at HEAD | |---|---:| -| `unit` | 6,020 | -| `artifact` | 3,395 | +| `unit` | 6,025 | +| `artifact` | 3,396 | | `integration_psid` | 1,341 | | `reproduction_legacy` | 520 | | `oracle_policyengine` | 220 | -| **Total** | **11,496** | +| **Total** | **11,502** | diff --git a/tests/tier_counts.json b/tests/tier_counts.json index 8067fc14..3b12917a 100644 --- a/tests/tier_counts.json +++ b/tests/tier_counts.json @@ -1,8 +1,8 @@ { "schema_version": 1, "counts": { - "unit": 6020, - "artifact": 3395, + "unit": 6025, + "artifact": 3396, "integration_psid": 1341, "reproduction_legacy": 520, "oracle_policyengine": 220 From 84c7d8e4429125069d13ba72ba37455c48487dc5 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 05:20:59 -0400 Subject: [PATCH 06/13] EPUF career fills: confirmation review fixes for #516 The score script checks the staged candidates against the manifest before leaving any marker and binds the cell and wage-base helpers; tests tie its pin to the manifest and check that it refuses another manifest, an unlocked gate and an existing result. fill_careers' docstring says which pre-career years it adds. The two nits left in the pinned fill module are recorded. Co-Authored-By: Claude Opus 5.5 --- .../gate_epuf_fill_candidates_registration.md | 15 ++++++++ scripts/score_epuf_fill_test.py | 20 +++++++--- .../cohorts/psid2010_epuf_fill.py | 7 ++-- tests/README-tiers.md | 4 +- tests/test_epuf_fill_candidates_manifest.py | 37 +++++++++++++++++++ tests/tier_counts.json | 2 +- 6 files changed, 73 insertions(+), 12 deletions(-) diff --git a/docs/amendments/gate_epuf_fill_candidates_registration.md b/docs/amendments/gate_epuf_fill_candidates_registration.md index 4efb7efe..45316077 100644 --- a/docs/amendments/gate_epuf_fill_candidates_registration.md +++ b/docs/amendments/gate_epuf_fill_candidates_registration.md @@ -165,6 +165,21 @@ application, which EPUF scoring does not use; the cache's key; the draws for persons of uncoded sex, who are never scored; and the bank summaries' years, which on EPUF end in 2006 anyway. +**The confirmation review** of both PRs +(`reviews/gate_epuf_fill_pr515_pr516_confirmation_review_20261004.md`) +returned APPROVE WITH NITS. Its fixes in unpinned files are applied: +- The score script checks the staged files against the manifest before it + leaves any marker. +- Tests tie the score script's pin to the manifest and check its refusals. +- A run with any injected input is marked as not the registered scoring. + +Two nits in the pinned `epuf_fill.py` are left as they are, because fixing +them would need a refit. Neither changes a registered draw: +- `_BANK_DIGESTS` keeps each digested donor fill alive for the process's + life. +- The zero-year forest has no empty-leaf check. An empty leaf would give a + zero probability. + ## Procedure on TEST (after lock) 1. Confirm that `gates.yaml` locks `gate_epuf_fill` and that the staged files diff --git a/scripts/score_epuf_fill_test.py b/scripts/score_epuf_fill_test.py index 79594ebc..6cf00885 100644 --- a/scripts/score_epuf_fill_test.py +++ b/scripts/score_epuf_fill_test.py @@ -40,6 +40,8 @@ CODE_FILES = ( "src/populace_dynamics/harness/epuf_fill_gate.py", "src/populace_dynamics/harness/epuf_fill_scoring.py", + "src/populace_dynamics/harness/epuf_cells.py", + "src/populace_dynamics/harness/epuf_operator.py", "src/populace_dynamics/estimates/epuf_fill.py", "scripts/score_epuf_fill_test.py", ) @@ -57,7 +59,7 @@ def candidates_from(manifest: dict, fills_dir: Path) -> dict: return spec -def main() -> None: +def main(argv: list[str] | None = None) -> None: parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) parser.add_argument("--manifest", type=Path, required=True) parser.add_argument("--output", type=Path, required=True) @@ -68,7 +70,7 @@ def main() -> None: os.environ.get("POPULACE_DYNAMICS_EPUF_FILLS_DIR", DEFAULT_DIR) ), ) - args = parser.parse_args() + args = parser.parse_args(argv) if args.output.exists(): raise FileExistsError(f"{args.output} exists; TEST is scored once") manifest_bytes = args.manifest.read_bytes() @@ -94,6 +96,15 @@ def main() -> None: raise g.TestPartLocked( "gate_epuf_fill is not locked; TEST stays unread" ) + # The staged candidates must be the registered bytes before any marker. + manifest = json.loads(manifest_bytes) + spec = candidates_from(manifest, args.fills_dir) + for family, roles in spec.items(): + for role, (path, sha256) in roles.items(): + if hashlib.sha256(path.read_bytes()).hexdigest() != sha256: + raise ValueError( + f"{family} {role}: {path.name} is not the registered bytes" + ) # A started marker, so a run that fails after reading TEST leaves a # trace of the read. marker = Path(f"{args.output}.started.json") @@ -107,10 +118,7 @@ def main() -> None: }, handle, ) - manifest = json.loads(manifest_bytes) - record = scoring.score_registered( - candidates_from(manifest, args.fills_dir) - ) + record = scoring.score_registered(spec) # Published paths are the registered file names, not local folders. for family in record["candidates"].values(): for role in family.values(): diff --git a/src/populace_dynamics/cohorts/psid2010_epuf_fill.py b/src/populace_dynamics/cohorts/psid2010_epuf_fill.py index 4937e672..35726600 100644 --- a/src/populace_dynamics/cohorts/psid2010_epuf_fill.py +++ b/src/populace_dynamics/cohorts/psid2010_epuf_fill.py @@ -18,9 +18,10 @@ - a gap year with no visible neighbour keeps the assembler's value, which used a neighbour the fill does not see (an observed pre-career year, or the 2014 boundary year); -- every year from 1951 before the career start becomes the pre-career - fill's draw, with provenance - :attr:`EPUFFillProvenance.PRE_CAREER_EPUF_DONOR`. +- every year from 1951 before the career start that has no career row + becomes the pre-career fill's draw, with provenance + :attr:`EPUFFillProvenance.PRE_CAREER_EPUF_DONOR` (the assembler's careers + start at the career start, so in practice every such year). The fills see what the gate's scoring path gives them: - capped shares of the wage base for the career years the PSID recorded; diff --git a/tests/README-tiers.md b/tests/README-tiers.md index ee7bf28c..6421150f 100644 --- a/tests/README-tiers.md +++ b/tests/README-tiers.md @@ -40,8 +40,8 @@ pytest --collect-only -q -m oracle_policyengine | tail -1 | Tier | Tests at HEAD | |---|---:| | `unit` | 6,025 | -| `artifact` | 3,396 | +| `artifact` | 3,398 | | `integration_psid` | 1,341 | | `reproduction_legacy` | 520 | | `oracle_policyengine` | 220 | -| **Total** | **11,502** | +| **Total** | **11,504** | diff --git a/tests/test_epuf_fill_candidates_manifest.py b/tests/test_epuf_fill_candidates_manifest.py index c4b2dae6..46f290d9 100644 --- a/tests/test_epuf_fill_candidates_manifest.py +++ b/tests/test_epuf_fill_candidates_manifest.py @@ -103,3 +103,40 @@ def test_the_registered_copula_binds_for_each_coded_sex(manifest): for sex in ("1", "2"): rho = diagnostics[sex]["rho"] assert any(value > 0 for value in rho[int(sex)]) + + +def _score_script(): + spec = importlib.util.spec_from_file_location( + "score_epuf_fill_test", ROOT / "scripts" / "score_epuf_fill_test.py" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_the_score_script_pins_this_manifest(): + script = _score_script() + assert script.REGISTERED_MANIFEST == "runs/epuf_fill_candidates_v1.json" + assert ( + script.REGISTERED_MANIFEST_SHA256 + == hashlib.sha256(MANIFEST.read_bytes()).hexdigest() + ) + + +def test_the_score_script_refuses_before_reading_test(tmp_path): + script = _score_script() + output = tmp_path / "result.json" + other = tmp_path / "other.json" + other.write_text("{}") + with pytest.raises(ValueError, match="not the registered"): + script.main(["--manifest", str(other), "--output", str(output)]) + # The registered manifest, but the gate is not locked in gates.yaml. + from populace_dynamics.harness import epuf_fill_gate as g + + with pytest.raises(g.TestPartLocked): + script.main(["--manifest", str(MANIFEST), "--output", str(output)]) + assert not output.exists() + assert not tmp_path.joinpath("result.json.started.json").exists() + output.write_text("{}") + with pytest.raises(FileExistsError, match="scored once"): + script.main(["--manifest", str(MANIFEST), "--output", str(output)]) diff --git a/tests/tier_counts.json b/tests/tier_counts.json index 3b12917a..8ec48733 100644 --- a/tests/tier_counts.json +++ b/tests/tier_counts.json @@ -2,7 +2,7 @@ "schema_version": 1, "counts": { "unit": 6025, - "artifact": 3396, + "artifact": 3398, "integration_psid": 1341, "reproduction_legacy": 520, "oracle_policyengine": 220 From b722382e2427cf84253daab96c206f875a8c0544 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 05:20:59 -0400 Subject: [PATCH 07/13] gate_epuf_fill: withdraw the candidate manifest again, to refit at a reachable commit Rebasing onto #515's fixes left the manifest's code commit unreachable from the branch. It is refitted at this commit; the artifacts' bytes are unchanged. The branch syncs with #515 by merge from here on, so the manifest's commit stays reachable. Co-Authored-By: Claude Opus 5.5 --- runs/epuf_fill_candidates_v1.json | 1404 -------------------- runs/epuf_fill_candidates_v1.json.env.json | 19 - 2 files changed, 1423 deletions(-) delete mode 100644 runs/epuf_fill_candidates_v1.json delete mode 100644 runs/epuf_fill_candidates_v1.json.env.json diff --git a/runs/epuf_fill_candidates_v1.json b/runs/epuf_fill_candidates_v1.json deleted file mode 100644 index ce2e3f49..00000000 --- a/runs/epuf_fill_candidates_v1.json +++ /dev/null @@ -1,1404 +0,0 @@ -{ - "schema": "populace_dynamics.epuf_fill_candidates.v1", - "registration_id": "2026-10-03-epuf-career-fill", - "code_commit": "598e44366621fff26469ec87f23ff4871215efdc", - "code_files_clean": true, - "built_at_utc": "2026-10-04T09:07:47+00:00", - "part": "train", - "n_persons": 2629944, - "versions": { - "numpy": "2.5.1", - "scipy": "1.18.0", - "scikit_learn": "1.9.0", - "python": "3.14.4", - "zlib_runtime": "1.2.12", - "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O" - }, - "staging": "files live outside the repository, like EPUF; a refit with this script at code_commit, on the same library, zlib and platform versions, reproduces their bytes", - "fills": { - "odd_forest": { - "family": "odd", - "role": "primary", - "params": { - "unit_years": [ - 1991, - 2005 - ], - "n_units": 3000000, - "n_trees": 10, - "min_leaf": 15, - "max_features": 0.8, - "seed": 0 - }, - "file": "odd_forest_v1.npz", - "sha256": "37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa", - "bytes": 44836853, - "fit_seconds": 67.5, - "diagnostics": { - "1": { - "n_units": 3000000, - "n_positive_units": 1267291, - "n_trees": 10, - "min_leaf": 15, - "n_leaves": 250386, - "n_nodes": 500762, - "calibration_persons": 136391, - "rho": [ - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ], - [ - 0.1, - 0.1, - 0.15, - 0.15, - 0.45, - 0.45 - ], - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ], - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ] - ], - "calibration": { - "1.1.rho_0.0": [ - -0.00937054964764883, - -0.006568577649590401 - ], - "1.2.rho_0.0": [ - -0.009185985434689847, - -0.017087922487923013 - ], - "1.3.rho_0.0": [ - -0.008760608986963514, - -0.005892891185897087 - ], - "1.4.rho_0.0": [ - -0.014822290100570679, - -0.0029128598220162782 - ], - "2.1.rho_0.0": [ - "nan", - "nan" - ], - "2.2.rho_0.0": [ - "nan", - "nan" - ], - "2.3.rho_0.0": [ - "nan", - "nan" - ], - "2.4.rho_0.0": [ - "nan", - "nan" - ], - "1.1.rho_0.05": [ - -0.0025221662657621824, - -0.0025405849711389594 - ], - "1.2.rho_0.05": [ - -0.005747487224877057, - -0.014480367835756236 - ], - "1.3.rho_0.05": [ - -0.006653418258712018, - -0.003659386315042923 - ], - "1.4.rho_0.05": [ - -0.013918461926527237, - -0.0014846818917918503 - ], - "2.1.rho_0.05": [ - "nan", - "nan" - ], - "2.2.rho_0.05": [ - "nan", - "nan" - ], - "2.3.rho_0.05": [ - "nan", - "nan" - ], - "2.4.rho_0.05": [ - "nan", - "nan" - ], - "1.1.rho_0.1": [ - 0.0022622655578613537, - 0.0017571621431731188 - ], - "1.2.rho_0.1": [ - -0.003463604263099551, - -0.013817366044770019 - ], - "1.3.rho_0.1": [ - -0.0043256050642532795, - -0.0020772149123101658 - ], - "1.4.rho_0.1": [ - -0.011090372488350764, - 0.0003620037980666124 - ], - "2.1.rho_0.1": [ - "nan", - "nan" - ], - "2.2.rho_0.1": [ - "nan", - "nan" - ], - "2.3.rho_0.1": [ - "nan", - "nan" - ], - "2.4.rho_0.1": [ - "nan", - "nan" - ], - "1.1.rho_0.15": [ - 0.00735065280082603, - 0.00547377504308344 - ], - "1.2.rho_0.15": [ - -0.00043591174742074745, - -0.011407436015852035 - ], - "1.3.rho_0.15": [ - -0.001469747919156772, - -6.807326989843876e-05 - ], - "1.4.rho_0.15": [ - -0.007842751587615049, - 0.0035570177498509548 - ], - "2.1.rho_0.15": [ - "nan", - "nan" - ], - "2.2.rho_0.15": [ - "nan", - "nan" - ], - "2.3.rho_0.15": [ - "nan", - "nan" - ], - "2.4.rho_0.15": [ - "nan", - "nan" - ], - "1.1.rho_0.2": [ - 0.01235436992080352, - 0.007278338434604126 - ], - "1.2.rho_0.2": [ - 0.0018633014396086667, - -0.00951255602154033 - ], - "1.3.rho_0.2": [ - 0.0014639774440903253, - 0.0017977312065096118 - ], - "1.4.rho_0.2": [ - -0.007672387545435644, - 0.005532053798539716 - ], - "2.1.rho_0.2": [ - "nan", - "nan" - ], - "2.2.rho_0.2": [ - "nan", - "nan" - ], - "2.3.rho_0.2": [ - "nan", - "nan" - ], - "2.4.rho_0.2": [ - "nan", - "nan" - ], - "1.1.rho_0.25": [ - 0.017532837584111616, - 0.012910343871726515 - ], - "1.2.rho_0.25": [ - 0.005432755081799634, - -0.0058967638868855365 - ], - "1.3.rho_0.25": [ - 0.004461578182951564, - 0.003245424028756938 - ], - "1.4.rho_0.25": [ - -0.006896856131266005, - 0.007402313522806625 - ], - "2.1.rho_0.25": [ - "nan", - "nan" - ], - "2.2.rho_0.25": [ - "nan", - "nan" - ], - "2.3.rho_0.25": [ - "nan", - "nan" - ], - "2.4.rho_0.25": [ - "nan", - "nan" - ], - "1.1.rho_0.3": [ - 0.02283322303498425, - 0.01782293559705972 - ], - "1.2.rho_0.3": [ - 0.007917659257822063, - -0.0042736292458098735 - ], - "1.3.rho_0.3": [ - 0.005912180793823385, - 0.004708972924462929 - ], - "1.4.rho_0.3": [ - -0.0037362954092261536, - 0.012169812222417309 - ], - "2.1.rho_0.3": [ - "nan", - "nan" - ], - "2.2.rho_0.3": [ - "nan", - "nan" - ], - "2.3.rho_0.3": [ - "nan", - "nan" - ], - "2.4.rho_0.3": [ - "nan", - "nan" - ], - "1.1.rho_0.35": [ - 0.02936744538086289, - 0.02269930584128066 - ], - "1.2.rho_0.35": [ - 0.010991221409968999, - -0.0020151785137914047 - ], - "1.3.rho_0.35": [ - 0.008154499506314195, - 0.006864335349464179 - ], - "1.4.rho_0.35": [ - -0.0033667339378640193, - 0.010716886425056416 - ], - "2.1.rho_0.35": [ - "nan", - "nan" - ], - "2.2.rho_0.35": [ - "nan", - "nan" - ], - "2.3.rho_0.35": [ - "nan", - "nan" - ], - "2.4.rho_0.35": [ - "nan", - "nan" - ], - "1.1.rho_0.4": [ - 0.033991025478408377, - 0.025266554928974005 - ], - "1.2.rho_0.4": [ - 0.01395682475791915, - 0.00013629350615307345 - ], - "1.3.rho_0.4": [ - 0.009812482470721196, - 0.008655649864332982 - ], - "1.4.rho_0.4": [ - -0.0018532086431369832, - 0.010892834474684587 - ], - "2.1.rho_0.4": [ - "nan", - "nan" - ], - "2.2.rho_0.4": [ - "nan", - "nan" - ], - "2.3.rho_0.4": [ - "nan", - "nan" - ], - "2.4.rho_0.4": [ - "nan", - "nan" - ], - "1.1.rho_0.45": [ - 0.04078840124406191, - 0.03125399303118248 - ], - "1.2.rho_0.45": [ - 0.016853239128019615, - 0.0023949314237127206 - ], - "1.3.rho_0.45": [ - 0.012218818061125014, - 0.0103161648513439 - ], - "1.4.rho_0.45": [ - 0.0007757976805343736, - 0.012061337583499476 - ], - "2.1.rho_0.45": [ - "nan", - "nan" - ], - "2.2.rho_0.45": [ - "nan", - "nan" - ], - "2.3.rho_0.45": [ - "nan", - "nan" - ], - "2.4.rho_0.45": [ - "nan", - "nan" - ], - "1.1.rho_0.5": [ - 0.04627958132301435, - 0.035137264874086305 - ], - "1.2.rho_0.5": [ - 0.020285078982263505, - 0.004963715627156029 - ], - "1.3.rho_0.5": [ - 0.014599779251665335, - 0.012138390969464785 - ], - "1.4.rho_0.5": [ - 0.004685831714231314, - 0.015445592554809928 - ], - "2.1.rho_0.5": [ - "nan", - "nan" - ], - "2.2.rho_0.5": [ - "nan", - "nan" - ], - "2.3.rho_0.5": [ - "nan", - "nan" - ], - "2.4.rho_0.5": [ - "nan", - "nan" - ], - "1.1.rho_0.55": [ - 0.0531305404913871, - 0.03958250535156038 - ], - "1.2.rho_0.55": [ - 0.023484969938003974, - 0.007224393712798816 - ], - "1.3.rho_0.55": [ - 0.016946622715777737, - 0.013757911329974615 - ], - "1.4.rho_0.55": [ - 0.005386466630163178, - 0.01558864988803843 - ], - "2.1.rho_0.55": [ - "nan", - "nan" - ], - "2.2.rho_0.55": [ - "nan", - "nan" - ], - "2.3.rho_0.55": [ - "nan", - "nan" - ], - "2.4.rho_0.55": [ - "nan", - "nan" - ], - "1.1.rho_0.6": [ - 0.0594398122408063, - 0.045308024035041305 - ], - "1.2.rho_0.6": [ - 0.02665635658052512, - 0.01009428430629955 - ], - "1.3.rho_0.6": [ - 0.020127026484403232, - 0.017144987639204246 - ], - "1.4.rho_0.6": [ - 0.0098509448536922, - 0.01789310382229714 - ], - "2.1.rho_0.6": [ - "nan", - "nan" - ], - "2.2.rho_0.6": [ - "nan", - "nan" - ], - "2.3.rho_0.6": [ - "nan", - "nan" - ], - "2.4.rho_0.6": [ - "nan", - "nan" - ], - "1.1.rho_0.65": [ - 0.06636452154965067, - 0.048690911284104854 - ], - "1.2.rho_0.65": [ - 0.02988279156747209, - 0.012593270355836461 - ], - "1.3.rho_0.65": [ - 0.023092885480960224, - 0.018772455583715764 - ], - "1.4.rho_0.65": [ - 0.01241442885124, - 0.02076244248410586 - ], - "2.1.rho_0.65": [ - "nan", - "nan" - ], - "2.2.rho_0.65": [ - "nan", - "nan" - ], - "2.3.rho_0.65": [ - "nan", - "nan" - ], - "2.4.rho_0.65": [ - "nan", - "nan" - ], - "1.1.rho_0.7": [ - 0.07254218457867379, - 0.0526910466051157 - ], - "1.2.rho_0.7": [ - 0.03329166905191294, - 0.015175105205589068 - ], - "1.3.rho_0.7": [ - 0.0258155301357077, - 0.020502762171670685 - ], - "1.4.rho_0.7": [ - 0.014453411163469543, - 0.020846671921446847 - ], - "2.1.rho_0.7": [ - "nan", - "nan" - ], - "2.2.rho_0.7": [ - "nan", - "nan" - ], - "2.3.rho_0.7": [ - "nan", - "nan" - ], - "2.4.rho_0.7": [ - "nan", - "nan" - ], - "1.1.rho_0.75": [ - 0.07916776188641705, - 0.057886291657737066 - ], - "1.2.rho_0.75": [ - 0.03725049246712253, - 0.018772837363772443 - ], - "1.3.rho_0.75": [ - 0.028609776086117367, - 0.02307257411267194 - ], - "1.4.rho_0.75": [ - 0.016869649042968726, - 0.02362923713906029 - ], - "2.1.rho_0.75": [ - "nan", - "nan" - ], - "2.2.rho_0.75": [ - "nan", - "nan" - ], - "2.3.rho_0.75": [ - "nan", - "nan" - ], - "2.4.rho_0.75": [ - "nan", - "nan" - ], - "1.1.rho_0.8": [ - 0.08680076817335436, - 0.06389736793273171 - ], - "1.2.rho_0.8": [ - 0.040963145105353815, - 0.021929454410539284 - ], - "1.3.rho_0.8": [ - 0.03147254961686874, - 0.02464394115084445 - ], - "1.4.rho_0.8": [ - 0.017611836623551924, - 0.02768136631318152 - ], - "2.1.rho_0.8": [ - "nan", - "nan" - ], - "2.2.rho_0.8": [ - "nan", - "nan" - ], - "2.3.rho_0.8": [ - "nan", - "nan" - ], - "2.4.rho_0.8": [ - "nan", - "nan" - ], - "1.1.rho_0.85": [ - 0.09307651453825694, - 0.06816742141440546 - ], - "1.2.rho_0.85": [ - 0.04423261009456436, - 0.024040513696466315 - ], - "1.3.rho_0.85": [ - 0.03373239871230438, - 0.02743872467269859 - ], - "1.4.rho_0.85": [ - 0.02157870721620181, - 0.02748553411253518 - ], - "2.1.rho_0.85": [ - "nan", - "nan" - ], - "2.2.rho_0.85": [ - "nan", - "nan" - ], - "2.3.rho_0.85": [ - "nan", - "nan" - ], - "2.4.rho_0.85": [ - "nan", - "nan" - ], - "1.1.rho_0.9": [ - 0.10107252485664509, - 0.07345061064036906 - ], - "1.2.rho_0.9": [ - 0.048178209491760327, - 0.026937093526634648 - ], - "1.3.rho_0.9": [ - 0.036002361151492135, - 0.02854935592244301 - ], - "1.4.rho_0.9": [ - 0.024308070051008657, - 0.02733002869966339 - ], - "2.1.rho_0.9": [ - "nan", - "nan" - ], - "2.2.rho_0.9": [ - "nan", - "nan" - ], - "2.3.rho_0.9": [ - "nan", - "nan" - ], - "2.4.rho_0.9": [ - "nan", - "nan" - ] - } - }, - "2": { - "n_units": 3000000, - "n_positive_units": 1215539, - "n_trees": 10, - "min_leaf": 15, - "n_leaves": 246831, - "n_nodes": 493652, - "calibration_persons": 126094, - "rho": [ - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ], - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ], - [ - 0.05, - 0.05, - 0.15, - 0.25, - 0.9, - 0.9 - ], - [ - 0.0, - 0.0, - 0.0, - 0.0, - 0.0, - 0.0 - ] - ], - "calibration": { - "1.1.rho_0.0": [ - "nan", - "nan" - ], - "1.2.rho_0.0": [ - "nan", - "nan" - ], - "1.3.rho_0.0": [ - "nan", - "nan" - ], - "1.4.rho_0.0": [ - "nan", - "nan" - ], - "2.1.rho_0.0": [ - -0.0007703488441346273, - -0.0010508879964993278 - ], - "2.2.rho_0.0": [ - -0.007493032484792828, - -0.013193098454176932 - ], - "2.3.rho_0.0": [ - -0.009429974516101503, - -0.01195790961172627 - ], - "2.4.rho_0.0": [ - -0.03692428228749378, - -0.046884412622328786 - ], - "1.1.rho_0.05": [ - "nan", - "nan" - ], - "1.2.rho_0.05": [ - "nan", - "nan" - ], - "1.3.rho_0.05": [ - "nan", - "nan" - ], - "1.4.rho_0.05": [ - "nan", - "nan" - ], - "2.1.rho_0.05": [ - 0.0011501760114371873, - 0.0002049050210158887 - ], - "2.2.rho_0.05": [ - -0.005131023108985944, - -0.01216928087935032 - ], - "2.3.rho_0.05": [ - -0.008520996492143773, - -0.010608354960477073 - ], - "2.4.rho_0.05": [ - -0.03473496584300617, - -0.05402554667776105 - ], - "1.1.rho_0.1": [ - "nan", - "nan" - ], - "1.2.rho_0.1": [ - "nan", - "nan" - ], - "1.3.rho_0.1": [ - "nan", - "nan" - ], - "1.4.rho_0.1": [ - "nan", - "nan" - ], - "2.1.rho_0.1": [ - 0.006927157675038598, - 0.004290292190204048 - ], - "2.2.rho_0.1": [ - -0.001703132719541478, - -0.010456836329726604 - ], - "2.3.rho_0.1": [ - -0.007321013146343036, - -0.009035871667865014 - ], - "2.4.rho_0.1": [ - -0.030882723546502233, - -0.048033202775634276 - ], - "1.1.rho_0.15": [ - "nan", - "nan" - ], - "1.2.rho_0.15": [ - "nan", - "nan" - ], - "1.3.rho_0.15": [ - "nan", - "nan" - ], - "1.4.rho_0.15": [ - "nan", - "nan" - ], - "2.1.rho_0.15": [ - 0.011751352091050937, - 0.008076268003208154 - ], - "2.2.rho_0.15": [ - 0.0020210219373677507, - -0.0070992294258820365 - ], - "2.3.rho_0.15": [ - -0.005528714504464016, - -0.0069653177500850205 - ], - "2.4.rho_0.15": [ - -0.029923398294732673, - -0.05002522943078225 - ], - "1.1.rho_0.2": [ - "nan", - "nan" - ], - "1.2.rho_0.2": [ - "nan", - "nan" - ], - "1.3.rho_0.2": [ - "nan", - "nan" - ], - "1.4.rho_0.2": [ - "nan", - "nan" - ], - "2.1.rho_0.2": [ - 0.016343948418381715, - 0.011187555046907938 - ], - "2.2.rho_0.2": [ - 0.0033473023560638415, - -0.0067290354973356115 - ], - "2.3.rho_0.2": [ - -0.002877760745891411, - -0.0033953050183980205 - ], - "2.4.rho_0.2": [ - -0.025096610145121545, - -0.05020372098724413 - ], - "1.1.rho_0.25": [ - "nan", - "nan" - ], - "1.2.rho_0.25": [ - "nan", - "nan" - ], - "1.3.rho_0.25": [ - "nan", - "nan" - ], - "1.4.rho_0.25": [ - "nan", - "nan" - ], - "2.1.rho_0.25": [ - 0.021425408842097315, - 0.01429463213074944 - ], - "2.2.rho_0.25": [ - 0.0065957040671158484, - -0.004705541238731792 - ], - "2.3.rho_0.25": [ - -1.7567476131130633e-05, - -0.0010814093051690898 - ], - "2.4.rho_0.25": [ - -0.0212615966241938, - -0.04842399617411386 - ], - "1.1.rho_0.3": [ - "nan", - "nan" - ], - "1.2.rho_0.3": [ - "nan", - "nan" - ], - "1.3.rho_0.3": [ - "nan", - "nan" - ], - "1.4.rho_0.3": [ - "nan", - "nan" - ], - "2.1.rho_0.3": [ - 0.02693682357503624, - 0.016009169813349877 - ], - "2.2.rho_0.3": [ - 0.009021593962986185, - -0.001825312400977941 - ], - "2.3.rho_0.3": [ - 0.002480943042680095, - -0.0003852668318783392 - ], - "2.4.rho_0.3": [ - -0.020630295377933372, - -0.05211604831280381 - ], - "1.1.rho_0.35": [ - "nan", - "nan" - ], - "1.2.rho_0.35": [ - "nan", - "nan" - ], - "1.3.rho_0.35": [ - "nan", - "nan" - ], - "1.4.rho_0.35": [ - "nan", - "nan" - ], - "2.1.rho_0.35": [ - 0.03296557024474411, - 0.021046225240840877 - ], - "2.2.rho_0.35": [ - 0.012335421323450668, - 0.0016580900689775468 - ], - "2.3.rho_0.35": [ - 0.004414993633125364, - 0.0012509622619388816 - ], - "2.4.rho_0.35": [ - -0.017200316022581097, - -0.04689517410403199 - ], - "1.1.rho_0.4": [ - "nan", - "nan" - ], - "1.2.rho_0.4": [ - "nan", - "nan" - ], - "1.3.rho_0.4": [ - "nan", - "nan" - ], - "1.4.rho_0.4": [ - "nan", - "nan" - ], - "2.1.rho_0.4": [ - 0.03881518696695174, - 0.02617586875204103 - ], - "2.2.rho_0.4": [ - 0.014916453205874203, - 0.0031898299704220534 - ], - "2.3.rho_0.4": [ - 0.006149846624828648, - 0.0020327028751055964 - ], - "2.4.rho_0.4": [ - -0.014400133887248034, - -0.04119061699894566 - ], - "1.1.rho_0.45": [ - "nan", - "nan" - ], - "1.2.rho_0.45": [ - "nan", - "nan" - ], - "1.3.rho_0.45": [ - "nan", - "nan" - ], - "1.4.rho_0.45": [ - "nan", - "nan" - ], - "2.1.rho_0.45": [ - 0.04408209328648782, - 0.03100375174528891 - ], - "2.2.rho_0.45": [ - 0.018276462063509857, - 0.0054696188221817765 - ], - "2.3.rho_0.45": [ - 0.0076902880826571485, - 0.00218766241834345 - ], - "2.4.rho_0.45": [ - -0.014908359815873351, - -0.0393019026587349 - ], - "1.1.rho_0.5": [ - "nan", - "nan" - ], - "1.2.rho_0.5": [ - "nan", - "nan" - ], - "1.3.rho_0.5": [ - "nan", - "nan" - ], - "1.4.rho_0.5": [ - "nan", - "nan" - ], - "2.1.rho_0.5": [ - 0.04970088094513325, - 0.034854865497334075 - ], - "2.2.rho_0.5": [ - 0.021724257795143087, - 0.008193822209299872 - ], - "2.3.rho_0.5": [ - 0.009745593525695817, - 0.003562541404541153 - ], - "2.4.rho_0.5": [ - -0.014383130743268246, - -0.04271338044982742 - ], - "1.1.rho_0.55": [ - "nan", - "nan" - ], - "1.2.rho_0.55": [ - "nan", - "nan" - ], - "1.3.rho_0.55": [ - "nan", - "nan" - ], - "1.4.rho_0.55": [ - "nan", - "nan" - ], - "2.1.rho_0.55": [ - 0.05502109102831798, - 0.03890967682059354 - ], - "2.2.rho_0.55": [ - 0.024839414066947896, - 0.010572170562098249 - ], - "2.3.rho_0.55": [ - 0.011772980929248611, - 0.003908378086523001 - ], - "2.4.rho_0.55": [ - -0.013131700928571188, - -0.04285851871646029 - ], - "1.1.rho_0.6": [ - "nan", - "nan" - ], - "1.2.rho_0.6": [ - "nan", - "nan" - ], - "1.3.rho_0.6": [ - "nan", - "nan" - ], - "1.4.rho_0.6": [ - "nan", - "nan" - ], - "2.1.rho_0.6": [ - 0.06088619981198029, - 0.0420376792891467 - ], - "2.2.rho_0.6": [ - 0.028468055929460223, - 0.012861481376605255 - ], - "2.3.rho_0.6": [ - 0.014264953161897465, - 0.006716731791151176 - ], - "2.4.rho_0.6": [ - -0.01411854107127919, - -0.04322951820182941 - ], - "1.1.rho_0.65": [ - "nan", - "nan" - ], - "1.2.rho_0.65": [ - "nan", - "nan" - ], - "1.3.rho_0.65": [ - "nan", - "nan" - ], - "1.4.rho_0.65": [ - "nan", - "nan" - ], - "2.1.rho_0.65": [ - 0.06682408268645768, - 0.04478633762350159 - ], - "2.2.rho_0.65": [ - 0.03150037859622068, - 0.014635098136048796 - ], - "2.3.rho_0.65": [ - 0.016061091785444348, - 0.008363861184702004 - ], - "2.4.rho_0.65": [ - -0.013183126314957772, - -0.04726560797289203 - ], - "1.1.rho_0.7": [ - "nan", - "nan" - ], - "1.2.rho_0.7": [ - "nan", - "nan" - ], - "1.3.rho_0.7": [ - "nan", - "nan" - ], - "1.4.rho_0.7": [ - "nan", - "nan" - ], - "2.1.rho_0.7": [ - 0.07410396740658753, - 0.05061327862615472 - ], - "2.2.rho_0.7": [ - 0.03457230831577063, - 0.01752794757556897 - ], - "2.3.rho_0.7": [ - 0.018780887727013362, - 0.009609081527960028 - ], - "2.4.rho_0.7": [ - -0.011927889063067187, - -0.04575175956846467 - ], - "1.1.rho_0.75": [ - "nan", - "nan" - ], - "1.2.rho_0.75": [ - "nan", - "nan" - ], - "1.3.rho_0.75": [ - "nan", - "nan" - ], - "1.4.rho_0.75": [ - "nan", - "nan" - ], - "2.1.rho_0.75": [ - 0.08092707240167418, - 0.05483242561629709 - ], - "2.2.rho_0.75": [ - 0.03658199157962372, - 0.018699837196579194 - ], - "2.3.rho_0.75": [ - 0.021946755597646472, - 0.011296140023272394 - ], - "2.4.rho_0.75": [ - -0.009788245012555041, - -0.045540955638491365 - ], - "1.1.rho_0.8": [ - "nan", - "nan" - ], - "1.2.rho_0.8": [ - "nan", - "nan" - ], - "1.3.rho_0.8": [ - "nan", - "nan" - ], - "1.4.rho_0.8": [ - "nan", - "nan" - ], - "2.1.rho_0.8": [ - 0.08743759706500098, - 0.060096830084552244 - ], - "2.2.rho_0.8": [ - 0.038972747910327676, - 0.01962918025075 - ], - "2.3.rho_0.8": [ - 0.02443160357398244, - 0.011703961899924842 - ], - "2.4.rho_0.8": [ - -0.008951498727275298, - -0.042132069352851076 - ], - "1.1.rho_0.85": [ - "nan", - "nan" - ], - "1.2.rho_0.85": [ - "nan", - "nan" - ], - "1.3.rho_0.85": [ - "nan", - "nan" - ], - "1.4.rho_0.85": [ - "nan", - "nan" - ], - "2.1.rho_0.85": [ - 0.09255511715216769, - 0.06287928620893612 - ], - "2.2.rho_0.85": [ - 0.04200109430794119, - 0.021762940153467025 - ], - "2.3.rho_0.85": [ - 0.027419910814330595, - 0.013140726589435658 - ], - "2.4.rho_0.85": [ - -0.0057698938150033685, - -0.03462993739531095 - ], - "1.1.rho_0.9": [ - "nan", - "nan" - ], - "1.2.rho_0.9": [ - "nan", - "nan" - ], - "1.3.rho_0.9": [ - "nan", - "nan" - ], - "1.4.rho_0.9": [ - "nan", - "nan" - ], - "2.1.rho_0.9": [ - 0.09982884654946289, - 0.06787532650906625 - ], - "2.2.rho_0.9": [ - 0.04549333428061564, - 0.02470270553781595 - ], - "2.3.rho_0.9": [ - 0.03002262611212725, - 0.014889240292517925 - ], - "2.4.rho_0.9": [ - -0.0024130894891942756, - -0.034578411611535076 - ] - } - } - } - }, - "odd_knn": { - "family": "odd", - "role": "alternative", - "params": { - "unit_years": [ - 1991, - 2005 - ], - "k": 10, - "seed": 0 - }, - "file": "odd_knn_v1.npz", - "sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", - "bytes": 4076580, - "fit_seconds": 6.2, - "diagnostics": { - "n_units": 27840831, - "bank": 1147199 - } - }, - "pre_chain": { - "family": "pre", - "role": "alternative", - "params": { - "unit_years": [ - 1951, - 2005 - ] - }, - "file": "pre_chain_v1.npz", - "sha256": "8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724", - "bytes": 236450, - "fit_seconds": 74.0, - "diagnostics": { - "n_units": 144646920 - } - }, - "pre_donor": { - "family": "pre", - "role": "primary", - "params": { - "k": 3, - "bank_size": 100000, - "birth_years": [ - 1905, - 1985 - ] - }, - "file": "pre_donor_v1.npz", - "sha256": "3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7", - "bytes": 31768107, - "fit_seconds": 2.9, - "diagnostics": { - "bank": 1518845 - } - } - }, - "elapsed_seconds": 182.1 -} diff --git a/runs/epuf_fill_candidates_v1.json.env.json b/runs/epuf_fill_candidates_v1.json.env.json deleted file mode 100644 index 7e9ba5df..00000000 --- a/runs/epuf_fill_candidates_v1.json.env.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "environment": { - "python": "3.14.4", - "numpy": "2.5.1", - "pandas": "3.0.3", - "sklearn": "1.9.0", - "scipy": "1.18.0", - "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O", - "fitting_stack": { - "populace_fit": "absent", - "populace_frame": "absent" - } - }, - "contract": { - "blob_sha": "b0c39af1e13a705f90b85d3e6b9a91e1d3c5485c", - "head_sha": "598e44366621fff26469ec87f23ff4871215efdc", - "path": "gates.yaml" - } -} From bddde85ea8ea7427f310fcb2791fec0fcbac7317 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 05:24:23 -0400 Subject: [PATCH 08/13] gate_epuf_fill: candidate manifest refitted at b722382e; pin updated The four artifacts' SHA-256 values are unchanged. The score script pins the new manifest. Co-Authored-By: Claude Opus 5.5 --- .../gate_epuf_fill_candidates_registration.md | 10 +- runs/epuf_fill_candidates_v1.json | 1404 +++++++++++++++++ runs/epuf_fill_candidates_v1.json.env.json | 19 + scripts/score_epuf_fill_test.py | 2 +- 4 files changed, 1430 insertions(+), 5 deletions(-) create mode 100644 runs/epuf_fill_candidates_v1.json create mode 100644 runs/epuf_fill_candidates_v1.json.env.json diff --git a/docs/amendments/gate_epuf_fill_candidates_registration.md b/docs/amendments/gate_epuf_fill_candidates_registration.md index 45316077..219175df 100644 --- a/docs/amendments/gate_epuf_fill_candidates_registration.md +++ b/docs/amendments/gate_epuf_fill_candidates_registration.md @@ -20,8 +20,8 @@ ## The registered artifacts -- **Manifest**: `runs/epuf_fill_candidates_v1.json`, SHA-256 `a304311343f3c78f7702ec6918b991b6bea30dad2a23529df6f0f975ce7f62d0`. -- **Fitted at**: `598e4436` on TRAIN, with the code files clean. +- **Manifest**: `runs/epuf_fill_candidates_v1.json`, SHA-256 `83d17a14f960033c7c0ed0d602ae395ea4b8d66ff5facab85115f493f3b96e2c`. +- **Fitted at**: `b722382e` on TRAIN, with the code files clean. - **Environment**: numpy 2.5.1, scipy 1.18.0, scikit-learn 1.9.0, Python 3.14.4, zlib 1.2.12, on macOS-26.6.2-arm64-arm-64bit-Mach-O. - **Reproducibility**: - A second fit at the same commit reproduced all four files byte for @@ -156,8 +156,10 @@ An independent code review of PR #516 returned REQUEST CHANGES was clean, and publishes no local paths. **The manifest.** The first manifest (SHA-256 `8d42153...`) was withdrawn -and refitted at the reviewed code (`598e4436`). The four artifacts' -SHA-256 values are unchanged. +and refitted at the reviewed code (`598e4436`, SHA-256 `a3043113...`). +Rebasing onto #515's fixes then left that commit off the branch, so the +manifest was refitted once more at `b722382e`. The four +artifacts' SHA-256 values are the same in all three manifests. **The DEV dry run still stands.** It scored the same artifacts, and none of the fixes changes a draw for a person the gate scores. They touch the PSID diff --git a/runs/epuf_fill_candidates_v1.json b/runs/epuf_fill_candidates_v1.json new file mode 100644 index 00000000..1504efd3 --- /dev/null +++ b/runs/epuf_fill_candidates_v1.json @@ -0,0 +1,1404 @@ +{ + "schema": "populace_dynamics.epuf_fill_candidates.v1", + "registration_id": "2026-10-03-epuf-career-fill", + "code_commit": "b722382e2427cf84253daab96c206f875a8c0544", + "code_files_clean": true, + "built_at_utc": "2026-10-04T09:21:00+00:00", + "part": "train", + "n_persons": 2629944, + "versions": { + "numpy": "2.5.1", + "scipy": "1.18.0", + "scikit_learn": "1.9.0", + "python": "3.14.4", + "zlib_runtime": "1.2.12", + "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O" + }, + "staging": "files live outside the repository, like EPUF; a refit with this script at code_commit, on the same library, zlib and platform versions, reproduces their bytes", + "fills": { + "odd_forest": { + "family": "odd", + "role": "primary", + "params": { + "unit_years": [ + 1991, + 2005 + ], + "n_units": 3000000, + "n_trees": 10, + "min_leaf": 15, + "max_features": 0.8, + "seed": 0 + }, + "file": "odd_forest_v1.npz", + "sha256": "37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa", + "bytes": 44836853, + "fit_seconds": 71.5, + "diagnostics": { + "1": { + "n_units": 3000000, + "n_positive_units": 1267291, + "n_trees": 10, + "min_leaf": 15, + "n_leaves": 250386, + "n_nodes": 500762, + "calibration_persons": 136391, + "rho": [ + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.1, + 0.1, + 0.15, + 0.15, + 0.45, + 0.45 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + ], + "calibration": { + "1.1.rho_0.0": [ + -0.00937054964764883, + -0.006568577649590401 + ], + "1.2.rho_0.0": [ + -0.009185985434689847, + -0.017087922487923013 + ], + "1.3.rho_0.0": [ + -0.008760608986963514, + -0.005892891185897087 + ], + "1.4.rho_0.0": [ + -0.014822290100570679, + -0.0029128598220162782 + ], + "2.1.rho_0.0": [ + "nan", + "nan" + ], + "2.2.rho_0.0": [ + "nan", + "nan" + ], + "2.3.rho_0.0": [ + "nan", + "nan" + ], + "2.4.rho_0.0": [ + "nan", + "nan" + ], + "1.1.rho_0.05": [ + -0.0025221662657621824, + -0.0025405849711389594 + ], + "1.2.rho_0.05": [ + -0.005747487224877057, + -0.014480367835756236 + ], + "1.3.rho_0.05": [ + -0.006653418258712018, + -0.003659386315042923 + ], + "1.4.rho_0.05": [ + -0.013918461926527237, + -0.0014846818917918503 + ], + "2.1.rho_0.05": [ + "nan", + "nan" + ], + "2.2.rho_0.05": [ + "nan", + "nan" + ], + "2.3.rho_0.05": [ + "nan", + "nan" + ], + "2.4.rho_0.05": [ + "nan", + "nan" + ], + "1.1.rho_0.1": [ + 0.0022622655578613537, + 0.0017571621431731188 + ], + "1.2.rho_0.1": [ + -0.003463604263099551, + -0.013817366044770019 + ], + "1.3.rho_0.1": [ + -0.0043256050642532795, + -0.0020772149123101658 + ], + "1.4.rho_0.1": [ + -0.011090372488350764, + 0.0003620037980666124 + ], + "2.1.rho_0.1": [ + "nan", + "nan" + ], + "2.2.rho_0.1": [ + "nan", + "nan" + ], + "2.3.rho_0.1": [ + "nan", + "nan" + ], + "2.4.rho_0.1": [ + "nan", + "nan" + ], + "1.1.rho_0.15": [ + 0.00735065280082603, + 0.00547377504308344 + ], + "1.2.rho_0.15": [ + -0.00043591174742074745, + -0.011407436015852035 + ], + "1.3.rho_0.15": [ + -0.001469747919156772, + -6.807326989843876e-05 + ], + "1.4.rho_0.15": [ + -0.007842751587615049, + 0.0035570177498509548 + ], + "2.1.rho_0.15": [ + "nan", + "nan" + ], + "2.2.rho_0.15": [ + "nan", + "nan" + ], + "2.3.rho_0.15": [ + "nan", + "nan" + ], + "2.4.rho_0.15": [ + "nan", + "nan" + ], + "1.1.rho_0.2": [ + 0.01235436992080352, + 0.007278338434604126 + ], + "1.2.rho_0.2": [ + 0.0018633014396086667, + -0.00951255602154033 + ], + "1.3.rho_0.2": [ + 0.0014639774440903253, + 0.0017977312065096118 + ], + "1.4.rho_0.2": [ + -0.007672387545435644, + 0.005532053798539716 + ], + "2.1.rho_0.2": [ + "nan", + "nan" + ], + "2.2.rho_0.2": [ + "nan", + "nan" + ], + "2.3.rho_0.2": [ + "nan", + "nan" + ], + "2.4.rho_0.2": [ + "nan", + "nan" + ], + "1.1.rho_0.25": [ + 0.017532837584111616, + 0.012910343871726515 + ], + "1.2.rho_0.25": [ + 0.005432755081799634, + -0.0058967638868855365 + ], + "1.3.rho_0.25": [ + 0.004461578182951564, + 0.003245424028756938 + ], + "1.4.rho_0.25": [ + -0.006896856131266005, + 0.007402313522806625 + ], + "2.1.rho_0.25": [ + "nan", + "nan" + ], + "2.2.rho_0.25": [ + "nan", + "nan" + ], + "2.3.rho_0.25": [ + "nan", + "nan" + ], + "2.4.rho_0.25": [ + "nan", + "nan" + ], + "1.1.rho_0.3": [ + 0.02283322303498425, + 0.01782293559705972 + ], + "1.2.rho_0.3": [ + 0.007917659257822063, + -0.0042736292458098735 + ], + "1.3.rho_0.3": [ + 0.005912180793823385, + 0.004708972924462929 + ], + "1.4.rho_0.3": [ + -0.0037362954092261536, + 0.012169812222417309 + ], + "2.1.rho_0.3": [ + "nan", + "nan" + ], + "2.2.rho_0.3": [ + "nan", + "nan" + ], + "2.3.rho_0.3": [ + "nan", + "nan" + ], + "2.4.rho_0.3": [ + "nan", + "nan" + ], + "1.1.rho_0.35": [ + 0.02936744538086289, + 0.02269930584128066 + ], + "1.2.rho_0.35": [ + 0.010991221409968999, + -0.0020151785137914047 + ], + "1.3.rho_0.35": [ + 0.008154499506314195, + 0.006864335349464179 + ], + "1.4.rho_0.35": [ + -0.0033667339378640193, + 0.010716886425056416 + ], + "2.1.rho_0.35": [ + "nan", + "nan" + ], + "2.2.rho_0.35": [ + "nan", + "nan" + ], + "2.3.rho_0.35": [ + "nan", + "nan" + ], + "2.4.rho_0.35": [ + "nan", + "nan" + ], + "1.1.rho_0.4": [ + 0.033991025478408377, + 0.025266554928974005 + ], + "1.2.rho_0.4": [ + 0.01395682475791915, + 0.00013629350615307345 + ], + "1.3.rho_0.4": [ + 0.009812482470721196, + 0.008655649864332982 + ], + "1.4.rho_0.4": [ + -0.0018532086431369832, + 0.010892834474684587 + ], + "2.1.rho_0.4": [ + "nan", + "nan" + ], + "2.2.rho_0.4": [ + "nan", + "nan" + ], + "2.3.rho_0.4": [ + "nan", + "nan" + ], + "2.4.rho_0.4": [ + "nan", + "nan" + ], + "1.1.rho_0.45": [ + 0.04078840124406191, + 0.03125399303118248 + ], + "1.2.rho_0.45": [ + 0.016853239128019615, + 0.0023949314237127206 + ], + "1.3.rho_0.45": [ + 0.012218818061125014, + 0.0103161648513439 + ], + "1.4.rho_0.45": [ + 0.0007757976805343736, + 0.012061337583499476 + ], + "2.1.rho_0.45": [ + "nan", + "nan" + ], + "2.2.rho_0.45": [ + "nan", + "nan" + ], + "2.3.rho_0.45": [ + "nan", + "nan" + ], + "2.4.rho_0.45": [ + "nan", + "nan" + ], + "1.1.rho_0.5": [ + 0.04627958132301435, + 0.035137264874086305 + ], + "1.2.rho_0.5": [ + 0.020285078982263505, + 0.004963715627156029 + ], + "1.3.rho_0.5": [ + 0.014599779251665335, + 0.012138390969464785 + ], + "1.4.rho_0.5": [ + 0.004685831714231314, + 0.015445592554809928 + ], + "2.1.rho_0.5": [ + "nan", + "nan" + ], + "2.2.rho_0.5": [ + "nan", + "nan" + ], + "2.3.rho_0.5": [ + "nan", + "nan" + ], + "2.4.rho_0.5": [ + "nan", + "nan" + ], + "1.1.rho_0.55": [ + 0.0531305404913871, + 0.03958250535156038 + ], + "1.2.rho_0.55": [ + 0.023484969938003974, + 0.007224393712798816 + ], + "1.3.rho_0.55": [ + 0.016946622715777737, + 0.013757911329974615 + ], + "1.4.rho_0.55": [ + 0.005386466630163178, + 0.01558864988803843 + ], + "2.1.rho_0.55": [ + "nan", + "nan" + ], + "2.2.rho_0.55": [ + "nan", + "nan" + ], + "2.3.rho_0.55": [ + "nan", + "nan" + ], + "2.4.rho_0.55": [ + "nan", + "nan" + ], + "1.1.rho_0.6": [ + 0.0594398122408063, + 0.045308024035041305 + ], + "1.2.rho_0.6": [ + 0.02665635658052512, + 0.01009428430629955 + ], + "1.3.rho_0.6": [ + 0.020127026484403232, + 0.017144987639204246 + ], + "1.4.rho_0.6": [ + 0.0098509448536922, + 0.01789310382229714 + ], + "2.1.rho_0.6": [ + "nan", + "nan" + ], + "2.2.rho_0.6": [ + "nan", + "nan" + ], + "2.3.rho_0.6": [ + "nan", + "nan" + ], + "2.4.rho_0.6": [ + "nan", + "nan" + ], + "1.1.rho_0.65": [ + 0.06636452154965067, + 0.048690911284104854 + ], + "1.2.rho_0.65": [ + 0.02988279156747209, + 0.012593270355836461 + ], + "1.3.rho_0.65": [ + 0.023092885480960224, + 0.018772455583715764 + ], + "1.4.rho_0.65": [ + 0.01241442885124, + 0.02076244248410586 + ], + "2.1.rho_0.65": [ + "nan", + "nan" + ], + "2.2.rho_0.65": [ + "nan", + "nan" + ], + "2.3.rho_0.65": [ + "nan", + "nan" + ], + "2.4.rho_0.65": [ + "nan", + "nan" + ], + "1.1.rho_0.7": [ + 0.07254218457867379, + 0.0526910466051157 + ], + "1.2.rho_0.7": [ + 0.03329166905191294, + 0.015175105205589068 + ], + "1.3.rho_0.7": [ + 0.0258155301357077, + 0.020502762171670685 + ], + "1.4.rho_0.7": [ + 0.014453411163469543, + 0.020846671921446847 + ], + "2.1.rho_0.7": [ + "nan", + "nan" + ], + "2.2.rho_0.7": [ + "nan", + "nan" + ], + "2.3.rho_0.7": [ + "nan", + "nan" + ], + "2.4.rho_0.7": [ + "nan", + "nan" + ], + "1.1.rho_0.75": [ + 0.07916776188641705, + 0.057886291657737066 + ], + "1.2.rho_0.75": [ + 0.03725049246712253, + 0.018772837363772443 + ], + "1.3.rho_0.75": [ + 0.028609776086117367, + 0.02307257411267194 + ], + "1.4.rho_0.75": [ + 0.016869649042968726, + 0.02362923713906029 + ], + "2.1.rho_0.75": [ + "nan", + "nan" + ], + "2.2.rho_0.75": [ + "nan", + "nan" + ], + "2.3.rho_0.75": [ + "nan", + "nan" + ], + "2.4.rho_0.75": [ + "nan", + "nan" + ], + "1.1.rho_0.8": [ + 0.08680076817335436, + 0.06389736793273171 + ], + "1.2.rho_0.8": [ + 0.040963145105353815, + 0.021929454410539284 + ], + "1.3.rho_0.8": [ + 0.03147254961686874, + 0.02464394115084445 + ], + "1.4.rho_0.8": [ + 0.017611836623551924, + 0.02768136631318152 + ], + "2.1.rho_0.8": [ + "nan", + "nan" + ], + "2.2.rho_0.8": [ + "nan", + "nan" + ], + "2.3.rho_0.8": [ + "nan", + "nan" + ], + "2.4.rho_0.8": [ + "nan", + "nan" + ], + "1.1.rho_0.85": [ + 0.09307651453825694, + 0.06816742141440546 + ], + "1.2.rho_0.85": [ + 0.04423261009456436, + 0.024040513696466315 + ], + "1.3.rho_0.85": [ + 0.03373239871230438, + 0.02743872467269859 + ], + "1.4.rho_0.85": [ + 0.02157870721620181, + 0.02748553411253518 + ], + "2.1.rho_0.85": [ + "nan", + "nan" + ], + "2.2.rho_0.85": [ + "nan", + "nan" + ], + "2.3.rho_0.85": [ + "nan", + "nan" + ], + "2.4.rho_0.85": [ + "nan", + "nan" + ], + "1.1.rho_0.9": [ + 0.10107252485664509, + 0.07345061064036906 + ], + "1.2.rho_0.9": [ + 0.048178209491760327, + 0.026937093526634648 + ], + "1.3.rho_0.9": [ + 0.036002361151492135, + 0.02854935592244301 + ], + "1.4.rho_0.9": [ + 0.024308070051008657, + 0.02733002869966339 + ], + "2.1.rho_0.9": [ + "nan", + "nan" + ], + "2.2.rho_0.9": [ + "nan", + "nan" + ], + "2.3.rho_0.9": [ + "nan", + "nan" + ], + "2.4.rho_0.9": [ + "nan", + "nan" + ] + } + }, + "2": { + "n_units": 3000000, + "n_positive_units": 1215539, + "n_trees": 10, + "min_leaf": 15, + "n_leaves": 246831, + "n_nodes": 493652, + "calibration_persons": 126094, + "rho": [ + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.05, + 0.05, + 0.15, + 0.25, + 0.9, + 0.9 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + ], + "calibration": { + "1.1.rho_0.0": [ + "nan", + "nan" + ], + "1.2.rho_0.0": [ + "nan", + "nan" + ], + "1.3.rho_0.0": [ + "nan", + "nan" + ], + "1.4.rho_0.0": [ + "nan", + "nan" + ], + "2.1.rho_0.0": [ + -0.0007703488441346273, + -0.0010508879964993278 + ], + "2.2.rho_0.0": [ + -0.007493032484792828, + -0.013193098454176932 + ], + "2.3.rho_0.0": [ + -0.009429974516101503, + -0.01195790961172627 + ], + "2.4.rho_0.0": [ + -0.03692428228749378, + -0.046884412622328786 + ], + "1.1.rho_0.05": [ + "nan", + "nan" + ], + "1.2.rho_0.05": [ + "nan", + "nan" + ], + "1.3.rho_0.05": [ + "nan", + "nan" + ], + "1.4.rho_0.05": [ + "nan", + "nan" + ], + "2.1.rho_0.05": [ + 0.0011501760114371873, + 0.0002049050210158887 + ], + "2.2.rho_0.05": [ + -0.005131023108985944, + -0.01216928087935032 + ], + "2.3.rho_0.05": [ + -0.008520996492143773, + -0.010608354960477073 + ], + "2.4.rho_0.05": [ + -0.03473496584300617, + -0.05402554667776105 + ], + "1.1.rho_0.1": [ + "nan", + "nan" + ], + "1.2.rho_0.1": [ + "nan", + "nan" + ], + "1.3.rho_0.1": [ + "nan", + "nan" + ], + "1.4.rho_0.1": [ + "nan", + "nan" + ], + "2.1.rho_0.1": [ + 0.006927157675038598, + 0.004290292190204048 + ], + "2.2.rho_0.1": [ + -0.001703132719541478, + -0.010456836329726604 + ], + "2.3.rho_0.1": [ + -0.007321013146343036, + -0.009035871667865014 + ], + "2.4.rho_0.1": [ + -0.030882723546502233, + -0.048033202775634276 + ], + "1.1.rho_0.15": [ + "nan", + "nan" + ], + "1.2.rho_0.15": [ + "nan", + "nan" + ], + "1.3.rho_0.15": [ + "nan", + "nan" + ], + "1.4.rho_0.15": [ + "nan", + "nan" + ], + "2.1.rho_0.15": [ + 0.011751352091050937, + 0.008076268003208154 + ], + "2.2.rho_0.15": [ + 0.0020210219373677507, + -0.0070992294258820365 + ], + "2.3.rho_0.15": [ + -0.005528714504464016, + -0.0069653177500850205 + ], + "2.4.rho_0.15": [ + -0.029923398294732673, + -0.05002522943078225 + ], + "1.1.rho_0.2": [ + "nan", + "nan" + ], + "1.2.rho_0.2": [ + "nan", + "nan" + ], + "1.3.rho_0.2": [ + "nan", + "nan" + ], + "1.4.rho_0.2": [ + "nan", + "nan" + ], + "2.1.rho_0.2": [ + 0.016343948418381715, + 0.011187555046907938 + ], + "2.2.rho_0.2": [ + 0.0033473023560638415, + -0.0067290354973356115 + ], + "2.3.rho_0.2": [ + -0.002877760745891411, + -0.0033953050183980205 + ], + "2.4.rho_0.2": [ + -0.025096610145121545, + -0.05020372098724413 + ], + "1.1.rho_0.25": [ + "nan", + "nan" + ], + "1.2.rho_0.25": [ + "nan", + "nan" + ], + "1.3.rho_0.25": [ + "nan", + "nan" + ], + "1.4.rho_0.25": [ + "nan", + "nan" + ], + "2.1.rho_0.25": [ + 0.021425408842097315, + 0.01429463213074944 + ], + "2.2.rho_0.25": [ + 0.0065957040671158484, + -0.004705541238731792 + ], + "2.3.rho_0.25": [ + -1.7567476131130633e-05, + -0.0010814093051690898 + ], + "2.4.rho_0.25": [ + -0.0212615966241938, + -0.04842399617411386 + ], + "1.1.rho_0.3": [ + "nan", + "nan" + ], + "1.2.rho_0.3": [ + "nan", + "nan" + ], + "1.3.rho_0.3": [ + "nan", + "nan" + ], + "1.4.rho_0.3": [ + "nan", + "nan" + ], + "2.1.rho_0.3": [ + 0.02693682357503624, + 0.016009169813349877 + ], + "2.2.rho_0.3": [ + 0.009021593962986185, + -0.001825312400977941 + ], + "2.3.rho_0.3": [ + 0.002480943042680095, + -0.0003852668318783392 + ], + "2.4.rho_0.3": [ + -0.020630295377933372, + -0.05211604831280381 + ], + "1.1.rho_0.35": [ + "nan", + "nan" + ], + "1.2.rho_0.35": [ + "nan", + "nan" + ], + "1.3.rho_0.35": [ + "nan", + "nan" + ], + "1.4.rho_0.35": [ + "nan", + "nan" + ], + "2.1.rho_0.35": [ + 0.03296557024474411, + 0.021046225240840877 + ], + "2.2.rho_0.35": [ + 0.012335421323450668, + 0.0016580900689775468 + ], + "2.3.rho_0.35": [ + 0.004414993633125364, + 0.0012509622619388816 + ], + "2.4.rho_0.35": [ + -0.017200316022581097, + -0.04689517410403199 + ], + "1.1.rho_0.4": [ + "nan", + "nan" + ], + "1.2.rho_0.4": [ + "nan", + "nan" + ], + "1.3.rho_0.4": [ + "nan", + "nan" + ], + "1.4.rho_0.4": [ + "nan", + "nan" + ], + "2.1.rho_0.4": [ + 0.03881518696695174, + 0.02617586875204103 + ], + "2.2.rho_0.4": [ + 0.014916453205874203, + 0.0031898299704220534 + ], + "2.3.rho_0.4": [ + 0.006149846624828648, + 0.0020327028751055964 + ], + "2.4.rho_0.4": [ + -0.014400133887248034, + -0.04119061699894566 + ], + "1.1.rho_0.45": [ + "nan", + "nan" + ], + "1.2.rho_0.45": [ + "nan", + "nan" + ], + "1.3.rho_0.45": [ + "nan", + "nan" + ], + "1.4.rho_0.45": [ + "nan", + "nan" + ], + "2.1.rho_0.45": [ + 0.04408209328648782, + 0.03100375174528891 + ], + "2.2.rho_0.45": [ + 0.018276462063509857, + 0.0054696188221817765 + ], + "2.3.rho_0.45": [ + 0.0076902880826571485, + 0.00218766241834345 + ], + "2.4.rho_0.45": [ + -0.014908359815873351, + -0.0393019026587349 + ], + "1.1.rho_0.5": [ + "nan", + "nan" + ], + "1.2.rho_0.5": [ + "nan", + "nan" + ], + "1.3.rho_0.5": [ + "nan", + "nan" + ], + "1.4.rho_0.5": [ + "nan", + "nan" + ], + "2.1.rho_0.5": [ + 0.04970088094513325, + 0.034854865497334075 + ], + "2.2.rho_0.5": [ + 0.021724257795143087, + 0.008193822209299872 + ], + "2.3.rho_0.5": [ + 0.009745593525695817, + 0.003562541404541153 + ], + "2.4.rho_0.5": [ + -0.014383130743268246, + -0.04271338044982742 + ], + "1.1.rho_0.55": [ + "nan", + "nan" + ], + "1.2.rho_0.55": [ + "nan", + "nan" + ], + "1.3.rho_0.55": [ + "nan", + "nan" + ], + "1.4.rho_0.55": [ + "nan", + "nan" + ], + "2.1.rho_0.55": [ + 0.05502109102831798, + 0.03890967682059354 + ], + "2.2.rho_0.55": [ + 0.024839414066947896, + 0.010572170562098249 + ], + "2.3.rho_0.55": [ + 0.011772980929248611, + 0.003908378086523001 + ], + "2.4.rho_0.55": [ + -0.013131700928571188, + -0.04285851871646029 + ], + "1.1.rho_0.6": [ + "nan", + "nan" + ], + "1.2.rho_0.6": [ + "nan", + "nan" + ], + "1.3.rho_0.6": [ + "nan", + "nan" + ], + "1.4.rho_0.6": [ + "nan", + "nan" + ], + "2.1.rho_0.6": [ + 0.06088619981198029, + 0.0420376792891467 + ], + "2.2.rho_0.6": [ + 0.028468055929460223, + 0.012861481376605255 + ], + "2.3.rho_0.6": [ + 0.014264953161897465, + 0.006716731791151176 + ], + "2.4.rho_0.6": [ + -0.01411854107127919, + -0.04322951820182941 + ], + "1.1.rho_0.65": [ + "nan", + "nan" + ], + "1.2.rho_0.65": [ + "nan", + "nan" + ], + "1.3.rho_0.65": [ + "nan", + "nan" + ], + "1.4.rho_0.65": [ + "nan", + "nan" + ], + "2.1.rho_0.65": [ + 0.06682408268645768, + 0.04478633762350159 + ], + "2.2.rho_0.65": [ + 0.03150037859622068, + 0.014635098136048796 + ], + "2.3.rho_0.65": [ + 0.016061091785444348, + 0.008363861184702004 + ], + "2.4.rho_0.65": [ + -0.013183126314957772, + -0.04726560797289203 + ], + "1.1.rho_0.7": [ + "nan", + "nan" + ], + "1.2.rho_0.7": [ + "nan", + "nan" + ], + "1.3.rho_0.7": [ + "nan", + "nan" + ], + "1.4.rho_0.7": [ + "nan", + "nan" + ], + "2.1.rho_0.7": [ + 0.07410396740658753, + 0.05061327862615472 + ], + "2.2.rho_0.7": [ + 0.03457230831577063, + 0.01752794757556897 + ], + "2.3.rho_0.7": [ + 0.018780887727013362, + 0.009609081527960028 + ], + "2.4.rho_0.7": [ + -0.011927889063067187, + -0.04575175956846467 + ], + "1.1.rho_0.75": [ + "nan", + "nan" + ], + "1.2.rho_0.75": [ + "nan", + "nan" + ], + "1.3.rho_0.75": [ + "nan", + "nan" + ], + "1.4.rho_0.75": [ + "nan", + "nan" + ], + "2.1.rho_0.75": [ + 0.08092707240167418, + 0.05483242561629709 + ], + "2.2.rho_0.75": [ + 0.03658199157962372, + 0.018699837196579194 + ], + "2.3.rho_0.75": [ + 0.021946755597646472, + 0.011296140023272394 + ], + "2.4.rho_0.75": [ + -0.009788245012555041, + -0.045540955638491365 + ], + "1.1.rho_0.8": [ + "nan", + "nan" + ], + "1.2.rho_0.8": [ + "nan", + "nan" + ], + "1.3.rho_0.8": [ + "nan", + "nan" + ], + "1.4.rho_0.8": [ + "nan", + "nan" + ], + "2.1.rho_0.8": [ + 0.08743759706500098, + 0.060096830084552244 + ], + "2.2.rho_0.8": [ + 0.038972747910327676, + 0.01962918025075 + ], + "2.3.rho_0.8": [ + 0.02443160357398244, + 0.011703961899924842 + ], + "2.4.rho_0.8": [ + -0.008951498727275298, + -0.042132069352851076 + ], + "1.1.rho_0.85": [ + "nan", + "nan" + ], + "1.2.rho_0.85": [ + "nan", + "nan" + ], + "1.3.rho_0.85": [ + "nan", + "nan" + ], + "1.4.rho_0.85": [ + "nan", + "nan" + ], + "2.1.rho_0.85": [ + 0.09255511715216769, + 0.06287928620893612 + ], + "2.2.rho_0.85": [ + 0.04200109430794119, + 0.021762940153467025 + ], + "2.3.rho_0.85": [ + 0.027419910814330595, + 0.013140726589435658 + ], + "2.4.rho_0.85": [ + -0.0057698938150033685, + -0.03462993739531095 + ], + "1.1.rho_0.9": [ + "nan", + "nan" + ], + "1.2.rho_0.9": [ + "nan", + "nan" + ], + "1.3.rho_0.9": [ + "nan", + "nan" + ], + "1.4.rho_0.9": [ + "nan", + "nan" + ], + "2.1.rho_0.9": [ + 0.09982884654946289, + 0.06787532650906625 + ], + "2.2.rho_0.9": [ + 0.04549333428061564, + 0.02470270553781595 + ], + "2.3.rho_0.9": [ + 0.03002262611212725, + 0.014889240292517925 + ], + "2.4.rho_0.9": [ + -0.0024130894891942756, + -0.034578411611535076 + ] + } + } + } + }, + "odd_knn": { + "family": "odd", + "role": "alternative", + "params": { + "unit_years": [ + 1991, + 2005 + ], + "k": 10, + "seed": 0 + }, + "file": "odd_knn_v1.npz", + "sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", + "bytes": 4076580, + "fit_seconds": 5.8, + "diagnostics": { + "n_units": 27840831, + "bank": 1147199 + } + }, + "pre_chain": { + "family": "pre", + "role": "alternative", + "params": { + "unit_years": [ + 1951, + 2005 + ] + }, + "file": "pre_chain_v1.npz", + "sha256": "8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724", + "bytes": 236450, + "fit_seconds": 73.6, + "diagnostics": { + "n_units": 144646920 + } + }, + "pre_donor": { + "family": "pre", + "role": "primary", + "params": { + "k": 3, + "bank_size": 100000, + "birth_years": [ + 1905, + 1985 + ] + }, + "file": "pre_donor_v1.npz", + "sha256": "3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7", + "bytes": 31768107, + "fit_seconds": 2.9, + "diagnostics": { + "bank": 1518845 + } + } + }, + "elapsed_seconds": 182.1 +} diff --git a/runs/epuf_fill_candidates_v1.json.env.json b/runs/epuf_fill_candidates_v1.json.env.json new file mode 100644 index 00000000..a3f2cfbe --- /dev/null +++ b/runs/epuf_fill_candidates_v1.json.env.json @@ -0,0 +1,19 @@ +{ + "environment": { + "python": "3.14.4", + "numpy": "2.5.1", + "pandas": "3.0.3", + "sklearn": "1.9.0", + "scipy": "1.18.0", + "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O", + "fitting_stack": { + "populace_fit": "absent", + "populace_frame": "absent" + } + }, + "contract": { + "blob_sha": "b0c39af1e13a705f90b85d3e6b9a91e1d3c5485c", + "head_sha": "b722382e2427cf84253daab96c206f875a8c0544", + "path": "gates.yaml" + } +} diff --git a/scripts/score_epuf_fill_test.py b/scripts/score_epuf_fill_test.py index 6cf00885..f61434a2 100644 --- a/scripts/score_epuf_fill_test.py +++ b/scripts/score_epuf_fill_test.py @@ -34,7 +34,7 @@ #: The registered manifest; any other manifest is refused. REGISTERED_MANIFEST = "runs/epuf_fill_candidates_v1.json" REGISTERED_MANIFEST_SHA256 = ( - "a304311343f3c78f7702ec6918b991b6bea30dad2a23529df6f0f975ce7f62d0" + "83d17a14f960033c7c0ed0d602ae395ea4b8d66ff5facab85115f493f3b96e2c" ) #: Files whose state the record reports. CODE_FILES = ( From 7de0644d4373d9fc22df8f67d9205bd2076483e5 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 05:30:48 -0400 Subject: [PATCH 09/13] EPUF career fills: the score-script refusal test can never reach TEST Delta review of #516 (D1): once gates.yaml locks the gate, the refusal test would have passed the lock check and read TEST. It now patches the lock status, points --fills-dir at a test folder, and replaces test_part with a function that fails the test if reached. It also covers the missing case: gate locked, a staged file with other bytes, refused before any .started.json marker. Test-only; no pinned file changes and no refit. Co-Authored-By: Claude Opus 5.5 --- tests/test_epuf_fill_candidates_manifest.py | 46 +++++++++++++++++---- 1 file changed, 39 insertions(+), 7 deletions(-) diff --git a/tests/test_epuf_fill_candidates_manifest.py b/tests/test_epuf_fill_candidates_manifest.py index 46f290d9..7dbf7c83 100644 --- a/tests/test_epuf_fill_candidates_manifest.py +++ b/tests/test_epuf_fill_candidates_manifest.py @@ -123,20 +123,52 @@ def test_the_score_script_pins_this_manifest(): ) -def test_the_score_script_refuses_before_reading_test(tmp_path): +def test_the_score_script_refuses_before_reading_test(tmp_path, monkeypatch): + """No path here can reach TEST, even after the gate locks. + + The lock status is patched, and the fills folder is an empty or a + deliberately wrong one, so ``test_part`` is never called. + """ + script = _score_script() output = tmp_path / "result.json" + empty = tmp_path / "fills" + empty.mkdir() other = tmp_path / "other.json" other.write_text("{}") + base = ["--output", str(output), "--fills-dir", str(empty)] with pytest.raises(ValueError, match="not the registered"): - script.main(["--manifest", str(other), "--output", str(output)]) - # The registered manifest, but the gate is not locked in gates.yaml. - from populace_dynamics.harness import epuf_fill_gate as g + script.main(["--manifest", str(other), *base]) - with pytest.raises(g.TestPartLocked): - script.main(["--manifest", str(MANIFEST), "--output", str(output)]) + def reached_test(**_): + raise AssertionError("the score script reached test_part") + + monkeypatch.setattr(script.scoring.g, "test_part", reached_test) + monkeypatch.setattr( + script.g, + "_gate_lock_status", + lambda _: {"locked": False, "registration_id": None}, + ) + with pytest.raises(script.g.TestPartLocked): + script.main(["--manifest", str(MANIFEST), *base]) assert not output.exists() assert not tmp_path.joinpath("result.json.started.json").exists() + # Locked, but a staged file is not the registered bytes: refused before + # any marker or read. + monkeypatch.setattr( + script.g, + "_gate_lock_status", + lambda _: { + "locked": True, + "registration_id": script.g.REGISTRATION_ID, + }, + ) + manifest = json.loads(MANIFEST.read_text()) + for record in manifest["fills"].values(): + (empty / record["file"]).write_bytes(b"not the registered bytes") + with pytest.raises(ValueError, match="not the registered bytes"): + script.main(["--manifest", str(MANIFEST), *base]) + assert not tmp_path.joinpath("result.json.started.json").exists() output.write_text("{}") with pytest.raises(FileExistsError, match="scored once"): - script.main(["--manifest", str(MANIFEST), "--output", str(output)]) + script.main(["--manifest", str(MANIFEST), *base]) From 881f0377bad83bcce6fbb640f2d4ea5bddf5c10f Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 06:08:04 -0400 Subject: [PATCH 10/13] EPUF career fills: keep estimates/epuf_fill.py off the first-estimates surface CI shard 1 on #516 failed test__estimator_surface__pins_complete_module_tuple: the test globs estimates/*.py and expects every module that is not a named exclusion to be on the registered first-estimates surface. The fills module is opt-in (only the gate's scoring and its fit script load it), so it gets its own named exclusion, like the COLA tabulation and Track U modules. coordinator._ESTIMATOR_SURFACE_SOURCES is unchanged. Test-only; no pinned file changes and no refit. Co-Authored-By: Claude Opus 5.5 --- tests/estimates/test_coordinator.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tests/estimates/test_coordinator.py b/tests/estimates/test_coordinator.py index da1ef010..e23f211a 100644 --- a/tests/estimates/test_coordinator.py +++ b/tests/estimates/test_coordinator.py @@ -1538,12 +1538,19 @@ def test__estimator_surface__pins_complete_module_tuple(): for path in observed if path.name in ("adjusted_poverty.py", "uniform_cut_tabulation.py") ) + # The EPUF-learned career fills (gate_epuf_fill) are opt-in and outside + # the registered first-estimates surface: the gate's scoring and its fit + # script load them, and nothing on the first-estimates path imports them. + epuf_fill_surface = tuple( + path for path in observed if path.name == "epuf_fill.py" + ) first_estimates_surface = tuple( path for path in observed if path not in context_surface and path not in tabulation_surface and path not in track_u_surface + and path not in epuf_fill_surface ) assert coordinator._ESTIMATOR_SURFACE_SOURCES == expected @@ -1555,6 +1562,9 @@ def test__estimator_surface__pins_complete_module_tuple(): Path("src/populace_dynamics/estimates/adjusted_poverty.py"), Path("src/populace_dynamics/estimates/uniform_cut_tabulation.py"), ) + assert epuf_fill_surface == ( + Path("src/populace_dynamics/estimates/epuf_fill.py"), + ) assert context_surface == ( Path("src/populace_dynamics/estimates/anchor_context_coordinator.py"), Path("src/populace_dynamics/estimates/anchor_context_publication.py"), From 28428900faff77888f7331772c1f78b71b04af70 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 13:50:51 -0400 Subject: [PATCH 11/13] gate_epuf_fill candidates: describe pre_chain exactly; note the exploratory QRF follow-up The pre alternative is a one-step binned conditional-quantile chain, not a QRF: each year conditions on one later share (22 bins), sex and age. The document now says so where it describes the candidate and its DEV result, and the TEST procedure requires that wording, so its result is not read as a QRF result. Its five worst DEV cells are youth levels and zero shares, which is what the text reports. Adds an exploratory follow-up after TEST: a full-career microcosm-fit QRF fill reported beside the registered results. It is not a candidate and cannot change a tier or adoption. Doc-only; no registered rule, parameter or pinned file changes. Mechanism claims checked against epuf_fill.py by an independent reviewer. Co-Authored-By: Claude Opus 5.5 --- .../gate_epuf_fill_candidates_registration.md | 32 +++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/docs/amendments/gate_epuf_fill_candidates_registration.md b/docs/amendments/gate_epuf_fill_candidates_registration.md index 219175df..9f2149f4 100644 --- a/docs/amendments/gate_epuf_fill_candidates_registration.md +++ b/docs/amendments/gate_epuf_fill_candidates_registration.md @@ -73,6 +73,20 @@ nearest TRAIN donors of the same sex and birth year. **`pre_chain`** (pre alternative) draws year `y` from the next known later year's share, sex and age, backward from the career start. Ages 15-24 are single years. +- **What it is**: a one-step binned conditional-quantile chain, not a + quantile regression forest. Each year conditions on one earnings value + only: the nearest later share that is known or already drawn (`y+1` in + the TRAIN fit; `y+2` when `y+1` is the career start and a masked odd + year; zero when none is known). That share picks one of 22 bins (zero, 20 TRAIN quantile bins, + and the cap) and scales the draw. A sex-age-bin cell with fewer than 200 + TRAIN person-years falls back to sex and bin, then to bin alone. Each + cell stores `P(zero)` and 65 quantiles of the log ratio to the next share + (of the log share when that is zero). Each person-year's draw uses its + own uniform, is capped at the wage base, and is zero below age 15. +- **What its result can show**: how a fill conditioned on one later year + scores on the gate's cells. It says nothing about a QRF that conditions + on many of a person's real years. Describe its DEV and TEST results in + these terms. Each candidate's exact fit parameters are in the manifest and in `scripts/fit_epuf_fills.py` (`REGISTERED`). @@ -100,6 +114,11 @@ draw seeds; see the dry run below. | `pre_donor` | 0 of 136 (worst: 0.77 tolerances) | certified | | `pre_chain` | 50 of 136 (worst: youth `ylevel`, 20 tolerances) | not adopted | +`pre_chain`'s 50 failures are those of the one-step binned chain described +above, not of a QRF. Its five worst DEV cells are youth earnings levels and +zero shares over ages 15-21 (`ylevel`, `yzero`) and one pre-career level +(`plevel`); the log records only the five worst cells. + **The registered procedure, dry-run on DEV.** On 2026-10-04, `epuf_fill_scoring.score_registered` ran with the DEV matrix in place of TEST, the registered artifacts loaded by SHA-256, and all 20 draw seeds. It @@ -193,3 +212,16 @@ them would need a refit. Neither changes a registered draw: and adoption under the registered rule. 4. Record the adoption in this document. A family whose primary and alternative both fail to be adopted keeps the current rule. +5. Describe `pre_chain`'s result as that of a one-step binned + conditional-quantile chain (see "The four candidates"), never as a QRF + result. + +## Exploratory follow-up (after TEST, not a candidate) + +A full-career QRF fill, using microcosm-fit's QRF, in which each +pre-career year conditions on all of the person's recorded years. It would +show whether a QRF given many real predictors scores better on the cells +the one-step chain fails. If it is run, it is run only after the +registered TEST scoring and is reported beside the registered results, +labelled exploratory. It is not a candidate: it cannot change either +family's tier or adoption, and adopting it would need a new registration. From 0634936710a0d898047a8c28fdfdd2a8ad6518a4 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 15:10:45 -0400 Subject: [PATCH 12/13] gate_epuf_fill: disclose the DEV sweep of K shown to Max before d927 Max is ruling on d927 (ratify K = 1). The brief he sees includes a DEV-only sweep of K over the registered dry run, so the sweep is disclosed in the DEV log before the lock closes it: - gate_epuf_fill_dev_registered_dryrun.json: the registered TEST procedure dry-run on DEV, with every cell (previously summarized in the log only); - gate_epuf_fill_dev_k_sweep.py and .txt: the re-scoring at K from 0.25 to 4 with the repository's own score/adoption_tier/adopt/combined_current; K = 1 reproduces the record exactly; - a DEV-log line and a paragraph in the candidates registration. K = 1 was registered at 14045be4, before any DEV score; nothing registered changes. No pinned file, manifest or bound file is touched. Co-Authored-By: Claude Opus 5.5 --- .../gate_epuf_fill_candidates_registration.md | 12 + docs/amendments/gate_epuf_fill_dev_k_sweep.py | 225 ++++++++++++++++++ .../amendments/gate_epuf_fill_dev_k_sweep.txt | 41 ++++ .../gate_epuf_fill_dev_registered_dryrun.json | 1 + ...e_epuf_fill_dev_scores_after_round_2.jsonl | 1 + 5 files changed, 280 insertions(+) create mode 100644 docs/amendments/gate_epuf_fill_dev_k_sweep.py create mode 100644 docs/amendments/gate_epuf_fill_dev_k_sweep.txt create mode 100644 docs/amendments/gate_epuf_fill_dev_registered_dryrun.json diff --git a/docs/amendments/gate_epuf_fill_candidates_registration.md b/docs/amendments/gate_epuf_fill_candidates_registration.md index 9f2149f4..fd41fad8 100644 --- a/docs/amendments/gate_epuf_fill_candidates_registration.md +++ b/docs/amendments/gate_epuf_fill_candidates_registration.md @@ -119,6 +119,18 @@ above, not of a QRF. Its five worst DEV cells are youth earnings levels and zero shares over ages 15-21 (`ylevel`, `yzero`) and one pre-career level (`plevel`); the log records only the five worst cells. +**A sweep of K, shown to the ratifier.** Before Max ruled on d927 (ratify +`K = 1`), the dry run below was re-scored at every `K` from 0.25 to 4 with +the repository's own scoring and adoption functions. `K = 1` reproduces the +record exactly. The same primaries are adopted for every `K` from about 0.9 +to 4. The odd primary is certified from `K = 2.69` and not adopted below +about 0.9. The pre primary is certified from `K = 0.77`. `K = 1` was +registered at `14045be4`, before any DEV score; the sweep is disclosed so +the ratification is read with it in view. Script, input and output: +`gate_epuf_fill_dev_k_sweep.py`, `gate_epuf_fill_dev_registered_dryrun.json` +(the dry run with every cell) and `gate_epuf_fill_dev_k_sweep.txt`, logged +as the last line of the DEV log. + **The registered procedure, dry-run on DEV.** On 2026-10-04, `epuf_fill_scoring.score_registered` ran with the DEV matrix in place of TEST, the registered artifacts loaded by SHA-256, and all 20 draw seeds. It diff --git a/docs/amendments/gate_epuf_fill_dev_k_sweep.py b/docs/amendments/gate_epuf_fill_dev_k_sweep.py new file mode 100644 index 00000000..d8c7bbfe --- /dev/null +++ b/docs/amendments/gate_epuf_fill_dev_k_sweep.py @@ -0,0 +1,225 @@ +"""DEV-only, post hoc: sensitivity of gate_epuf_fill's verdicts to K. + +Shown to Max before he ruled on d927 (ratify K = 1), and disclosed in the +DEV log for that reason. K = 1 was registered at 14045be4, before any DEV +score; this sweep does not change it. + +Reads only gate_epuf_fill_dev_registered_dryrun.json (the registered TEST +procedure dry-run on DEV, with every cell) and the registered floor build +(hash-checked via epuf_fill_scoring.load_registered_floors). No EPUF +microdata, no TEST. Run from the repository root with PYTHONPATH=src; +its output is gate_epuf_fill_dev_k_sweep.txt. + +For each K, every gating cell's tolerance is K times the registered +(K=1) tolerance, and each recorded score is re-scored with the +repository's own epuf_fill_gate.score / adoption_tier / adopt and +epuf_fill_scoring.combined_current. + +Reconstruction: score() takes truth CellValues and per-draw filled +CellValues and uses only .value (epuf_fill_gate.py:1015-1053). The dry run +recorded each cell's truth and the mean filled value over draws, so one +"draw" equal to the recorded mean reproduces the recorded gap exactly +(the mean of one value is that value; gap() is recomputed from it). +""" + +import json +import math +import sys +from pathlib import Path + +import numpy as np + +from populace_dynamics.harness import epuf_fill_gate as g +from populace_dynamics.harness import epuf_fill_scoring as scoring +from populace_dynamics.harness.epuf_cells import CellValue + +DRY = Path("docs/amendments/gate_epuf_fill_dev_registered_dryrun.json") +KS = (0.5, 0.75, 1.0, 1.25, 1.5, 2.0, 2.5, 3.0) +GRID = np.round(np.arange(0.25, 4.0001, 0.005), 3) + + +def num(x): + return float(x) if not isinstance(x, str) else float(x) # "inf"/"nan" + + +rec = json.loads(DRY.read_text()) +floors = scoring.load_registered_floors() # SHA-256 checked +assert rec["floors_sha256"] == floors["sha256"] +assert g.K_TOLERANCE == 1.0 + +fams = {} +for fam, e in rec["families"].items(): + truth = { + c: CellValue(num(v["value"]), int(v["events"]), int(v["events"])) + for c, v in e["truth"].items() + } + gating = [ + c + for c in floors["gating"] + if c.startswith(f"{fam}.") and c not in e["dropped_undefined_truth"] + ] + rec_gating = sorted( + c for c, r in e["primary"]["cells"].items() if "tolerance" in r + ) + assert rec_gating == sorted(gating), fam + for c in gating: # recorded tolerances are the floor build's (K=1) + assert e["primary"]["cells"][c]["tolerance"] == floors["tolerance"][c] + + def draws(block): + return [ + {c: CellValue(num(r["filled"]), 0, 0) for c, r in block.items()} + ] + + roles = {r: draws(e[r]["cells"]) for r in ("primary", "alternative")} + readings = {k: draws(v["cells"]) for k, v in e["current_rule"].items()} + fams[fam] = dict( + e=e, truth=truth, gating=gating, roles=roles, readings=readings + ) + + +def run(fam, k): + f = fams[fam] + tol = {c: k * floors["tolerance"][c] for c in f["gating"]} + sc = lambda d: g.score(f["truth"], d, tol, f["gating"]) # noqa: E731 + cur = {n: sc(d) for n, d in f["readings"].items()} + ref = scoring.combined_current(*cur.values()) + out = { + "current": {n: s["n_failing"] for n, s in cur.items()}, + "ref_failing": ref["n_failing"], + "scores": {}, + } + tiers = {} + for role, d in f["roles"].items(): + s = sc(d) + tiers[role] = g.adoption_tier(s, ref) + out["scores"][role] = s + out[role] = {"n_failing": s["n_failing"], "tier": tiers[role]} + out["adopted"] = g.adopt(tiers["primary"], tiers["alternative"]) + return out + + +# ---- K = 1 must reproduce the record exactly -------------------------------- +mismatch = [] +for fam, f in fams.items(): + e, r = f["e"], run(fam, 1.0) + if r["adopted"] != e["adopted"]: + mismatch.append((fam, "adopted", r["adopted"], e["adopted"])) + for role in ("primary", "alternative"): + for key in ("n_failing", "tier"): + if r[role][key] != e[role][key]: + mismatch.append((fam, role, key, r[role][key], e[role][key])) + for c, row in r["scores"][role]["cells"].items(): + want = e[role]["cells"][c] + same_gap = (row["gap"] == num(want["gap"])) or ( + math.isnan(row["gap"]) and math.isnan(num(want["gap"])) + ) + if not same_gap or row.get("passes") != want.get("passes"): + mismatch.append((fam, role, c)) + for n, v in r["current"].items(): + if v != e["current_rule"][n]["n_failing"]: + mismatch.append( + (fam, "current", n, v, e["current_rule"][n]["n_failing"]) + ) +print("K=1 reproduction mismatches:", mismatch or "none") +if mismatch: + sys.exit(1) + +# ---- table ------------------------------------------------------------------ +print( + "\n| K | odd primary | odd alt | odd cur fb/2s | odd adopted " + "| pre primary | pre alt | pre cur | pre adopted |" +) +print("|---|---|---|---|---|---|---|---|---|") +for k in KS: + o, p = run("odd", k), run("pre", k) + cell = lambda r, role: f"{r[role]['n_failing']} {r[role]['tier']}" # noqa + print( + f"| {k} | {cell(o,'primary')} | {cell(o,'alternative')} | " + f"{o['current']['fallback']}/{o['current']['two_sided']} | " + f"{o['adopted']} | {cell(p,'primary')} | {cell(p,'alternative')} " + f"| {p['current']['fallback']} | {p['adopted']} |" + ) + +# ---- odd primary worst cells, |gap| / sigma (sigma = K=1 tolerance) --------- +e = fams["odd"]["e"] +ratios = sorted( + ( + (abs(num(r["gap"])) / r["tolerance"], c) + for c, r in e["primary"]["cells"].items() + if "tolerance" in r + ), + reverse=True, +) +print("\nodd primary worst 10 |gap|/sigma:") +for x, c in ratios[:10]: + print(f" {x:.3f} {c}") +kmin = ratios[0][0] +print(f"smallest K certifying odd primary = max |gap|/sigma = {kmin:.4f}") +r = run("odd", kmin) +print( + " check at that K:", + r["primary"], + "; at K-1e-9:", + run("odd", kmin - 1e-9)["primary"], +) + +# ---- fine grid: tier transitions ------------------------------------------- +print("\ntransitions on K grid 0.25..4.0 step 0.005:") +for fam in ("odd", "pre"): + prev = None + for k in GRID: + r = run(fam, float(k)) + state = (r["primary"]["tier"], r["alternative"]["tier"], r["adopted"]) + if state != prev: + print( + f" {fam} K>={k}: primary={state[0]} " + f"(fail {r['primary']['n_failing']}), " + f"alt={state[1]} (fail {r['alternative']['n_failing']}), " + f"cur={r['current']}, adopted={state[2]}" + ) + prev = state + +# odd primary: which cells block "improves" below K=1 +for k in (0.5, 0.75): + f = fams["odd"] + tol = {c: k * floors["tolerance"][c] for c in f["gating"]} + cur = [ + g.score(f["truth"], d, tol, f["gating"]) + for d in f["readings"].values() + ] + ref = scoring.combined_current(*cur) + s = g.score(f["truth"], f["roles"]["primary"], tol, f["gating"]) + blockers = [] + for c, row in s["cells"].items(): + if "passes" not in row: + continue + b = float(ref["cells"][c]["gap"]) + b = abs(b) if np.isfinite(b) else np.inf + allowed = max( + row["tolerance"], min(b, g.IMPROVES_CAP * row["tolerance"]) + ) + if abs(row["gap"]) > allowed: + blockers.append( + ( + c, + round(abs(row["gap"]) / floors["tolerance"][c], 3), + round(b / floors["tolerance"][c], 3), + ) + ) + print( + f"\nodd primary improves-blockers at K={k} " + f"(cell, |gap|/sigma, |cur gap|/sigma): {blockers}" + ) + +# pre primary: max |gap|/sigma (headroom of the certification) +pr = sorted( + ( + (abs(num(r["gap"])) / r["tolerance"], c) + for c, r in fams["pre"]["e"]["primary"]["cells"].items() + if "tolerance" in r + ), + reverse=True, +) +print( + "\npre primary worst 5 |gap|/sigma:", [(round(x, 3), c) for x, c in pr[:5]] +) diff --git a/docs/amendments/gate_epuf_fill_dev_k_sweep.txt b/docs/amendments/gate_epuf_fill_dev_k_sweep.txt new file mode 100644 index 00000000..d75f9299 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_dev_k_sweep.txt @@ -0,0 +1,41 @@ +K=1 reproduction mismatches: none + +| K | odd primary | odd alt | odd cur fb/2s | odd adopted | pre primary | pre alt | pre cur | pre adopted | +|---|---|---|---|---|---|---|---|---| +| 0.5 | 29 not_adopted | 65 not_adopted | 103/101 | None | 6 improves | 74 not_adopted | 136 | primary | +| 0.75 | 14 not_adopted | 53 not_adopted | 101/100 | None | 1 improves | 63 not_adopted | 132 | primary | +| 1.0 | 7 improves | 46 not_adopted | 100/99 | primary | 0 certified | 50 not_adopted | 131 | primary | +| 1.25 | 4 improves | 43 not_adopted | 98/99 | primary | 0 certified | 42 not_adopted | 126 | primary | +| 1.5 | 3 improves | 39 not_adopted | 97/98 | primary | 0 certified | 39 not_adopted | 123 | primary | +| 2.0 | 2 improves | 29 not_adopted | 96/97 | primary | 0 certified | 29 not_adopted | 112 | primary | +| 2.5 | 1 improves | 22 not_adopted | 91/93 | primary | 0 certified | 23 not_adopted | 98 | primary | +| 3.0 | 0 certified | 19 not_adopted | 85/88 | primary | 0 certified | 18 not_adopted | 88 | primary | + +odd primary worst 10 |gap|/sigma: + 2.690 odd.men.a22_29.r1 + 2.019 odd.women.a22_29.r1 + 1.585 odd.women.a22_29.zint + 1.365 odd.men.a22_29.zint + 1.232 odd.women.a22_74.zint + 1.170 odd.women.a60_74.r1 + 1.093 odd.women.a22_74.r1 + 0.969 odd.men.a22_74.r1 + 0.928 odd.men.a22_29.r3 + 0.858 odd.women.a22_29.wint +smallest K certifying odd primary = max |gap|/sigma = 2.6903 + check at that K: {'n_failing': 0, 'tier': 'certified'} ; at K-1e-9: {'n_failing': 1, 'tier': 'improves'} + +transitions on K grid 0.25..4.0 step 0.005: + odd K>=0.25: primary=not_adopted (fail 49), alt=not_adopted (fail 81), cur={'fallback': 108, 'two_sided': 104}, adopted=None + odd K>=0.9: primary=improves (fail 9), alt=not_adopted (fail 48), cur={'fallback': 100, 'two_sided': 99}, adopted=primary + odd K>=2.695: primary=certified (fail 0), alt=not_adopted (fail 20), cur={'fallback': 87, 'two_sided': 90}, adopted=primary + odd K>=3.415: primary=certified (fail 0), alt=improves (fail 14), cur={'fallback': 81, 'two_sided': 86}, adopted=primary + pre K>=0.25: primary=not_adopted (fail 34), alt=not_adopted (fail 97), cur={'fallback': 136}, adopted=None + pre K>=0.26: primary=improves (fail 32), alt=not_adopted (fail 96), cur={'fallback': 136}, adopted=primary + pre K>=0.77: primary=certified (fail 0), alt=not_adopted (fail 60), cur={'fallback': 132}, adopted=primary + +odd primary improves-blockers at K=0.5 (cell, |gap|/sigma, |cur gap|/sigma): [('odd.men.a22_29.r1', 2.69, 11.583), ('odd.women.a22_29.r1', 2.019, 13.643), ('odd.women.a22_29.zint', 1.585, inf), ('odd.women.a60_74.r3', 0.651, 0.617)] + +odd primary improves-blockers at K=0.75 (cell, |gap|/sigma, |cur gap|/sigma): [('odd.men.a22_29.r1', 2.69, 11.583)] + +pre primary worst 5 |gap|/sigma: [(0.769, 'pre.men.b1946_1980.yzero'), (0.723, 'pre.women.b1956_1965.yr_cross'), (0.718, 'pre.women.b1946_1980.yzero'), (0.657, 'pre.men.b1956_1965.yzero'), (0.607, 'pre.men.b1946_1980.ylevel')] diff --git a/docs/amendments/gate_epuf_fill_dev_registered_dryrun.json b/docs/amendments/gate_epuf_fill_dev_registered_dryrun.json new file mode 100644 index 00000000..94ca4427 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_dev_registered_dryrun.json @@ -0,0 +1 @@ +{"registration_id": "2026-10-03-epuf-career-fill", "floors_sha256": "d403a824416f00524fadceefb897f5bdcaa197c12ebee0b1f1fd52304d98e25e", "constants_sha256": "cabb7611fd088bffc94fb8e1ee5d454df2cb8afbfbf909a6ed837c471fb4e04a", "candidates": {"odd": {"primary": {"path": "/Users/maxghenis/PolicyEngine/epuf-data/fills/odd_forest_v1.npz", "sha256": "37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa"}, "alternative": {"path": "/Users/maxghenis/PolicyEngine/epuf-data/fills/odd_knn_v1.npz", "sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291"}}, "pre": {"alternative": {"path": "/Users/maxghenis/PolicyEngine/epuf-data/fills/pre_chain_v1.npz", "sha256": "8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724"}, "primary": {"path": "/Users/maxghenis/PolicyEngine/epuf-data/fills/pre_donor_v1.npz", "sha256": "3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7"}}}, "n_persons": 875829, "seeds": [7100, 7101, 7102, 7103, 7104, 7105, 7106, 7107, 7108, 7109, 7110, 7111, 7112, 7113, 7114, 7115, 7116, 7117, 7118, 7119], "families": {"odd": {"dropped_undefined_truth": [], "truth": {"odd.men.a22_29.r1": {"value": 0.8114342592359591, "events": 24883}, "odd.men.a22_29.r3": {"value": 0.6020834292042946, "events": 23658}, "odd.men.a22_29.r2": {"value": 0.7090594261869381, "events": 24265}, "odd.men.a22_29.r4": {"value": 0.5990582702554191, "events": 23718}, "odd.men.a22_29.zint": {"value": 0.027589420573962787, "events": 3434}, "odd.men.a22_29.zexit": {"value": 0.4099387400566883, "events": 8967}, "odd.men.a22_29.wint": {"value": 0.08288543140028289, "events": 1172}, "odd.men.a22_29.atcap": {"value": 0.014676604027739744, "events": 1983}, "odd.men.a22_29.level": {"value": 0.23657185599421368, "events": 135113}, "odd.men.a22_29.q10": {"value": 0.040229885057471264, "events": 135113}, "odd.men.a22_29.q50": {"value": 0.24222222222222223, "events": 135113}, "odd.men.a22_29.q90": {"value": 0.5661157024793388, "events": 135113}, "odd.men.a30_44.r1": {"value": 0.9000861304840623, "events": 51112}, "odd.men.a30_44.r3": {"value": 0.8030903893396322, "events": 50209}, "odd.men.a30_44.r2": {"value": 0.8479233224100629, "events": 52027}, "odd.men.a30_44.r4": {"value": 0.7869108820616492, "events": 52730}, "odd.men.a30_44.zint": {"value": 0.017933882760031997, "events": 4820}, "odd.men.a30_44.zexit": {"value": 0.4342024282437196, "events": 15521}, "odd.men.a30_44.wint": {"value": 0.07239145006330527, "events": 1944}, "odd.men.a30_44.atcap": {"value": 0.10574805846620577, "events": 30256}, "odd.men.a30_44.level": {"value": 0.3982899917120792, "events": 286114}, "odd.men.a30_44.q10": {"value": 0.08706467661691543, "events": 286114}, "odd.men.a30_44.q50": {"value": 0.41333333333333333, "events": 286114}, "odd.men.a30_44.q90": {"value": 1.0, "events": 286114}, "odd.men.a45_59.r1": {"value": 0.9077661818197654, "events": 35553}, "odd.men.a45_59.r3": {"value": 0.8146289743583418, "events": 33477}, "odd.men.a45_59.r2": {"value": 0.850421444941529, "events": 34460}, "odd.men.a45_59.r4": {"value": 0.7598303885400162, "events": 32366}, "odd.men.a45_59.zint": {"value": 0.015835938476928848, "events": 3166}, "odd.men.a45_59.zexit": {"value": 0.45185706741770815, "events": 12835}, "odd.men.a45_59.wint": {"value": 0.05825168693530555, "events": 1528}, "odd.men.a45_59.atcap": {"value": 0.15826463477931516, "events": 33846}, "odd.men.a45_59.level": {"value": 0.4279852808128974, "events": 213857}, "odd.men.a45_59.q10": {"value": 0.08868501529051988, "events": 213857}, "odd.men.a45_59.q50": {"value": 0.4701492537313433, "events": 213857}, "odd.men.a45_59.q90": {"value": 1.0, "events": 213857}, "odd.men.a60_74.r1": {"value": 0.8712782553777247, "events": 9471}, "odd.men.a60_74.r3": {"value": 0.7283746799072385, "events": 7450}, "odd.men.a60_74.r2": {"value": 0.7888015745113253, "events": 8329}, "odd.men.a60_74.r4": {"value": 0.6781549627202678, "events": 6609}, "odd.men.a60_74.zint": {"value": 0.027132152458672565, "events": 1423}, "odd.men.a60_74.zexit": {"value": 0.5047452182800409, "events": 10176}, "odd.men.a60_74.wint": {"value": 0.03084255319148936, "events": 906}, "odd.men.a60_74.atcap": {"value": 0.10024796315975912, "events": 6226}, "odd.men.a60_74.level": {"value": 0.20453937198442498, "events": 62106}, "odd.men.a60_74.q10": {"value": 0.01529051987767584, "events": 62106}, "odd.men.a60_74.q50": {"value": 0.21666666666666667, "events": 62106}, "odd.men.a60_74.q90": {"value": 1.0, "events": 62106}, "odd.men.a22_74.r1": {"value": 0.8935990156167083, "events": 127726}, "odd.men.a22_74.r3": {"value": 0.7806492619651291, "events": 121284}, "odd.men.a22_74.r2": {"value": 0.8291027164330249, "events": 124053}, "odd.men.a22_74.r4": {"value": 0.7414979369803346, "events": 118165}, "odd.men.a22_74.zint": {"value": 0.019892968610837895, "events": 12843}, "odd.men.a22_74.zexit": {"value": 0.44752843148294114, "events": 47694}, "odd.men.a22_74.wint": {"value": 0.05745341614906832, "events": 5550}, "odd.men.a22_74.atcap": {"value": 0.10371778137953785, "events": 72311}, "odd.men.a22_74.level": {"value": 0.35325148977531445, "events": 697190}, "odd.men.a22_74.q10": {"value": 0.06111111111111111, "events": 697190}, "odd.men.a22_74.q50": {"value": 0.37052341597796146, "events": 697190}, "odd.men.a22_74.q90": {"value": 1.0, "events": 697190}, "odd.men.b1936_1940.aime_p10": {"value": 346.0, "events": 13032}, "odd.men.b1936_1940.aime_p25": {"value": 1131.0, "events": 13032}, "odd.men.b1936_1940.aime_p50": {"value": 2430.0, "events": 13032}, "odd.men.b1936_1940.aime_p75": {"value": 3582.0, "events": 13032}, "odd.men.b1936_1940.aime_p90": {"value": 4382.0, "events": 13032}, "odd.men.b1941_1945.aime_p10": {"value": 338.0, "events": 15907}, "odd.men.b1941_1945.aime_p25": {"value": 1174.25, "events": 15907}, "odd.men.b1941_1945.aime_p50": {"value": 2821.5, "events": 15907}, "odd.men.b1941_1945.aime_p75": {"value": 4421.0, "events": 15907}, "odd.men.b1941_1945.aime_p90": {"value": 5528.300000000001, "events": 15907}, "odd.men.b1936_1945.aime_p10": {"value": 342.8000000000002, "events": 28939}, "odd.men.b1936_1945.aime_p25": {"value": 1152.0, "events": 28939}, "odd.men.b1936_1945.aime_p50": {"value": 2625.0, "events": 28939}, "odd.men.b1936_1945.aime_p75": {"value": 3991.5, "events": 28939}, "odd.men.b1936_1945.aime_p90": {"value": 5075.0, "events": 28939}, "odd.men.b1946_1955.paime_p10": {"value": 308.0, "events": 44074}, "odd.men.b1946_1955.paime_p25": {"value": 1026.0, "events": 44074}, "odd.men.b1946_1955.paime_p50": {"value": 2614.0, "events": 44074}, "odd.men.b1946_1955.paime_p75": {"value": 4390.0, "events": 44074}, "odd.men.b1946_1955.paime_p90": {"value": 5810.0, "events": 44074}, "odd.men.b1956_1965.paime_p10": {"value": 253.0, "events": 49852}, "odd.men.b1956_1965.paime_p25": {"value": 765.0, "events": 49852}, "odd.men.b1956_1965.paime_p50": {"value": 1764.0, "events": 49852}, "odd.men.b1956_1965.paime_p75": {"value": 2958.0, "events": 49852}, "odd.men.b1956_1965.paime_p90": {"value": 4111.0, "events": 49852}, "odd.men.b1966_1980.paime_p10": {"value": 112.0, "events": 61607}, "odd.men.b1966_1980.paime_p25": {"value": 303.25, "events": 61607}, "odd.men.b1966_1980.paime_p50": {"value": 655.0, "events": 61607}, "odd.men.b1966_1980.paime_p75": {"value": 1225.0, "events": 61607}, "odd.men.b1966_1980.paime_p90": {"value": 1888.9000000000015, "events": 61607}, "odd.men.b1946_1980.paime_p10": {"value": 175.0, "events": 155533}, "odd.men.b1946_1980.paime_p25": {"value": 487.0, "events": 155533}, "odd.men.b1946_1980.paime_p50": {"value": 1245.0, "events": 155533}, "odd.men.b1946_1980.paime_p75": {"value": 2644.0, "events": 155533}, "odd.men.b1946_1980.paime_p90": {"value": 4310.0, "events": 155533}, "odd.women.a22_29.r1": {"value": 0.7942016511246733, "events": 22841}, "odd.women.a22_29.r3": {"value": 0.5535614164842262, "events": 21369}, "odd.women.a22_29.r2": {"value": 0.6731585499668502, "events": 22004}, "odd.women.a22_29.r4": {"value": 0.552312691565311, "events": 21259}, "odd.women.a22_29.zint": {"value": 0.0315547703180212, "events": 3572}, "odd.women.a22_29.zexit": {"value": 0.41757828810020875, "events": 10001}, "odd.women.a22_29.wint": {"value": 0.08501885498800137, "events": 1240}, "odd.women.a22_29.atcap": {"value": 0.005728386357627567, "events": 715}, "odd.women.a22_29.level": {"value": 0.1854665923300802, "events": 124817}, "odd.women.a22_29.q10": {"value": 0.028735632183908046, "events": 124817}, "odd.women.a22_29.q50": {"value": 0.191131498470948, "events": 124817}, "odd.women.a22_29.q90": {"value": 0.46005509641873277, "events": 124817}, "odd.women.a30_44.r1": {"value": 0.888241419643356, "events": 44822}, "odd.women.a30_44.r3": {"value": 0.7685608654079126, "events": 42936}, "odd.women.a30_44.r2": {"value": 0.8242277470219269, "events": 44797}, "odd.women.a30_44.r4": {"value": 0.7490435327097834, "events": 44806}, "odd.women.a30_44.zint": {"value": 0.022222222222222223, "events": 5098}, "odd.women.a30_44.zexit": {"value": 0.45066799061202384, "events": 19970}, "odd.women.a30_44.wint": {"value": 0.06073485056210584, "events": 2215}, "odd.women.a30_44.atcap": {"value": 0.034619662054697874, "events": 8685}, "odd.women.a30_44.level": {"value": 0.2579924798566703, "events": 250869}, "odd.women.a30_44.q10": {"value": 0.04228855721393035, "events": 250869}, "odd.women.a30_44.q50": {"value": 0.26859504132231404, "events": 250869}, "odd.women.a30_44.q90": {"value": 0.6716417910447762, "events": 250869}, "odd.women.a45_59.r1": {"value": 0.9101006329634369, "events": 31760}, "odd.women.a45_59.r3": {"value": 0.8096130207660378, "events": 29595}, "odd.women.a45_59.r2": {"value": 0.8498173267452248, "events": 30540}, "odd.women.a45_59.r4": {"value": 0.7620167777938601, "events": 28452}, "odd.women.a45_59.zint": {"value": 0.01632776600720909, "events": 2967}, "odd.women.a45_59.zexit": {"value": 0.4693136110029843, "events": 14468}, "odd.women.a45_59.wint": {"value": 0.050115932427956277, "events": 1513}, "odd.women.a45_59.atcap": {"value": 0.04053483605515179, "events": 7970}, "odd.women.a45_59.level": {"value": 0.28104586423928846, "events": 196621}, "odd.women.a45_59.q10": {"value": 0.053516819571865444, "events": 196621}, "odd.women.a45_59.q50": {"value": 0.29338842975206614, "events": 196621}, "odd.women.a45_59.q90": {"value": 0.7300275482093664, "events": 196621}, "odd.women.a60_74.r1": {"value": 0.8610627385333514, "events": 6985}, "odd.women.a60_74.r3": {"value": 0.7141078828415282, "events": 5282}, "odd.women.a60_74.r2": {"value": 0.7722738686836694, "events": 6041}, "odd.women.a60_74.r4": {"value": 0.6501204938024134, "events": 4624}, "odd.women.a60_74.zint": {"value": 0.022290698541288734, "events": 897}, "odd.women.a60_74.zexit": {"value": 0.5155483759303063, "events": 8397}, "odd.women.a60_74.wint": {"value": 0.023107482596493787, "events": 634}, "odd.women.a60_74.atcap": {"value": 0.01812919896640827, "events": 877}, "odd.women.a60_74.level": {"value": 0.12926943691615103, "events": 48375}, "odd.women.a60_74.q10": {"value": 0.017412935323383085, "events": 48375}, "odd.women.a60_74.q50": {"value": 0.15517241379310345, "events": 48375}, "odd.women.a60_74.q90": {"value": 0.5360696517412935, "events": 48375}, "odd.women.a22_74.r1": {"value": 0.8812684347612898, "events": 110453}, "odd.women.a22_74.r3": {"value": 0.7468316241328095, "events": 103479}, "odd.women.a22_74.r2": {"value": 0.8056395091372208, "events": 106291}, "odd.women.a22_74.r4": {"value": 0.7116744894807431, "events": 100276}, "odd.women.a22_74.zint": {"value": 0.022201124403524123, "events": 12534}, "odd.women.a22_74.zexit": {"value": 0.45845752128015943, "events": 53375}, "odd.women.a22_74.wint": {"value": 0.051544874036178946, "events": 5602}, "odd.women.a22_74.atcap": {"value": 0.029398307023564402, "events": 18247}, "odd.women.a22_74.level": {"value": 0.2372854094489719, "events": 620682}, "odd.women.a22_74.q10": {"value": 0.03777777777777778, "events": 620682}, "odd.women.a22_74.q50": {"value": 0.24875621890547264, "events": 620682}, "odd.women.a22_74.q90": {"value": 0.6444444444444445, "events": 620682}, "odd.women.b1936_1940.aime_p10": {"value": 82.29999999999995, "events": 11967}, "odd.women.b1936_1940.aime_p25": {"value": 287.75, "events": 11967}, "odd.women.b1936_1940.aime_p50": {"value": 786.0, "events": 11967}, "odd.women.b1936_1940.aime_p75": {"value": 1566.0, "events": 11967}, "odd.women.b1936_1940.aime_p90": {"value": 2438.7000000000007, "events": 11967}, "odd.women.b1941_1945.aime_p10": {"value": 115.0, "events": 15070}, "odd.women.b1941_1945.aime_p25": {"value": 399.0, "events": 15070}, "odd.women.b1941_1945.aime_p50": {"value": 1077.0, "events": 15070}, "odd.women.b1941_1945.aime_p75": {"value": 2116.0, "events": 15070}, "odd.women.b1941_1945.aime_p90": {"value": 3268.6000000000004, "events": 15070}, "odd.women.b1936_1945.aime_p10": {"value": 98.0, "events": 27037}, "odd.women.b1936_1945.aime_p25": {"value": 345.0, "events": 27037}, "odd.women.b1936_1945.aime_p50": {"value": 932.0, "events": 27037}, "odd.women.b1936_1945.aime_p75": {"value": 1865.0, "events": 27037}, "odd.women.b1936_1945.aime_p90": {"value": 2920.0, "events": 27037}, "odd.women.b1946_1955.paime_p10": {"value": 158.0, "events": 41669}, "odd.women.b1946_1955.paime_p25": {"value": 519.0, "events": 41669}, "odd.women.b1946_1955.paime_p50": {"value": 1313.0, "events": 41669}, "odd.women.b1946_1955.paime_p75": {"value": 2486.0, "events": 41669}, "odd.women.b1946_1955.paime_p90": {"value": 3780.0, "events": 41669}, "odd.women.b1956_1965.paime_p10": {"value": 140.0, "events": 46740}, "odd.women.b1956_1965.paime_p25": {"value": 426.0, "events": 46740}, "odd.women.b1956_1965.paime_p50": {"value": 1000.0, "events": 46740}, "odd.women.b1956_1965.paime_p75": {"value": 1842.0, "events": 46740}, "odd.women.b1956_1965.paime_p90": {"value": 2793.0, "events": 46740}, "odd.women.b1966_1980.paime_p10": {"value": 80.0, "events": 58000}, "odd.women.b1966_1980.paime_p25": {"value": 211.0, "events": 58000}, "odd.women.b1966_1980.paime_p50": {"value": 463.0, "events": 58000}, "odd.women.b1966_1980.paime_p75": {"value": 866.75, "events": 58000}, "odd.women.b1966_1980.paime_p90": {"value": 1387.0, "events": 58000}, "odd.women.b1946_1980.paime_p10": {"value": 109.0, "events": 146409}, "odd.women.b1946_1980.paime_p25": {"value": 309.0, "events": 146409}, "odd.women.b1946_1980.paime_p50": {"value": 758.0, "events": 146409}, "odd.women.b1946_1980.paime_p75": {"value": 1590.0, "events": 146409}, "odd.women.b1946_1980.paime_p90": {"value": 2696.0, "events": 146409}}, "current_rule": {"fallback": {"passes": false, "n_gating": 183, "n_failing": 100, "cells": {"odd.men.a22_29.atcap": {"truth": 0.014676604027739744, "filled": 0.004275744117403658, "gap": -1.2332965133723093, "seed_sd": 0.0, "tolerance": 0.18023370223284765, "passes": false}, "odd.men.a22_29.level": {"truth": 0.23657185599421368, "filled": 0.2422219209391976, "gap": 0.02360234199005551, "seed_sd": 0.0, "tolerance": 0.02171146931539837, "passes": false}, "odd.men.a22_29.q10": {"truth": 0.040229885057471264, "filled": 0.038505747126436785, "gap": -0.04380262265839274, "seed_sd": 0.0, "tolerance": 0.07329266358707544, "passes": true}, "odd.men.a22_29.q50": {"truth": 0.24222222222222223, "filled": 0.231651376146789, "gap": -0.04462202597255316, "seed_sd": 0.0, "tolerance": 0.022980444394274636, "passes": false}, "odd.men.a22_29.q90": {"truth": 0.5661157024793388, "filled": 0.54, "gap": -0.04722933909525551, "seed_sd": 0.0, "tolerance": 0.02353059621893112, "passes": false}, "odd.men.a22_29.r1": {"truth": 0.8114342592359591, "filled": 0.8956863944371684, "gap": 0.08425213520120922, "seed_sd": 0.0, "tolerance": 0.007273490161268873, "passes": false}, "odd.men.a22_29.r2": {"truth": 0.7090594261869381, "filled": 0.863164005737995, "gap": 0.15410457955105694, "seed_sd": 0.0, "tolerance": 0.014356564488812985, "passes": false}, "odd.men.a22_29.r3": {"truth": 0.6020834292042946, "filled": 0.6369726646985666, "gap": 0.03488923549427203, "seed_sd": 0.0, "tolerance": 0.013280307041936976, "passes": false}, "odd.men.a22_29.r4": {"truth": 0.5990582702554191, "filled": 0.6928618295102605, "gap": 0.0938035592548414, "seed_sd": 0.0, "tolerance": 0.02119963449400435, "passes": false}, "odd.men.a22_29.wint": {"truth": 0.08288543140028289, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.men.a22_29.zexit": {"truth": 0.4099387400566883, "filled": 0.06116851056048277, "gap": -1.9023752102638736, "seed_sd": 0.0, "tolerance": 0.052311207438163365, "passes": false}, "odd.men.a22_29.zint": {"truth": 0.027589420573962787, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.12140160693630424, "passes": false}, "odd.men.a22_74.atcap": {"truth": 0.10371778137953785, "filled": 0.0461710166893302, "gap": -0.8093213131248014, "seed_sd": 0.0, "tolerance": 0.03690628859850606, "passes": false}, "odd.men.a22_74.level": {"truth": 0.35325148977531445, "filled": 0.3545980146560176, "gap": 0.003804555902007456, "seed_sd": 0.0, "tolerance": 0.01106822547283481, "passes": true}, "odd.men.a22_74.q10": {"truth": 0.06111111111111111, "filled": 0.04700413223140496, "gap": -0.26245818322783254, "seed_sd": 0.0, "tolerance": 0.03462864626687699, "passes": false}, "odd.men.a22_74.q50": {"truth": 0.37052341597796146, "filled": 0.3390804597701149, "gap": -0.08867922008585272, "seed_sd": 0.0, "tolerance": 0.012463773427145929, "passes": false}, "odd.men.a22_74.q90": {"truth": 1.0, "filled": 0.9277777777777778, "gap": -0.07496303847345523, "seed_sd": 0.0, "tolerance": 0.003451823067723808, "passes": false}, "odd.men.a22_74.r1": {"truth": 0.8935990156167083, "filled": 0.9477352806293625, "gap": 0.05413626501265423, "seed_sd": 0.0, "tolerance": 0.00258049589094515, "passes": false}, "odd.men.a22_74.r2": {"truth": 0.8291027164330249, "filled": 0.9096502660547399, "gap": 0.08054754962171495, "seed_sd": 0.0, "tolerance": 0.004576681103770273, "passes": false}, "odd.men.a22_74.r3": {"truth": 0.7806492619651291, "filled": 0.7999851781213285, "gap": 0.01933591615619945, "seed_sd": 0.0, "tolerance": 0.004722018684826323, "passes": false}, "odd.men.a22_74.r4": {"truth": 0.7414979369803346, "filled": 0.7846725429897913, "gap": 0.04317460600945666, "seed_sd": 0.0, "tolerance": 0.007000982741779954, "passes": false}, "odd.men.a22_74.wint": {"truth": 0.05745341614906832, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07337302886151653, "passes": false}, "odd.men.a22_74.zexit": {"truth": 0.44752843148294114, "filled": 0.01255489246706452, "gap": -3.573629642112994, "seed_sd": 0.0, "tolerance": 0.01916127382099423, "passes": false}, "odd.men.a22_74.zint": {"truth": 0.019892968610837895, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.05481211657295162, "passes": false}, "odd.men.a30_44.atcap": {"truth": 0.10574805846620577, "filled": 0.04650078322293776, "gap": -0.82159030214472, "seed_sd": 0.0, "tolerance": 0.049095580366705964, "passes": false}, "odd.men.a30_44.level": {"truth": 0.3982899917120792, "filled": 0.39901752453379963, "gap": 0.001824974702530513, "seed_sd": 0.0, "tolerance": 0.015220021463125325, "passes": true}, "odd.men.a30_44.q10": {"truth": 0.08706467661691543, "filled": 0.06778606965174129, "gap": -0.25029454038016086, "seed_sd": 0.0, "tolerance": 0.047669064141889934, "passes": false}, "odd.men.a30_44.q50": {"truth": 0.41333333333333333, "filled": 0.38685015290519875, "gap": -0.06621695367851432, "seed_sd": 0.0, "tolerance": 0.016660415997544607, "passes": false}, "odd.men.a30_44.q90": {"truth": 1.0, "filled": 0.9390547263681592, "gap": -0.06288151992994163, "seed_sd": 0.0, "tolerance": 0.0037405831290436837, "passes": false}, "odd.men.a30_44.r1": {"truth": 0.9000861304840623, "filled": 0.9541063174829294, "gap": 0.054020186998867126, "seed_sd": 0.0, "tolerance": 0.003556208950302802, "passes": false}, "odd.men.a30_44.r2": {"truth": 0.8479233224100629, "filled": 0.9269327325683842, "gap": 0.0790094101583213, "seed_sd": 0.0, "tolerance": 0.006500650570758907, "passes": false}, "odd.men.a30_44.r3": {"truth": 0.8030903893396322, "filled": 0.8276167881663179, "gap": 0.024526398826685725, "seed_sd": 0.0, "tolerance": 0.006087370619188444, "passes": false}, "odd.men.a30_44.r4": {"truth": 0.7869108820616492, "filled": 0.8327212421045639, "gap": 0.04581036004291472, "seed_sd": 0.0, "tolerance": 0.009218001294372115, "passes": false}, "odd.men.a30_44.wint": {"truth": 0.07239145006330527, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.12001713872675526, "passes": false}, "odd.men.a30_44.zexit": {"truth": 0.4342024282437196, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.031953074094816486, "passes": false}, "odd.men.a30_44.zint": {"truth": 0.017933882760031997, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.08989768311292742, "passes": false}, "odd.men.a45_59.atcap": {"truth": 0.15826463477931516, "filled": 0.07469452108789909, "gap": -0.750861791701275, "seed_sd": 0.0, "tolerance": 0.04711609868499565, "passes": false}, "odd.men.a45_59.level": {"truth": 0.4279852808128974, "filled": 0.4276293059548939, "gap": -0.0008320916544616308, "seed_sd": 0.0, "tolerance": 0.015834008317210692, "passes": true}, "odd.men.a45_59.q10": {"truth": 0.08868501529051988, "filled": 0.06592039800995025, "gap": -0.29664301471606525, "seed_sd": 0.0, "tolerance": 0.055701610600039544, "passes": false}, "odd.men.a45_59.q50": {"truth": 0.4701492537313433, "filled": 0.4345730027548209, "gap": -0.07868625928414963, "seed_sd": 0.0, "tolerance": 0.020385938992396834, "passes": false}, "odd.men.a45_59.q90": {"truth": 1.0, "filled": 0.9958677685950413, "gap": -0.004140792666031388, "seed_sd": 0.0}, "odd.men.a45_59.r1": {"truth": 0.9077661818197654, "filled": 0.9568133762839999, "gap": 0.049047194464234445, "seed_sd": 0.0, "tolerance": 0.004032009924506552, "passes": false}, "odd.men.a45_59.r2": {"truth": 0.850421444941529, "filled": 0.9210348902278979, "gap": 0.07061344528636881, "seed_sd": 0.0, "tolerance": 0.0074262368998509335, "passes": false}, "odd.men.a45_59.r3": {"truth": 0.8146289743583418, "filled": 0.8339594060387736, "gap": 0.019330431680431803, "seed_sd": 0.0, "tolerance": 0.007198396861260139, "passes": false}, "odd.men.a45_59.r4": {"truth": 0.7598303885400162, "filled": 0.7967982689055731, "gap": 0.03696788036555698, "seed_sd": 0.0, "tolerance": 0.012107705260459355, "passes": false}, "odd.men.a45_59.wint": {"truth": 0.05825168693530555, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.13344110142481172, "passes": false}, "odd.men.a45_59.zexit": {"truth": 0.45185706741770815, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.036314919811016276, "passes": false}, "odd.men.a45_59.zint": {"truth": 0.015835938476928848, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09287476169914739, "passes": false}, "odd.men.a60_74.atcap": {"truth": 0.10024796315975912, "filled": 0.038797709400772665, "gap": -0.949285539547915, "seed_sd": 0.0, "tolerance": 0.12472307183935011, "passes": false}, "odd.men.a60_74.level": {"truth": 0.20453937198442498, "filled": 0.20537657883929814, "gap": 0.004084778927416988, "seed_sd": 0.0, "tolerance": 0.0522939828282486, "passes": true}, "odd.men.a60_74.q10": {"truth": 0.01529051987767584, "filled": 0.009950248756218905, "gap": -0.4296354690359774, "seed_sd": 0.0, "tolerance": 0.20186846491463123, "passes": false}, "odd.men.a60_74.q50": {"truth": 0.21666666666666667, "filled": 0.17507645259938837, "gap": -0.21313732370234062, "seed_sd": 0.0, "tolerance": 0.07446457229432804, "passes": false}, "odd.men.a60_74.q90": {"truth": 1.0, "filled": 0.8222222222222222, "gap": -0.19574457712609536, "seed_sd": 0.0, "tolerance": 0.04069424979546094, "passes": false}, "odd.men.a60_74.r1": {"truth": 0.8712782553777247, "filled": 0.9364895298673185, "gap": 0.06521127448959374, "seed_sd": 0.0, "tolerance": 0.009750532797445479, "passes": false}, "odd.men.a60_74.r2": {"truth": 0.7888015745113253, "filled": 0.8691602172532571, "gap": 0.08035864274193183, "seed_sd": 0.0, "tolerance": 0.020757725353477027, "passes": false}, "odd.men.a60_74.r3": {"truth": 0.7283746799072385, "filled": 0.7453814856416601, "gap": 0.017006805734421593, "seed_sd": 0.0, "tolerance": 0.02064690204069302, "passes": true}, "odd.men.a60_74.r4": {"truth": 0.6781549627202678, "filled": 0.6972387280502411, "gap": 0.019083765329973357, "seed_sd": 0.0, "tolerance": 0.03820786124425089, "passes": true}, "odd.men.a60_74.wint": {"truth": 0.03084255319148936, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.men.a60_74.zexit": {"truth": 0.5047452182800409, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04434932710092839, "passes": false}, "odd.men.a60_74.zint": {"truth": 0.027132152458672565, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.17758159695027936, "passes": false}, "odd.men.b1936_1940.aime_p10": {"truth": 346.0, "filled": 345.0, "gap": -0.0028943580263645075, "seed_sd": 0.0, "tolerance": 0.30817736367768905, "passes": true}, "odd.men.b1936_1940.aime_p25": {"truth": 1131.0, "filled": 1131.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.15436566295506535, "passes": true}, "odd.men.b1936_1940.aime_p50": {"truth": 2430.0, "filled": 2426.0, "gap": -0.0016474468305984757, "seed_sd": 0.0, "tolerance": 0.0822816707411384, "passes": true}, "odd.men.b1936_1940.aime_p75": {"truth": 3582.0, "filled": 3576.0, "gap": -0.001676446327252279, "seed_sd": 0.0, "tolerance": 0.04915519601813279, "passes": true}, "odd.men.b1936_1940.aime_p90": {"truth": 4382.0, "filled": 4372.0, "gap": -0.0022846708589838727, "seed_sd": 0.0, "tolerance": 0.03801502836217192, "passes": true}, "odd.men.b1936_1945.aime_p10": {"truth": 342.8000000000002, "filled": 341.0, "gap": -0.005264709440107929, "seed_sd": 0.0, "tolerance": 0.19387427864829543, "passes": true}, "odd.men.b1936_1945.aime_p25": {"truth": 1152.0, "filled": 1152.5, "gap": 0.00043393361496679717, "seed_sd": 0.0, "tolerance": 0.09928513681983075, "passes": true}, "odd.men.b1936_1945.aime_p50": {"truth": 2625.0, "filled": 2621.0, "gap": -0.001524971702317579, "seed_sd": 0.0, "tolerance": 0.05013200755321697, "passes": true}, "odd.men.b1936_1945.aime_p75": {"truth": 3991.5, "filled": 3985.5, "gap": -0.0015043252178763566, "seed_sd": 0.0, "tolerance": 0.030282584009309735, "passes": true}, "odd.men.b1936_1945.aime_p90": {"truth": 5075.0, "filled": 5060.0, "gap": -0.0029600416284765174, "seed_sd": 0.0, "tolerance": 0.029246187918705275, "passes": true}, "odd.men.b1941_1945.aime_p10": {"truth": 338.0, "filled": 336.0, "gap": -0.005934735519814716, "seed_sd": 0.0, "tolerance": 0.2434370939708618, "passes": true}, "odd.men.b1941_1945.aime_p25": {"truth": 1174.25, "filled": 1176.25, "gap": 0.0017017659924842832, "seed_sd": 0.0, "tolerance": 0.15717605214643726, "passes": true}, "odd.men.b1941_1945.aime_p50": {"truth": 2821.5, "filled": 2818.0, "gap": -0.0012412449505694312, "seed_sd": 0.0, "tolerance": 0.06794319334096698, "passes": true}, "odd.men.b1941_1945.aime_p75": {"truth": 4421.0, "filled": 4414.0, "gap": -0.0015846070095602016, "seed_sd": 0.0, "tolerance": 0.039611439643930886, "passes": true}, "odd.men.b1941_1945.aime_p90": {"truth": 5528.300000000001, "filled": 5520.300000000001, "gap": -0.0014481475296577173, "seed_sd": 0.0, "tolerance": 0.024092287379584104, "passes": true}, "odd.men.b1946_1955.paime_p10": {"truth": 308.0, "filled": 309.0, "gap": 0.0032414939241718344, "seed_sd": 0.0, "tolerance": 0.11641286886749184, "passes": true}, "odd.men.b1946_1955.paime_p25": {"truth": 1026.0, "filled": 1028.0, "gap": 0.001947420284395207, "seed_sd": 0.0, "tolerance": 0.07923354406873329, "passes": true}, "odd.men.b1946_1955.paime_p50": {"truth": 2614.0, "filled": 2609.0, "gap": -0.0019146090474384536, "seed_sd": 0.0, "tolerance": 0.03884145918889322, "passes": true}, "odd.men.b1946_1955.paime_p75": {"truth": 4390.0, "filled": 4389.0, "gap": -0.0002278163809830147, "seed_sd": 0.0, "tolerance": 0.02464882648778213, "passes": true}, "odd.men.b1946_1955.paime_p90": {"truth": 5810.0, "filled": 5808.0, "gap": -0.00034429334132468625, "seed_sd": 0.0, "tolerance": 0.01935089604023114, "passes": true}, "odd.men.b1946_1980.paime_p10": {"truth": 175.0, "filled": 176.0, "gap": 0.005698021114636909, "seed_sd": 0.0, "tolerance": 0.05432313062129484, "passes": true}, "odd.men.b1946_1980.paime_p25": {"truth": 487.0, "filled": 489.0, "gap": 0.004098366392282671, "seed_sd": 0.0, "tolerance": 0.03064646902486962, "passes": true}, "odd.men.b1946_1980.paime_p50": {"truth": 1245.0, "filled": 1248.0, "gap": 0.002406740030565402, "seed_sd": 0.0, "tolerance": 0.025757396517852603, "passes": true}, "odd.men.b1946_1980.paime_p75": {"truth": 2644.0, "filled": 2645.0, "gap": 0.0003781433208231988, "seed_sd": 0.0, "tolerance": 0.0206417529534006, "passes": true}, "odd.men.b1946_1980.paime_p90": {"truth": 4310.0, "filled": 4310.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.01797004074133569, "passes": true}, "odd.men.b1956_1965.paime_p10": {"truth": 253.0, "filled": 253.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.10325340800222761, "passes": true}, "odd.men.b1956_1965.paime_p25": {"truth": 765.0, "filled": 767.0, "gap": 0.0026109675407202104, "seed_sd": 0.0, "tolerance": 0.06714959239176221, "passes": true}, "odd.men.b1956_1965.paime_p50": {"truth": 1764.0, "filled": 1765.0, "gap": 0.000566732800660219, "seed_sd": 0.0, "tolerance": 0.03783036836459788, "passes": true}, "odd.men.b1956_1965.paime_p75": {"truth": 2958.0, "filled": 2961.0, "gap": 0.0010136848308457402, "seed_sd": 0.0, "tolerance": 0.02582244416868233, "passes": true}, "odd.men.b1956_1965.paime_p90": {"truth": 4111.0, "filled": 4110.0, "gap": -0.00024327940759860667, "seed_sd": 0.0, "tolerance": 0.022664845011086624, "passes": true}, "odd.men.b1966_1980.paime_p10": {"truth": 112.0, "filled": 112.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.07763263566160054, "passes": true}, "odd.men.b1966_1980.paime_p25": {"truth": 303.25, "filled": 305.0, "gap": 0.00575422878325238, "seed_sd": 0.0, "tolerance": 0.04145124387911718, "passes": true}, "odd.men.b1966_1980.paime_p50": {"truth": 655.0, "filled": 661.0, "gap": 0.009118604216434179, "seed_sd": 0.0, "tolerance": 0.02859495381248372, "passes": true}, "odd.men.b1966_1980.paime_p75": {"truth": 1225.0, "filled": 1230.0, "gap": 0.00407332538763594, "seed_sd": 0.0, "tolerance": 0.024541637055534683, "passes": true}, "odd.men.b1966_1980.paime_p90": {"truth": 1888.9000000000015, "filled": 1891.0, "gap": 0.00111114062068296, "seed_sd": 0.0, "tolerance": 0.025554311796701996, "passes": true}, "odd.women.a22_29.atcap": {"truth": 0.005728386357627567, "filled": 0.0014958367106329674, "gap": -1.3427481551382767, "seed_sd": 0.0}, "odd.women.a22_29.level": {"truth": 0.1854665923300802, "filled": 0.1897248573494353, "gap": 0.022700132836240172, "seed_sd": 0.0, "tolerance": 0.018517427407013797, "passes": false}, "odd.women.a22_29.q10": {"truth": 0.028735632183908046, "filled": 0.026111111111111113, "gap": -0.09577695539376885, "seed_sd": 0.0, "tolerance": 0.0684466769976442, "passes": false}, "odd.women.a22_29.q50": {"truth": 0.191131498470948, "filled": 0.17813455657492355, "gap": -0.07042246429654586, "seed_sd": 0.0, "tolerance": 0.021849929026002995, "passes": false}, "odd.women.a22_29.q90": {"truth": 0.46005509641873277, "filled": 0.43654434250764523, "gap": -0.05245630051303818, "seed_sd": 0.0, "tolerance": 0.020751504651450387, "passes": false}, "odd.women.a22_29.r1": {"truth": 0.7942016511246733, "filled": 0.8821403084868372, "gap": 0.08793865736216389, "seed_sd": 0.0, "tolerance": 0.006445547562698701, "passes": false}, "odd.women.a22_29.r2": {"truth": 0.6731585499668502, "filled": 0.8489459186770097, "gap": 0.17578736871015954, "seed_sd": 0.0, "tolerance": 0.013111971249341917, "passes": false}, "odd.women.a22_29.r3": {"truth": 0.5535614164842262, "filled": 0.5908983271961422, "gap": 0.03733691071191603, "seed_sd": 0.0, "tolerance": 0.012054115899361516, "passes": false}, "odd.women.a22_29.r4": {"truth": 0.552312691565311, "filled": 0.656690179347997, "gap": 0.10437748778268596, "seed_sd": 0.0, "tolerance": 0.01982762523559322, "passes": false}, "odd.women.a22_29.wint": {"truth": 0.08501885498800137, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.15037337545812676, "passes": false}, "odd.women.a22_29.zexit": {"truth": 0.41757828810020875, "filled": 0.06012526096033403, "gap": -1.9380419744064699, "seed_sd": 0.0, "tolerance": 0.041287246240518535, "passes": false}, "odd.women.a22_29.zint": {"truth": 0.0315547703180212, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09342121518593281, "passes": false}, "odd.women.a22_74.atcap": {"truth": 0.029398307023564402, "filled": 0.011993248463319055, "gap": -0.8965932250573987, "seed_sd": 0.0, "tolerance": 0.06709882226274566, "passes": false}, "odd.women.a22_74.level": {"truth": 0.2372854094489719, "filled": 0.23838128106693332, "gap": 0.0046077372189554655, "seed_sd": 0.0, "tolerance": 0.010902725750872375, "passes": true}, "odd.women.a22_74.q10": {"truth": 0.03777777777777778, "filled": 0.027186544342507644, "gap": -0.3289988826632908, "seed_sd": 0.0, "tolerance": 0.031129845660342752, "passes": false}, "odd.women.a22_74.q50": {"truth": 0.24875621890547264, "filled": 0.22111111111111112, "gap": -0.11780803596888845, "seed_sd": 0.0, "tolerance": 0.012092425838785583, "passes": false}, "odd.women.a22_74.q90": {"truth": 0.6444444444444445, "filled": 0.6086206896551725, "gap": -0.057193386823322534, "seed_sd": 0.0, "tolerance": 0.013338685039058783, "passes": false}, "odd.women.a22_74.r1": {"truth": 0.8812684347612898, "filled": 0.9394347478682354, "gap": 0.058166313106945644, "seed_sd": 0.0, "tolerance": 0.002475922564645967, "passes": false}, "odd.women.a22_74.r2": {"truth": 0.8056395091372208, "filled": 0.8970038287402264, "gap": 0.09136431960300562, "seed_sd": 0.0, "tolerance": 0.004772836686607259, "passes": false}, "odd.women.a22_74.r3": {"truth": 0.7468316241328095, "filled": 0.7676196288675307, "gap": 0.0207880047347212, "seed_sd": 0.0, "tolerance": 0.005171487109756781, "passes": false}, "odd.women.a22_74.r4": {"truth": 0.7116744894807431, "filled": 0.7579270771987335, "gap": 0.04625258771799046, "seed_sd": 0.0, "tolerance": 0.007666262733486471, "passes": false}, "odd.women.a22_74.wint": {"truth": 0.051544874036178946, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07133033012317215, "passes": false}, "odd.women.a22_74.zexit": {"truth": 0.45845752128015943, "filled": 0.01236869003547409, "gap": -3.6126993579608797, "seed_sd": 0.0, "tolerance": 0.016039678883419197, "passes": false}, "odd.women.a22_74.zint": {"truth": 0.022201124403524123, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04476234327525859, "passes": false}, "odd.women.a30_44.atcap": {"truth": 0.034619662054697874, "filled": 0.013831551720358612, "gap": -0.917469449162823, "seed_sd": 0.0, "tolerance": 0.08510646139536093, "passes": false}, "odd.women.a30_44.level": {"truth": 0.2579924798566703, "filled": 0.2586995070802617, "gap": 0.0027367471637131935, "seed_sd": 0.0, "tolerance": 0.01578879525774269, "passes": true}, "odd.women.a30_44.q10": {"truth": 0.04228855721393035, "filled": 0.030303030303030304, "gap": -0.33326881690367527, "seed_sd": 0.0, "tolerance": 0.04573918824793483, "passes": false}, "odd.women.a30_44.q50": {"truth": 0.26859504132231404, "filled": 0.23850574712643677, "gap": -0.11881141571682718, "seed_sd": 0.0, "tolerance": 0.018488627942446087, "passes": false}, "odd.women.a30_44.q90": {"truth": 0.6716417910447762, "filled": 0.636085626911315, "gap": -0.054391961575289194, "seed_sd": 0.0, "tolerance": 0.018806950089013175, "passes": false}, "odd.women.a30_44.r1": {"truth": 0.888241419643356, "filled": 0.9468343895814233, "gap": 0.058592969938067285, "seed_sd": 0.0, "tolerance": 0.0034876380830913202, "passes": false}, "odd.women.a30_44.r2": {"truth": 0.8242277470219269, "filled": 0.9096684494259086, "gap": 0.08544070240398172, "seed_sd": 0.0, "tolerance": 0.007078708407960211, "passes": false}, "odd.women.a30_44.r3": {"truth": 0.7685608654079126, "filled": 0.7927366027222539, "gap": 0.024175737314341306, "seed_sd": 0.0, "tolerance": 0.0073196290204859075, "passes": false}, "odd.women.a30_44.r4": {"truth": 0.7490435327097834, "filled": 0.7920119748169124, "gap": 0.04296844210712902, "seed_sd": 0.0, "tolerance": 0.010187042430903364, "passes": false}, "odd.women.a30_44.wint": {"truth": 0.06073485056210584, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.10093703169083498, "passes": false}, "odd.women.a30_44.zexit": {"truth": 0.45066799061202384, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.024719366356404402, "passes": false}, "odd.women.a30_44.zint": {"truth": 0.022222222222222223, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07029190354044747, "passes": false}, "odd.women.a45_59.atcap": {"truth": 0.04053483605515179, "filled": 0.01771406256616308, "gap": -0.827802934506876, "seed_sd": 0.0, "tolerance": 0.09107631963722376, "passes": false}, "odd.women.a45_59.level": {"truth": 0.28104586423928846, "filled": 0.28103084273355944, "gap": -5.345002041967639e-05, "seed_sd": 0.0, "tolerance": 0.01605667435242628, "passes": true}, "odd.women.a45_59.q10": {"truth": 0.053516819571865444, "filled": 0.03581267217630854, "gap": -0.40169418683552927, "seed_sd": 0.0, "tolerance": 0.051646333169991114, "passes": false}, "odd.women.a45_59.q50": {"truth": 0.29338842975206614, "filled": 0.26666666666666666, "gap": -0.09549799086694866, "seed_sd": 0.0, "tolerance": 0.0182941133196311, "passes": false}, "odd.women.a45_59.q90": {"truth": 0.7300275482093664, "filled": 0.6954022988505747, "gap": -0.0485917453391595, "seed_sd": 0.0, "tolerance": 0.0208569023153089, "passes": false}, "odd.women.a45_59.r1": {"truth": 0.9101006329634369, "filled": 0.95644972873215, "gap": 0.046349095768713044, "seed_sd": 0.0, "tolerance": 0.0037361334549905496, "passes": false}, "odd.women.a45_59.r2": {"truth": 0.8498173267452248, "filled": 0.9175157606106729, "gap": 0.06769843386544805, "seed_sd": 0.0, "tolerance": 0.0074184353060767995, "passes": false}, "odd.women.a45_59.r3": {"truth": 0.8096130207660378, "filled": 0.8275574170028757, "gap": 0.017944396236837856, "seed_sd": 0.0, "tolerance": 0.007740061426920219, "passes": false}, "odd.women.a45_59.r4": {"truth": 0.7620167777938601, "filled": 0.7929407426303564, "gap": 0.030923964836496287, "seed_sd": 0.0, "tolerance": 0.012512985825002505, "passes": false}, "odd.women.a45_59.wint": {"truth": 0.050115932427956277, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.1271041647107106, "passes": false}, "odd.women.a45_59.zexit": {"truth": 0.4693136110029843, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.030024476078798885, "passes": false}, "odd.women.a45_59.zint": {"truth": 0.01632776600720909, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09017841563442361, "passes": false}, "odd.women.a60_74.atcap": {"truth": 0.01812919896640827, "filled": 0.006878104700038212, "gap": -0.9691807066740816, "seed_sd": 0.0}, "odd.women.a60_74.level": {"truth": 0.12926943691615103, "filled": 0.12931157523148915, "gap": 0.00032591964399220075, "seed_sd": 0.0, "tolerance": 0.04408494999775865, "passes": true}, "odd.women.a60_74.q10": {"truth": 0.017412935323383085, "filled": 0.009938837920489297, "gap": -0.5607632349918994, "seed_sd": 0.0, "tolerance": 0.13013445963696416, "passes": false}, "odd.women.a60_74.q50": {"truth": 0.15517241379310345, "filled": 0.12241379310344827, "gap": -0.23712979328894956, "seed_sd": 0.0, "tolerance": 0.05365060769995596, "passes": false}, "odd.women.a60_74.q90": {"truth": 0.5360696517412935, "filled": 0.47055555555555556, "gap": -0.13035007015698297, "seed_sd": 0.0, "tolerance": 0.04960336282899433, "passes": false}, "odd.women.a60_74.r1": {"truth": 0.8610627385333514, "filled": 0.9308441096943134, "gap": 0.06978137116096206, "seed_sd": 0.0, "tolerance": 0.009751654703991025, "passes": false}, "odd.women.a60_74.r2": {"truth": 0.7722738686836694, "filled": 0.8536224830395716, "gap": 0.08134861435590213, "seed_sd": 0.0, "tolerance": 0.022174026993056747, "passes": false}, "odd.women.a60_74.r3": {"truth": 0.7141078828415282, "filled": 0.7255034910997732, "gap": 0.011395608258245038, "seed_sd": 0.0, "tolerance": 0.018479646240433655, "passes": true}, "odd.women.a60_74.r4": {"truth": 0.6501204938024134, "filled": 0.6629121666010215, "gap": 0.012791672798608045, "seed_sd": 0.0, "tolerance": 0.036314307016843086, "passes": true}, "odd.women.a60_74.wint": {"truth": 0.023107482596493787, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.women.a60_74.zexit": {"truth": 0.5155483759303063, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.03896610501214408, "passes": false}, "odd.women.a60_74.zint": {"truth": 0.022290698541288734, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.women.b1936_1940.aime_p10": {"truth": 82.29999999999995, "filled": 83.0, "gap": 0.00846950011357439, "seed_sd": 0.0, "tolerance": 0.32371200850504067, "passes": true}, "odd.women.b1936_1940.aime_p25": {"truth": 287.75, "filled": 287.75, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.20139115911902655, "passes": true}, "odd.women.b1936_1940.aime_p50": {"truth": 786.0, "filled": 786.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.12524071173214943, "passes": true}, "odd.women.b1936_1940.aime_p75": {"truth": 1566.0, "filled": 1565.0, "gap": -0.0006387735764947777, "seed_sd": 0.0, "tolerance": 0.10110924143579125, "passes": true}, "odd.women.b1936_1940.aime_p90": {"truth": 2438.7000000000007, "filled": 2435.7000000000007, "gap": -0.0012309208841250197, "seed_sd": 0.0, "tolerance": 0.08967006892401257, "passes": true}, "odd.women.b1936_1945.aime_p10": {"truth": 98.0, "filled": 98.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.1995625804481066, "passes": true}, "odd.women.b1936_1945.aime_p25": {"truth": 345.0, "filled": 344.0, "gap": -0.0029027596579620507, "seed_sd": 0.0, "tolerance": 0.11108088073804002, "passes": true}, "odd.women.b1936_1945.aime_p50": {"truth": 932.0, "filled": 930.0, "gap": -0.002148228538289665, "seed_sd": 0.0, "tolerance": 0.07343389253317807, "passes": true}, "odd.women.b1936_1945.aime_p75": {"truth": 1865.0, "filled": 1863.0, "gap": -0.0010729614763267392, "seed_sd": 0.0, "tolerance": 0.06283602704614394, "passes": true}, "odd.women.b1936_1945.aime_p90": {"truth": 2920.0, "filled": 2924.2000000000007, "gap": 0.001437322721010048, "seed_sd": 0.0, "tolerance": 0.0536209031123137, "passes": true}, "odd.women.b1941_1945.aime_p10": {"truth": 115.0, "filled": 116.0, "gap": 0.008658062743114314, "seed_sd": 0.0, "tolerance": 0.2647761737592696, "passes": true}, "odd.women.b1941_1945.aime_p25": {"truth": 399.0, "filled": 398.0, "gap": -0.0025094116054260596, "seed_sd": 0.0, "tolerance": 0.15691889792263147, "passes": true}, "odd.women.b1941_1945.aime_p50": {"truth": 1077.0, "filled": 1075.0, "gap": -0.001858736594625654, "seed_sd": 0.0, "tolerance": 0.08797933476607916, "passes": true}, "odd.women.b1941_1945.aime_p75": {"truth": 2116.0, "filled": 2116.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.07205053130731136, "passes": true}, "odd.women.b1941_1945.aime_p90": {"truth": 3268.6000000000004, "filled": 3270.6000000000004, "gap": 0.0006116956393338313, "seed_sd": 0.0, "tolerance": 0.06548294022251817, "passes": true}, "odd.women.b1946_1955.paime_p10": {"truth": 158.0, "filled": 159.0, "gap": 0.00630916919326463, "seed_sd": 0.0, "tolerance": 0.1085059910489472, "passes": true}, "odd.women.b1946_1955.paime_p25": {"truth": 519.0, "filled": 519.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.062889445073137, "passes": true}, "odd.women.b1946_1955.paime_p50": {"truth": 1313.0, "filled": 1313.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.04313158587991037, "passes": true}, "odd.women.b1946_1955.paime_p75": {"truth": 2486.0, "filled": 2486.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.03158401312013406, "passes": true}, "odd.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3779.0, "gap": -0.0002645852641443014, "seed_sd": 0.0, "tolerance": 0.02998098721497899, "passes": true}, "odd.women.b1946_1980.paime_p10": {"truth": 109.0, "filled": 108.0, "gap": -0.009216655104923532, "seed_sd": 0.0, "tolerance": 0.0500438197853722, "passes": true}, "odd.women.b1946_1980.paime_p25": {"truth": 309.0, "filled": 311.0, "gap": 0.006451635281488066, "seed_sd": 0.0, "tolerance": 0.030021121307637916, "passes": true}, "odd.women.b1946_1980.paime_p50": {"truth": 758.0, "filled": 759.0, "gap": 0.001318391753258652, "seed_sd": 0.0, "tolerance": 0.02113414008122686, "passes": true}, "odd.women.b1946_1980.paime_p75": {"truth": 1590.0, "filled": 1590.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.018434974631993433, "passes": true}, "odd.women.b1946_1980.paime_p90": {"truth": 2696.0, "filled": 2697.0, "gap": 0.00037085110753221073, "seed_sd": 0.0, "tolerance": 0.018162919937725473, "passes": true}, "odd.women.b1956_1965.paime_p10": {"truth": 140.0, "filled": 140.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0987946317785998, "passes": true}, "odd.women.b1956_1965.paime_p25": {"truth": 426.0, "filled": 425.0, "gap": -0.0023501773449536856, "seed_sd": 0.0, "tolerance": 0.05702004966656188, "passes": true}, "odd.women.b1956_1965.paime_p50": {"truth": 1000.0, "filled": 1000.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0353321966023362, "passes": true}, "odd.women.b1956_1965.paime_p75": {"truth": 1842.0, "filled": 1846.0, "gap": 0.0021691982475449123, "seed_sd": 0.0, "tolerance": 0.026005856561860215, "passes": true}, "odd.women.b1956_1965.paime_p90": {"truth": 2793.0, "filled": 2795.0999999999985, "gap": 0.0007515971793115028, "seed_sd": 0.0, "tolerance": 0.027293077620429543, "passes": true}, "odd.women.b1966_1980.paime_p10": {"truth": 80.0, "filled": 80.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.06283109225080236, "passes": true}, "odd.women.b1966_1980.paime_p25": {"truth": 211.0, "filled": 211.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.033157126538046186, "passes": true}, "odd.women.b1966_1980.paime_p50": {"truth": 463.0, "filled": 466.0, "gap": 0.006458580039411466, "seed_sd": 0.0, "tolerance": 0.02601990880890944, "passes": true}, "odd.women.b1966_1980.paime_p75": {"truth": 866.75, "filled": 870.0, "gap": 0.003742627083497041, "seed_sd": 0.0, "tolerance": 0.02735691355887302, "passes": true}, "odd.women.b1966_1980.paime_p90": {"truth": 1387.0, "filled": 1389.0, "gap": 0.0014409224395128817, "seed_sd": 0.0, "tolerance": 0.027821321010049294, "passes": true}}}, "two_sided": {"passes": false, "n_gating": 183, "n_failing": 99, "cells": {"odd.men.a22_29.atcap": {"truth": 0.014676604027739744, "filled": 0.003963318801164396, "gap": -1.309172907789832, "seed_sd": 0.0, "tolerance": 0.18023370223284765, "passes": false}, "odd.men.a22_29.level": {"truth": 0.23657185599421368, "filled": 0.23815412941641792, "gap": 0.006666074030311719, "seed_sd": 0.0, "tolerance": 0.02171146931539837, "passes": true}, "odd.men.a22_29.q10": {"truth": 0.040229885057471264, "filled": 0.03611111111111111, "gap": -0.10800952382940343, "seed_sd": 0.0, "tolerance": 0.07329266358707544, "passes": false}, "odd.men.a22_29.q50": {"truth": 0.24222222222222223, "filled": 0.22201492537313433, "gap": -0.08711096742405089, "seed_sd": 0.0, "tolerance": 0.022980444394274636, "passes": false}, "odd.men.a22_29.q90": {"truth": 0.5661157024793388, "filled": 0.5323266998341619, "gap": -0.06154108035996475, "seed_sd": 0.0, "tolerance": 0.02353059621893112, "passes": false}, "odd.men.a22_29.r1": {"truth": 0.8114342592359591, "filled": 0.9077680195907103, "gap": 0.09633376035475116, "seed_sd": 0.0, "tolerance": 0.007273490161268873, "passes": false}, "odd.men.a22_29.r2": {"truth": 0.7090594261869381, "filled": 0.8581398280676814, "gap": 0.14908040188074334, "seed_sd": 0.0, "tolerance": 0.014356564488812985, "passes": false}, "odd.men.a22_29.r3": {"truth": 0.6020834292042946, "filled": 0.6504088491993362, "gap": 0.048325419995041585, "seed_sd": 0.0, "tolerance": 0.013280307041936976, "passes": false}, "odd.men.a22_29.r4": {"truth": 0.5990582702554191, "filled": 0.6892222646133446, "gap": 0.09016399435792544, "seed_sd": 0.0, "tolerance": 0.02119963449400435, "passes": false}, "odd.men.a22_29.wint": {"truth": 0.08288543140028289, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.men.a22_29.zexit": {"truth": 0.4099387400566883, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.052311207438163365, "passes": false}, "odd.men.a22_29.zint": {"truth": 0.027589420573962787, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.12140160693630424, "passes": false}, "odd.men.a22_74.atcap": {"truth": 0.10371778137953785, "filled": 0.046035707021086794, "gap": -0.8122562350030691, "seed_sd": 0.0, "tolerance": 0.03690628859850606, "passes": false}, "odd.men.a22_74.level": {"truth": 0.35325148977531445, "filled": 0.3538288994241502, "gap": 0.0016332224341244483, "seed_sd": 0.0, "tolerance": 0.01106822547283481, "passes": true}, "odd.men.a22_74.q10": {"truth": 0.06111111111111111, "filled": 0.04611111111111111, "gap": -0.2816397579958183, "seed_sd": 0.0, "tolerance": 0.03462864626687699, "passes": false}, "odd.men.a22_74.q50": {"truth": 0.37052341597796146, "filled": 0.3367816091954023, "gap": -0.09548196740860526, "seed_sd": 0.0, "tolerance": 0.012463773427145929, "passes": false}, "odd.men.a22_74.q90": {"truth": 1.0, "filled": 0.9266169154228856, "gap": -0.07621505079940682, "seed_sd": 0.0, "tolerance": 0.003451823067723808, "passes": false}, "odd.men.a22_74.r1": {"truth": 0.8935990156167083, "filled": 0.9490345370051931, "gap": 0.05543552138848484, "seed_sd": 0.0, "tolerance": 0.00258049589094515, "passes": false}, "odd.men.a22_74.r2": {"truth": 0.8291027164330249, "filled": 0.9084834928643107, "gap": 0.07938077643128583, "seed_sd": 0.0, "tolerance": 0.004576681103770273, "passes": false}, "odd.men.a22_74.r3": {"truth": 0.7806492619651291, "filled": 0.8016928703964389, "gap": 0.021043608431309813, "seed_sd": 0.0, "tolerance": 0.004722018684826323, "passes": false}, "odd.men.a22_74.r4": {"truth": 0.7414979369803346, "filled": 0.7832471318914956, "gap": 0.041749194911161025, "seed_sd": 0.0, "tolerance": 0.007000982741779954, "passes": false}, "odd.men.a22_74.wint": {"truth": 0.05745341614906832, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07337302886151653, "passes": false}, "odd.men.a22_74.zexit": {"truth": 0.44752843148294114, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.01916127382099423, "passes": false}, "odd.men.a22_74.zint": {"truth": 0.019892968610837895, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.05481211657295162, "passes": false}, "odd.men.a30_44.atcap": {"truth": 0.10574805846620577, "filled": 0.04650078322293776, "gap": -0.82159030214472, "seed_sd": 0.0, "tolerance": 0.049095580366705964, "passes": false}, "odd.men.a30_44.level": {"truth": 0.3982899917120792, "filled": 0.39901752453379963, "gap": 0.001824974702530513, "seed_sd": 0.0, "tolerance": 0.015220021463125325, "passes": true}, "odd.men.a30_44.q10": {"truth": 0.08706467661691543, "filled": 0.06778606965174129, "gap": -0.25029454038016086, "seed_sd": 0.0, "tolerance": 0.047669064141889934, "passes": false}, "odd.men.a30_44.q50": {"truth": 0.41333333333333333, "filled": 0.38685015290519875, "gap": -0.06621695367851432, "seed_sd": 0.0, "tolerance": 0.016660415997544607, "passes": false}, "odd.men.a30_44.q90": {"truth": 1.0, "filled": 0.9390547263681592, "gap": -0.06288151992994163, "seed_sd": 0.0, "tolerance": 0.0037405831290436837, "passes": false}, "odd.men.a30_44.r1": {"truth": 0.9000861304840623, "filled": 0.9541064209919281, "gap": 0.05402029050786583, "seed_sd": 0.0, "tolerance": 0.003556208950302802, "passes": false}, "odd.men.a30_44.r2": {"truth": 0.8479233224100629, "filled": 0.9269323018569282, "gap": 0.07900897944686536, "seed_sd": 0.0, "tolerance": 0.006500650570758907, "passes": false}, "odd.men.a30_44.r3": {"truth": 0.8030903893396322, "filled": 0.8276167677302705, "gap": 0.024526378390638315, "seed_sd": 0.0, "tolerance": 0.006087370619188444, "passes": false}, "odd.men.a30_44.r4": {"truth": 0.7869108820616492, "filled": 0.832720868870299, "gap": 0.04580998680864978, "seed_sd": 0.0, "tolerance": 0.009218001294372115, "passes": false}, "odd.men.a30_44.wint": {"truth": 0.07239145006330527, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.12001713872675526, "passes": false}, "odd.men.a30_44.zexit": {"truth": 0.4342024282437196, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.031953074094816486, "passes": false}, "odd.men.a30_44.zint": {"truth": 0.017933882760031997, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.08989768311292742, "passes": false}, "odd.men.a45_59.atcap": {"truth": 0.15826463477931516, "filled": 0.07469452108789909, "gap": -0.750861791701275, "seed_sd": 0.0, "tolerance": 0.04711609868499565, "passes": false}, "odd.men.a45_59.level": {"truth": 0.4279852808128974, "filled": 0.4276293059548939, "gap": -0.0008320916544616308, "seed_sd": 0.0, "tolerance": 0.015834008317210692, "passes": true}, "odd.men.a45_59.q10": {"truth": 0.08868501529051988, "filled": 0.06592039800995025, "gap": -0.29664301471606525, "seed_sd": 0.0, "tolerance": 0.055701610600039544, "passes": false}, "odd.men.a45_59.q50": {"truth": 0.4701492537313433, "filled": 0.4345730027548209, "gap": -0.07868625928414963, "seed_sd": 0.0, "tolerance": 0.020385938992396834, "passes": false}, "odd.men.a45_59.q90": {"truth": 1.0, "filled": 0.9958677685950413, "gap": -0.004140792666031388, "seed_sd": 0.0}, "odd.men.a45_59.r1": {"truth": 0.9077661818197654, "filled": 0.9568133880651853, "gap": 0.049047206245419916, "seed_sd": 0.0, "tolerance": 0.004032009924506552, "passes": false}, "odd.men.a45_59.r2": {"truth": 0.850421444941529, "filled": 0.9210343284479916, "gap": 0.07061288350646255, "seed_sd": 0.0, "tolerance": 0.0074262368998509335, "passes": false}, "odd.men.a45_59.r3": {"truth": 0.8146289743583418, "filled": 0.8339593243046802, "gap": 0.01933034994633842, "seed_sd": 0.0, "tolerance": 0.007198396861260139, "passes": false}, "odd.men.a45_59.r4": {"truth": 0.7598303885400162, "filled": 0.796798027663896, "gap": 0.03696763912387979, "seed_sd": 0.0, "tolerance": 0.012107705260459355, "passes": false}, "odd.men.a45_59.wint": {"truth": 0.05825168693530555, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.13344110142481172, "passes": false}, "odd.men.a45_59.zexit": {"truth": 0.45185706741770815, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.036314919811016276, "passes": false}, "odd.men.a45_59.zint": {"truth": 0.015835938476928848, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09287476169914739, "passes": false}, "odd.men.a60_74.atcap": {"truth": 0.10024796315975912, "filled": 0.038797709400772665, "gap": -0.949285539547915, "seed_sd": 0.0, "tolerance": 0.12472307183935011, "passes": false}, "odd.men.a60_74.level": {"truth": 0.20453937198442498, "filled": 0.20537657883929814, "gap": 0.004084778927416988, "seed_sd": 0.0, "tolerance": 0.0522939828282486, "passes": true}, "odd.men.a60_74.q10": {"truth": 0.01529051987767584, "filled": 0.009950248756218905, "gap": -0.4296354690359774, "seed_sd": 0.0, "tolerance": 0.20186846491463123, "passes": false}, "odd.men.a60_74.q50": {"truth": 0.21666666666666667, "filled": 0.17507645259938837, "gap": -0.21313732370234062, "seed_sd": 0.0, "tolerance": 0.07446457229432804, "passes": false}, "odd.men.a60_74.q90": {"truth": 1.0, "filled": 0.8222222222222222, "gap": -0.19574457712609536, "seed_sd": 0.0, "tolerance": 0.04069424979546094, "passes": false}, "odd.men.a60_74.r1": {"truth": 0.8712782553777247, "filled": 0.9364895683685154, "gap": 0.06521131299079064, "seed_sd": 0.0, "tolerance": 0.009750532797445479, "passes": false}, "odd.men.a60_74.r2": {"truth": 0.7888015745113253, "filled": 0.8691600366633899, "gap": 0.08035846215206466, "seed_sd": 0.0, "tolerance": 0.020757725353477027, "passes": false}, "odd.men.a60_74.r3": {"truth": 0.7283746799072385, "filled": 0.7453822584843823, "gap": 0.017007578577143856, "seed_sd": 0.0, "tolerance": 0.02064690204069302, "passes": true}, "odd.men.a60_74.r4": {"truth": 0.6781549627202678, "filled": 0.6972394023844436, "gap": 0.019084439664175834, "seed_sd": 0.0, "tolerance": 0.03820786124425089, "passes": true}, "odd.men.a60_74.wint": {"truth": 0.03084255319148936, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.men.a60_74.zexit": {"truth": 0.5047452182800409, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04434932710092839, "passes": false}, "odd.men.a60_74.zint": {"truth": 0.027132152458672565, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.17758159695027936, "passes": false}, "odd.men.b1936_1940.aime_p10": {"truth": 346.0, "filled": 345.0, "gap": -0.0028943580263645075, "seed_sd": 0.0, "tolerance": 0.30817736367768905, "passes": true}, "odd.men.b1936_1940.aime_p25": {"truth": 1131.0, "filled": 1131.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.15436566295506535, "passes": true}, "odd.men.b1936_1940.aime_p50": {"truth": 2430.0, "filled": 2426.0, "gap": -0.0016474468305984757, "seed_sd": 0.0, "tolerance": 0.0822816707411384, "passes": true}, "odd.men.b1936_1940.aime_p75": {"truth": 3582.0, "filled": 3576.0, "gap": -0.001676446327252279, "seed_sd": 0.0, "tolerance": 0.04915519601813279, "passes": true}, "odd.men.b1936_1940.aime_p90": {"truth": 4382.0, "filled": 4372.0, "gap": -0.0022846708589838727, "seed_sd": 0.0, "tolerance": 0.03801502836217192, "passes": true}, "odd.men.b1936_1945.aime_p10": {"truth": 342.8000000000002, "filled": 341.0, "gap": -0.005264709440107929, "seed_sd": 0.0, "tolerance": 0.19387427864829543, "passes": true}, "odd.men.b1936_1945.aime_p25": {"truth": 1152.0, "filled": 1152.5, "gap": 0.00043393361496679717, "seed_sd": 0.0, "tolerance": 0.09928513681983075, "passes": true}, "odd.men.b1936_1945.aime_p50": {"truth": 2625.0, "filled": 2621.0, "gap": -0.001524971702317579, "seed_sd": 0.0, "tolerance": 0.05013200755321697, "passes": true}, "odd.men.b1936_1945.aime_p75": {"truth": 3991.5, "filled": 3985.5, "gap": -0.0015043252178763566, "seed_sd": 0.0, "tolerance": 0.030282584009309735, "passes": true}, "odd.men.b1936_1945.aime_p90": {"truth": 5075.0, "filled": 5060.0, "gap": -0.0029600416284765174, "seed_sd": 0.0, "tolerance": 0.029246187918705275, "passes": true}, "odd.men.b1941_1945.aime_p10": {"truth": 338.0, "filled": 336.0, "gap": -0.005934735519814716, "seed_sd": 0.0, "tolerance": 0.2434370939708618, "passes": true}, "odd.men.b1941_1945.aime_p25": {"truth": 1174.25, "filled": 1176.25, "gap": 0.0017017659924842832, "seed_sd": 0.0, "tolerance": 0.15717605214643726, "passes": true}, "odd.men.b1941_1945.aime_p50": {"truth": 2821.5, "filled": 2818.0, "gap": -0.0012412449505694312, "seed_sd": 0.0, "tolerance": 0.06794319334096698, "passes": true}, "odd.men.b1941_1945.aime_p75": {"truth": 4421.0, "filled": 4414.0, "gap": -0.0015846070095602016, "seed_sd": 0.0, "tolerance": 0.039611439643930886, "passes": true}, "odd.men.b1941_1945.aime_p90": {"truth": 5528.300000000001, "filled": 5520.300000000001, "gap": -0.0014481475296577173, "seed_sd": 0.0, "tolerance": 0.024092287379584104, "passes": true}, "odd.men.b1946_1955.paime_p10": {"truth": 308.0, "filled": 309.0, "gap": 0.0032414939241718344, "seed_sd": 0.0, "tolerance": 0.11641286886749184, "passes": true}, "odd.men.b1946_1955.paime_p25": {"truth": 1026.0, "filled": 1028.0, "gap": 0.001947420284395207, "seed_sd": 0.0, "tolerance": 0.07923354406873329, "passes": true}, "odd.men.b1946_1955.paime_p50": {"truth": 2614.0, "filled": 2609.0, "gap": -0.0019146090474384536, "seed_sd": 0.0, "tolerance": 0.03884145918889322, "passes": true}, "odd.men.b1946_1955.paime_p75": {"truth": 4390.0, "filled": 4389.0, "gap": -0.0002278163809830147, "seed_sd": 0.0, "tolerance": 0.02464882648778213, "passes": true}, "odd.men.b1946_1955.paime_p90": {"truth": 5810.0, "filled": 5808.0, "gap": -0.00034429334132468625, "seed_sd": 0.0, "tolerance": 0.01935089604023114, "passes": true}, "odd.men.b1946_1980.paime_p10": {"truth": 175.0, "filled": 176.0, "gap": 0.005698021114636909, "seed_sd": 0.0, "tolerance": 0.05432313062129484, "passes": true}, "odd.men.b1946_1980.paime_p25": {"truth": 487.0, "filled": 488.0, "gap": 0.002051282770557883, "seed_sd": 0.0, "tolerance": 0.03064646902486962, "passes": true}, "odd.men.b1946_1980.paime_p50": {"truth": 1245.0, "filled": 1247.0, "gap": 0.0016051367812286443, "seed_sd": 0.0, "tolerance": 0.025757396517852603, "passes": true}, "odd.men.b1946_1980.paime_p75": {"truth": 2644.0, "filled": 2645.0, "gap": 0.0003781433208231988, "seed_sd": 0.0, "tolerance": 0.0206417529534006, "passes": true}, "odd.men.b1946_1980.paime_p90": {"truth": 4310.0, "filled": 4310.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.01797004074133569, "passes": true}, "odd.men.b1956_1965.paime_p10": {"truth": 253.0, "filled": 253.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.10325340800222761, "passes": true}, "odd.men.b1956_1965.paime_p25": {"truth": 765.0, "filled": 767.0, "gap": 0.0026109675407202104, "seed_sd": 0.0, "tolerance": 0.06714959239176221, "passes": true}, "odd.men.b1956_1965.paime_p50": {"truth": 1764.0, "filled": 1765.0, "gap": 0.000566732800660219, "seed_sd": 0.0, "tolerance": 0.03783036836459788, "passes": true}, "odd.men.b1956_1965.paime_p75": {"truth": 2958.0, "filled": 2961.0, "gap": 0.0010136848308457402, "seed_sd": 0.0, "tolerance": 0.02582244416868233, "passes": true}, "odd.men.b1956_1965.paime_p90": {"truth": 4111.0, "filled": 4110.0, "gap": -0.00024327940759860667, "seed_sd": 0.0, "tolerance": 0.022664845011086624, "passes": true}, "odd.men.b1966_1980.paime_p10": {"truth": 112.0, "filled": 112.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.07763263566160054, "passes": true}, "odd.men.b1966_1980.paime_p25": {"truth": 303.25, "filled": 304.0, "gap": 0.0024701535820623732, "seed_sd": 0.0, "tolerance": 0.04145124387911718, "passes": true}, "odd.men.b1966_1980.paime_p50": {"truth": 655.0, "filled": 658.0, "gap": 0.0045696956900656005, "seed_sd": 0.0, "tolerance": 0.02859495381248372, "passes": true}, "odd.men.b1966_1980.paime_p75": {"truth": 1225.0, "filled": 1228.0, "gap": 0.0024459857282606023, "seed_sd": 0.0, "tolerance": 0.024541637055534683, "passes": true}, "odd.men.b1966_1980.paime_p90": {"truth": 1888.9000000000015, "filled": 1889.9000000000015, "gap": 0.0005292685632181104, "seed_sd": 0.0, "tolerance": 0.025554311796701996, "passes": true}, "odd.women.a22_29.atcap": {"truth": 0.005728386357627567, "filled": 0.0014145096609551587, "gap": -1.3986509364070692, "seed_sd": 0.0}, "odd.women.a22_29.level": {"truth": 0.1854665923300802, "filled": 0.18583978254257602, "gap": 0.002010147756208447, "seed_sd": 0.0, "tolerance": 0.018517427407013797, "passes": true}, "odd.women.a22_29.q10": {"truth": 0.028735632183908046, "filled": 0.02522935779816514, "gap": -0.13012958377023498, "seed_sd": 0.0, "tolerance": 0.0684466769976442, "passes": false}, "odd.women.a22_29.q50": {"truth": 0.191131498470948, "filled": 0.16944444444444445, "gap": -0.1204365531219469, "seed_sd": 0.0, "tolerance": 0.021849929026002995, "passes": false}, "odd.women.a22_29.q90": {"truth": 0.46005509641873277, "filled": 0.42838544878911294, "gap": -0.07132288554928723, "seed_sd": 0.0, "tolerance": 0.020751504651450387, "passes": false}, "odd.women.a22_29.r1": {"truth": 0.7942016511246733, "filled": 0.8964267346621329, "gap": 0.10222508353745952, "seed_sd": 0.0, "tolerance": 0.006445547562698701, "passes": false}, "odd.women.a22_29.r2": {"truth": 0.6731585499668502, "filled": 0.8403968435266072, "gap": 0.16723829355975695, "seed_sd": 0.0, "tolerance": 0.013111971249341917, "passes": false}, "odd.women.a22_29.r3": {"truth": 0.5535614164842262, "filled": 0.6047968332303263, "gap": 0.051235416746100104, "seed_sd": 0.0, "tolerance": 0.012054115899361516, "passes": false}, "odd.women.a22_29.r4": {"truth": 0.552312691565311, "filled": 0.649332972302572, "gap": 0.09702028073726099, "seed_sd": 0.0, "tolerance": 0.01982762523559322, "passes": false}, "odd.women.a22_29.wint": {"truth": 0.08501885498800137, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.15037337545812676, "passes": false}, "odd.women.a22_29.zexit": {"truth": 0.41757828810020875, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.041287246240518535, "passes": false}, "odd.women.a22_29.zint": {"truth": 0.0315547703180212, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09342121518593281, "passes": false}, "odd.women.a22_74.atcap": {"truth": 0.029398307023564402, "filled": 0.011954671808208356, "gap": -0.8998149401825817, "seed_sd": 0.0, "tolerance": 0.06709882226274566, "passes": false}, "odd.women.a22_74.level": {"truth": 0.2372854094489719, "filled": 0.2376347653333952, "gap": 0.001471219653273792, "seed_sd": 0.0, "tolerance": 0.010902725750872375, "passes": true}, "odd.women.a22_74.q10": {"truth": 0.03777777777777778, "filled": 0.026859504132231406, "gap": -0.3411013105469456, "seed_sd": 0.0, "tolerance": 0.031129845660342752, "passes": false}, "odd.women.a22_74.q50": {"truth": 0.24875621890547264, "filled": 0.21888888888888888, "gap": -0.12790913195539244, "seed_sd": 0.0, "tolerance": 0.012092425838785583, "passes": false}, "odd.women.a22_74.q90": {"truth": 0.6444444444444445, "filled": 0.6072222222222222, "gap": -0.059493795923871884, "seed_sd": 0.0, "tolerance": 0.013338685039058783, "passes": false}, "odd.women.a22_74.r1": {"truth": 0.8812684347612898, "filled": 0.9414914928166868, "gap": 0.06022305805539696, "seed_sd": 0.0, "tolerance": 0.002475922564645967, "passes": false}, "odd.women.a22_74.r2": {"truth": 0.8056395091372208, "filled": 0.894980676078455, "gap": 0.08934116694123417, "seed_sd": 0.0, "tolerance": 0.004772836686607259, "passes": false}, "odd.women.a22_74.r3": {"truth": 0.7468316241328095, "filled": 0.769997332317142, "gap": 0.0231657081843325, "seed_sd": 0.0, "tolerance": 0.005171487109756781, "passes": false}, "odd.women.a22_74.r4": {"truth": 0.7116744894807431, "filled": 0.7556167205497889, "gap": 0.04394223106904582, "seed_sd": 0.0, "tolerance": 0.007666262733486471, "passes": false}, "odd.women.a22_74.wint": {"truth": 0.051544874036178946, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07133033012317215, "passes": false}, "odd.women.a22_74.zexit": {"truth": 0.45845752128015943, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.016039678883419197, "passes": false}, "odd.women.a22_74.zint": {"truth": 0.022201124403524123, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04476234327525859, "passes": false}, "odd.women.a30_44.atcap": {"truth": 0.034619662054697874, "filled": 0.013831551720358612, "gap": -0.917469449162823, "seed_sd": 0.0, "tolerance": 0.08510646139536093, "passes": false}, "odd.women.a30_44.level": {"truth": 0.2579924798566703, "filled": 0.2586995070802617, "gap": 0.0027367471637131935, "seed_sd": 0.0, "tolerance": 0.01578879525774269, "passes": true}, "odd.women.a30_44.q10": {"truth": 0.04228855721393035, "filled": 0.030303030303030304, "gap": -0.33326881690367527, "seed_sd": 0.0, "tolerance": 0.04573918824793483, "passes": false}, "odd.women.a30_44.q50": {"truth": 0.26859504132231404, "filled": 0.23850574712643677, "gap": -0.11881141571682718, "seed_sd": 0.0, "tolerance": 0.018488627942446087, "passes": false}, "odd.women.a30_44.q90": {"truth": 0.6716417910447762, "filled": 0.636085626911315, "gap": -0.054391961575289194, "seed_sd": 0.0, "tolerance": 0.018806950089013175, "passes": false}, "odd.women.a30_44.r1": {"truth": 0.888241419643356, "filled": 0.9468346426950571, "gap": 0.058593223051701115, "seed_sd": 0.0, "tolerance": 0.0034876380830913202, "passes": false}, "odd.women.a30_44.r2": {"truth": 0.8242277470219269, "filled": 0.9096682274710212, "gap": 0.08544048044909425, "seed_sd": 0.0, "tolerance": 0.007078708407960211, "passes": false}, "odd.women.a30_44.r3": {"truth": 0.7685608654079126, "filled": 0.792737244038737, "gap": 0.024176378630824447, "seed_sd": 0.0, "tolerance": 0.0073196290204859075, "passes": false}, "odd.women.a30_44.r4": {"truth": 0.7490435327097834, "filled": 0.792012608430596, "gap": 0.042969075720812544, "seed_sd": 0.0, "tolerance": 0.010187042430903364, "passes": false}, "odd.women.a30_44.wint": {"truth": 0.06073485056210584, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.10093703169083498, "passes": false}, "odd.women.a30_44.zexit": {"truth": 0.45066799061202384, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.024719366356404402, "passes": false}, "odd.women.a30_44.zint": {"truth": 0.022222222222222223, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07029190354044747, "passes": false}, "odd.women.a45_59.atcap": {"truth": 0.04053483605515179, "filled": 0.01771406256616308, "gap": -0.827802934506876, "seed_sd": 0.0, "tolerance": 0.09107631963722376, "passes": false}, "odd.women.a45_59.level": {"truth": 0.28104586423928846, "filled": 0.28103084273355944, "gap": -5.345002041967639e-05, "seed_sd": 0.0, "tolerance": 0.01605667435242628, "passes": true}, "odd.women.a45_59.q10": {"truth": 0.053516819571865444, "filled": 0.03581267217630854, "gap": -0.40169418683552927, "seed_sd": 0.0, "tolerance": 0.051646333169991114, "passes": false}, "odd.women.a45_59.q50": {"truth": 0.29338842975206614, "filled": 0.26666666666666666, "gap": -0.09549799086694866, "seed_sd": 0.0, "tolerance": 0.0182941133196311, "passes": false}, "odd.women.a45_59.q90": {"truth": 0.7300275482093664, "filled": 0.6954022988505747, "gap": -0.0485917453391595, "seed_sd": 0.0, "tolerance": 0.0208569023153089, "passes": false}, "odd.women.a45_59.r1": {"truth": 0.9101006329634369, "filled": 0.9564498180345706, "gap": 0.046349185071133725, "seed_sd": 0.0, "tolerance": 0.0037361334549905496, "passes": false}, "odd.women.a45_59.r2": {"truth": 0.8498173267452248, "filled": 0.9175149282013461, "gap": 0.06769760145612125, "seed_sd": 0.0, "tolerance": 0.0074184353060767995, "passes": false}, "odd.women.a45_59.r3": {"truth": 0.8096130207660378, "filled": 0.8275570329366764, "gap": 0.017944012170638568, "seed_sd": 0.0, "tolerance": 0.007740061426920219, "passes": false}, "odd.women.a45_59.r4": {"truth": 0.7620167777938601, "filled": 0.7929402009981047, "gap": 0.03092342320424457, "seed_sd": 0.0, "tolerance": 0.012512985825002505, "passes": false}, "odd.women.a45_59.wint": {"truth": 0.050115932427956277, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.1271041647107106, "passes": false}, "odd.women.a45_59.zexit": {"truth": 0.4693136110029843, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.030024476078798885, "passes": false}, "odd.women.a45_59.zint": {"truth": 0.01632776600720909, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09017841563442361, "passes": false}, "odd.women.a60_74.atcap": {"truth": 0.01812919896640827, "filled": 0.006878104700038212, "gap": -0.9691807066740816, "seed_sd": 0.0}, "odd.women.a60_74.level": {"truth": 0.12926943691615103, "filled": 0.12931157523148915, "gap": 0.00032591964399220075, "seed_sd": 0.0, "tolerance": 0.04408494999775865, "passes": true}, "odd.women.a60_74.q10": {"truth": 0.017412935323383085, "filled": 0.009938837920489297, "gap": -0.5607632349918994, "seed_sd": 0.0, "tolerance": 0.13013445963696416, "passes": false}, "odd.women.a60_74.q50": {"truth": 0.15517241379310345, "filled": 0.12241379310344827, "gap": -0.23712979328894956, "seed_sd": 0.0, "tolerance": 0.05365060769995596, "passes": false}, "odd.women.a60_74.q90": {"truth": 0.5360696517412935, "filled": 0.47055555555555556, "gap": -0.13035007015698297, "seed_sd": 0.0, "tolerance": 0.04960336282899433, "passes": false}, "odd.women.a60_74.r1": {"truth": 0.8610627385333514, "filled": 0.9308438442070122, "gap": 0.0697811056736608, "seed_sd": 0.0, "tolerance": 0.009751654703991025, "passes": false}, "odd.women.a60_74.r2": {"truth": 0.7722738686836694, "filled": 0.8536239349818906, "gap": 0.08135006629822117, "seed_sd": 0.0, "tolerance": 0.022174026993056747, "passes": false}, "odd.women.a60_74.r3": {"truth": 0.7141078828415282, "filled": 0.7255042892168905, "gap": 0.011396406375362322, "seed_sd": 0.0, "tolerance": 0.018479646240433655, "passes": true}, "odd.women.a60_74.r4": {"truth": 0.6501204938024134, "filled": 0.6629196573096237, "gap": 0.01279916350721022, "seed_sd": 0.0, "tolerance": 0.036314307016843086, "passes": true}, "odd.women.a60_74.wint": {"truth": 0.023107482596493787, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.women.a60_74.zexit": {"truth": 0.5155483759303063, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.03896610501214408, "passes": false}, "odd.women.a60_74.zint": {"truth": 0.022290698541288734, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.women.b1936_1940.aime_p10": {"truth": 82.29999999999995, "filled": 83.0, "gap": 0.00846950011357439, "seed_sd": 0.0, "tolerance": 0.32371200850504067, "passes": true}, "odd.women.b1936_1940.aime_p25": {"truth": 287.75, "filled": 287.75, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.20139115911902655, "passes": true}, "odd.women.b1936_1940.aime_p50": {"truth": 786.0, "filled": 786.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.12524071173214943, "passes": true}, "odd.women.b1936_1940.aime_p75": {"truth": 1566.0, "filled": 1565.0, "gap": -0.0006387735764947777, "seed_sd": 0.0, "tolerance": 0.10110924143579125, "passes": true}, "odd.women.b1936_1940.aime_p90": {"truth": 2438.7000000000007, "filled": 2435.7000000000007, "gap": -0.0012309208841250197, "seed_sd": 0.0, "tolerance": 0.08967006892401257, "passes": true}, "odd.women.b1936_1945.aime_p10": {"truth": 98.0, "filled": 98.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.1995625804481066, "passes": true}, "odd.women.b1936_1945.aime_p25": {"truth": 345.0, "filled": 344.0, "gap": -0.0029027596579620507, "seed_sd": 0.0, "tolerance": 0.11108088073804002, "passes": true}, "odd.women.b1936_1945.aime_p50": {"truth": 932.0, "filled": 930.0, "gap": -0.002148228538289665, "seed_sd": 0.0, "tolerance": 0.07343389253317807, "passes": true}, "odd.women.b1936_1945.aime_p75": {"truth": 1865.0, "filled": 1863.0, "gap": -0.0010729614763267392, "seed_sd": 0.0, "tolerance": 0.06283602704614394, "passes": true}, "odd.women.b1936_1945.aime_p90": {"truth": 2920.0, "filled": 2924.2000000000007, "gap": 0.001437322721010048, "seed_sd": 0.0, "tolerance": 0.0536209031123137, "passes": true}, "odd.women.b1941_1945.aime_p10": {"truth": 115.0, "filled": 116.0, "gap": 0.008658062743114314, "seed_sd": 0.0, "tolerance": 0.2647761737592696, "passes": true}, "odd.women.b1941_1945.aime_p25": {"truth": 399.0, "filled": 398.0, "gap": -0.0025094116054260596, "seed_sd": 0.0, "tolerance": 0.15691889792263147, "passes": true}, "odd.women.b1941_1945.aime_p50": {"truth": 1077.0, "filled": 1075.0, "gap": -0.001858736594625654, "seed_sd": 0.0, "tolerance": 0.08797933476607916, "passes": true}, "odd.women.b1941_1945.aime_p75": {"truth": 2116.0, "filled": 2116.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.07205053130731136, "passes": true}, "odd.women.b1941_1945.aime_p90": {"truth": 3268.6000000000004, "filled": 3270.6000000000004, "gap": 0.0006116956393338313, "seed_sd": 0.0, "tolerance": 0.06548294022251817, "passes": true}, "odd.women.b1946_1955.paime_p10": {"truth": 158.0, "filled": 159.0, "gap": 0.00630916919326463, "seed_sd": 0.0, "tolerance": 0.1085059910489472, "passes": true}, "odd.women.b1946_1955.paime_p25": {"truth": 519.0, "filled": 519.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.062889445073137, "passes": true}, "odd.women.b1946_1955.paime_p50": {"truth": 1313.0, "filled": 1313.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.04313158587991037, "passes": true}, "odd.women.b1946_1955.paime_p75": {"truth": 2486.0, "filled": 2486.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.03158401312013406, "passes": true}, "odd.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3779.0, "gap": -0.0002645852641443014, "seed_sd": 0.0, "tolerance": 0.02998098721497899, "passes": true}, "odd.women.b1946_1980.paime_p10": {"truth": 109.0, "filled": 108.0, "gap": -0.009216655104923532, "seed_sd": 0.0, "tolerance": 0.0500438197853722, "passes": true}, "odd.women.b1946_1980.paime_p25": {"truth": 309.0, "filled": 310.0, "gap": 0.003231020581446309, "seed_sd": 0.0, "tolerance": 0.030021121307637916, "passes": true}, "odd.women.b1946_1980.paime_p50": {"truth": 758.0, "filled": 758.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.02113414008122686, "passes": true}, "odd.women.b1946_1980.paime_p75": {"truth": 1590.0, "filled": 1590.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.018434974631993433, "passes": true}, "odd.women.b1946_1980.paime_p90": {"truth": 2696.0, "filled": 2697.0, "gap": 0.00037085110753221073, "seed_sd": 0.0, "tolerance": 0.018162919937725473, "passes": true}, "odd.women.b1956_1965.paime_p10": {"truth": 140.0, "filled": 140.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0987946317785998, "passes": true}, "odd.women.b1956_1965.paime_p25": {"truth": 426.0, "filled": 425.0, "gap": -0.0023501773449536856, "seed_sd": 0.0, "tolerance": 0.05702004966656188, "passes": true}, "odd.women.b1956_1965.paime_p50": {"truth": 1000.0, "filled": 1000.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0353321966023362, "passes": true}, "odd.women.b1956_1965.paime_p75": {"truth": 1842.0, "filled": 1846.0, "gap": 0.0021691982475449123, "seed_sd": 0.0, "tolerance": 0.026005856561860215, "passes": true}, "odd.women.b1956_1965.paime_p90": {"truth": 2793.0, "filled": 2795.0999999999985, "gap": 0.0007515971793115028, "seed_sd": 0.0, "tolerance": 0.027293077620429543, "passes": true}, "odd.women.b1966_1980.paime_p10": {"truth": 80.0, "filled": 80.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.06283109225080236, "passes": true}, "odd.women.b1966_1980.paime_p25": {"truth": 211.0, "filled": 211.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.033157126538046186, "passes": true}, "odd.women.b1966_1980.paime_p50": {"truth": 463.0, "filled": 464.0, "gap": 0.0021574981400211968, "seed_sd": 0.0, "tolerance": 0.02601990880890944, "passes": true}, "odd.women.b1966_1980.paime_p75": {"truth": 866.75, "filled": 867.0, "gap": 0.0002883922154088836, "seed_sd": 0.0, "tolerance": 0.02735691355887302, "passes": true}, "odd.women.b1966_1980.paime_p90": {"truth": 1387.0, "filled": 1388.0, "gap": 0.0007207207519188685, "seed_sd": 0.0, "tolerance": 0.027821321010049294, "passes": true}}}}, "primary": {"passes": false, "n_gating": 183, "n_failing": 7, "cells": {"odd.men.a22_29.atcap": {"truth": 0.014676604027739744, "filled": 0.015478991505216875, "gap": 0.053229054633069595, "seed_sd": 0.00022562078995838478, "tolerance": 0.18023370223284765, "passes": true}, "odd.men.a22_29.level": {"truth": 0.23657185599421368, "filled": 0.23717255395952153, "gap": 0.0025359593680274184, "seed_sd": 0.0002515217198283508, "tolerance": 0.02171146931539837, "passes": true}, "odd.men.a22_29.q10": {"truth": 0.040229885057471264, "filled": 0.041617456321049816, "gap": 0.03390957352930313, "seed_sd": 0.0003830112083455766, "tolerance": 0.07329266358707544, "passes": true}, "odd.men.a22_29.q50": {"truth": 0.24222222222222223, "filled": 0.24384832532234682, "gap": 0.006690836030639913, "seed_sd": 0.0005141419280494356, "tolerance": 0.022980444394274636, "passes": true}, "odd.men.a22_29.q90": {"truth": 0.5661157024793388, "filled": 0.5690728618295566, "gap": 0.005209999699271828, "seed_sd": 0.0011631471597934602, "tolerance": 0.02353059621893112, "passes": true}, "odd.men.a22_29.r1": {"truth": 0.8114342592359591, "filled": 0.7918666865747263, "gap": -0.01956757266123288, "seed_sd": 0.0008536254079533948, "tolerance": 0.007273490161268873, "passes": false}, "odd.men.a22_29.r2": {"truth": 0.7090594261869381, "filled": 0.7086612870078752, "gap": -0.0003981391790628397, "seed_sd": 0.0017514876624430678, "tolerance": 0.014356564488812985, "passes": true}, "odd.men.a22_29.r3": {"truth": 0.6020834292042946, "filled": 0.5897635473854892, "gap": -0.01231988181880539, "seed_sd": 0.001092216462324124, "tolerance": 0.013280307041936976, "passes": true}, "odd.men.a22_29.r4": {"truth": 0.5990582702554191, "filled": 0.5975865098462816, "gap": -0.0014717604091375458, "seed_sd": 0.001832401877059017, "tolerance": 0.02119963449400435, "passes": true}, "odd.men.a22_29.wint": {"truth": 0.08288543140028289, "filled": 0.09262022630834513, "gap": 0.11104823478796844, "seed_sd": 0.0025397861808929443}, "odd.men.a22_29.zexit": {"truth": 0.4099387400566883, "filled": 0.42119411173082194, "gap": 0.027086066376702744, "seed_sd": 0.003339605159177347, "tolerance": 0.052311207438163365, "passes": true}, "odd.men.a22_29.zint": {"truth": 0.027589420573962787, "filled": 0.032563791496609575, "gap": 0.16576859410855027, "seed_sd": 0.00038799743953147253, "tolerance": 0.12140160693630424, "passes": false}, "odd.men.a22_74.atcap": {"truth": 0.10371778137953785, "filled": 0.10405772152852044, "gap": 0.0032721899121184173, "seed_sd": 0.0001421100987836538, "tolerance": 0.03690628859850606, "passes": true}, "odd.men.a22_74.level": {"truth": 0.35325148977531445, "filled": 0.3536267889985615, "gap": 0.0010618497406973404, "seed_sd": 0.00015811254712585015, "tolerance": 0.01106822547283481, "passes": true}, "odd.men.a22_74.q10": {"truth": 0.06111111111111111, "filled": 0.06104982070649271, "gap": -0.0010034371684821686, "seed_sd": 0.0001901280444885172, "tolerance": 0.03462864626687699, "passes": true}, "odd.men.a22_74.q50": {"truth": 0.37052341597796146, "filled": 0.3714999618524454, "gap": 0.0026321177134208673, "seed_sd": 0.00027634709713943287, "tolerance": 0.012463773427145929, "passes": true}, "odd.men.a22_74.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.003451823067723808, "passes": true}, "odd.men.a22_74.r1": {"truth": 0.8935990156167083, "filled": 0.8910980192174813, "gap": -0.0025009963992269624, "seed_sd": 0.0001975630980355699, "tolerance": 0.00258049589094515, "passes": true}, "odd.men.a22_74.r2": {"truth": 0.8291027164330249, "filled": 0.8277733291882811, "gap": -0.0013293872447438515, "seed_sd": 0.00042543786212773815, "tolerance": 0.004576681103770273, "passes": true}, "odd.men.a22_74.r3": {"truth": 0.7806492619651291, "filled": 0.7768047594054334, "gap": -0.003844502559695706, "seed_sd": 0.00021680874531272687, "tolerance": 0.004722018684826323, "passes": true}, "odd.men.a22_74.r4": {"truth": 0.7414979369803346, "filled": 0.7366250578680585, "gap": -0.004872879112276074, "seed_sd": 0.0006218303381177339, "tolerance": 0.007000982741779954, "passes": true}, "odd.men.a22_74.wint": {"truth": 0.05745341614906832, "filled": 0.059355590062111795, "gap": 0.03257183917090156, "seed_sd": 0.0007236709175504945, "tolerance": 0.07337302886151653, "passes": true}, "odd.men.a22_74.zexit": {"truth": 0.44752843148294114, "filled": 0.44796616372030174, "gap": 0.0009776324158056182, "seed_sd": 0.001535188345791419, "tolerance": 0.01916127382099423, "passes": true}, "odd.men.a22_74.zint": {"truth": 0.019892968610837895, "filled": 0.02071057380286708, "gap": 0.04027804843040528, "seed_sd": 0.0001501775466059639, "tolerance": 0.05481211657295162, "passes": true}, "odd.men.a30_44.atcap": {"truth": 0.10574805846620577, "filled": 0.10532608891605424, "gap": -0.003998311660010412, "seed_sd": 0.00031281429807503787, "tolerance": 0.049095580366705964, "passes": true}, "odd.men.a30_44.level": {"truth": 0.3982899917120792, "filled": 0.3988284449313848, "gap": 0.0013509994912968004, "seed_sd": 0.0002028287101855178, "tolerance": 0.015220021463125325, "passes": true}, "odd.men.a30_44.q10": {"truth": 0.08706467661691543, "filled": 0.08610589761196305, "gap": -0.011073345529330147, "seed_sd": 0.0004749422573459383, "tolerance": 0.047669064141889934, "passes": true}, "odd.men.a30_44.q50": {"truth": 0.41333333333333333, "filled": 0.41458838788433666, "gap": 0.0030318216812166288, "seed_sd": 0.00034599819975770856, "tolerance": 0.016660415997544607, "passes": true}, "odd.men.a30_44.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0037405831290436837, "passes": true}, "odd.men.a30_44.r1": {"truth": 0.9000861304840623, "filled": 0.8998266022341059, "gap": -0.00025952824995634227, "seed_sd": 0.0004413185664793287, "tolerance": 0.003556208950302802, "passes": true}, "odd.men.a30_44.r2": {"truth": 0.8479233224100629, "filled": 0.8483810977099816, "gap": 0.0004577752999187501, "seed_sd": 0.000997098561784646, "tolerance": 0.006500650570758907, "passes": true}, "odd.men.a30_44.r3": {"truth": 0.8030903893396322, "filled": 0.7992399426798853, "gap": -0.0038504466597468756, "seed_sd": 0.00048283750320468534, "tolerance": 0.006087370619188444, "passes": true}, "odd.men.a30_44.r4": {"truth": 0.7869108820616492, "filled": 0.7810755971849026, "gap": -0.00583528487674656, "seed_sd": 0.0010191541897079349, "tolerance": 0.009218001294372115, "passes": true}, "odd.men.a30_44.wint": {"truth": 0.07239145006330527, "filled": 0.07317904222834587, "gap": 0.010820872247405244, "seed_sd": 0.0017316235445208798, "tolerance": 0.12001713872675526, "passes": true}, "odd.men.a30_44.zexit": {"truth": 0.4342024282437196, "filled": 0.4297935433335199, "gap": -0.01020588827807345, "seed_sd": 0.002303556443411403, "tolerance": 0.031953074094816486, "passes": true}, "odd.men.a30_44.zint": {"truth": 0.017933882760031997, "filled": 0.017023794020798837, "gap": -0.05207980146296265, "seed_sd": 0.00021688098899260444, "tolerance": 0.08989768311292742, "passes": true}, "odd.men.a45_59.atcap": {"truth": 0.15826463477931516, "filled": 0.15934572375910133, "gap": 0.0068076693717182835, "seed_sd": 0.0003390418371717721, "tolerance": 0.04711609868499565, "passes": true}, "odd.men.a45_59.level": {"truth": 0.4279852808128974, "filled": 0.42780971681639934, "gap": -0.00041029452017671275, "seed_sd": 0.00024109620070850895, "tolerance": 0.015834008317210692, "passes": true}, "odd.men.a45_59.q10": {"truth": 0.08868501529051988, "filled": 0.08761303120469977, "gap": -0.012161193147874894, "seed_sd": 0.00042934976933140727, "tolerance": 0.055701610600039544, "passes": true}, "odd.men.a45_59.q50": {"truth": 0.4701492537313433, "filled": 0.4705012588693065, "gap": 0.0007484291980476288, "seed_sd": 0.000414975117828889, "tolerance": 0.020385938992396834, "passes": true}, "odd.men.a45_59.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0}, "odd.men.a45_59.r1": {"truth": 0.9077661818197654, "filled": 0.9078114869048537, "gap": 4.530508508826525e-05, "seed_sd": 0.00034430581774453007, "tolerance": 0.004032009924506552, "passes": true}, "odd.men.a45_59.r2": {"truth": 0.850421444941529, "filled": 0.8484356528621042, "gap": -0.001985792079424842, "seed_sd": 0.0006903711344554875, "tolerance": 0.0074262368998509335, "passes": true}, "odd.men.a45_59.r3": {"truth": 0.8146289743583418, "filled": 0.8121562172502159, "gap": -0.0024727571081258892, "seed_sd": 0.00036683181976336675, "tolerance": 0.007198396861260139, "passes": true}, "odd.men.a45_59.r4": {"truth": 0.7598303885400162, "filled": 0.756280816003992, "gap": -0.0035495725360241703, "seed_sd": 0.0014138669035669439, "tolerance": 0.012107705260459355, "passes": true}, "odd.men.a45_59.wint": {"truth": 0.05825168693530555, "filled": 0.0575464145476726, "gap": -0.012181220579572383, "seed_sd": 0.0013702186370924433, "tolerance": 0.13344110142481172, "passes": true}, "odd.men.a45_59.zexit": {"truth": 0.45185706741770815, "filled": 0.45060728744939266, "gap": -0.0027697066604916998, "seed_sd": 0.002801554802201966, "tolerance": 0.036314919811016276, "passes": true}, "odd.men.a45_59.zint": {"truth": 0.015835938476928848, "filled": 0.01581418031761911, "gap": -0.0013749182351325828, "seed_sd": 0.00022747480774057554, "tolerance": 0.09287476169914739, "passes": true}, "odd.men.a60_74.atcap": {"truth": 0.10024796315975912, "filled": 0.09946648841993258, "gap": -0.00782596073680164, "seed_sd": 0.0004015553201233456, "tolerance": 0.12472307183935011, "passes": true}, "odd.men.a60_74.level": {"truth": 0.20453937198442498, "filled": 0.20540302537010527, "gap": 0.004213541556197242, "seed_sd": 0.00035587347264424323, "tolerance": 0.0522939828282486, "passes": true}, "odd.men.a60_74.q10": {"truth": 0.01529051987767584, "filled": 0.015350881208514536, "gap": 0.003939859587278605, "seed_sd": 0.00029774282812945784, "tolerance": 0.20186846491463123, "passes": true}, "odd.men.a60_74.q50": {"truth": 0.21666666666666667, "filled": 0.2233566033417258, "gap": 0.030409538134932967, "seed_sd": 0.001070762123976171, "tolerance": 0.07446457229432804, "passes": true}, "odd.men.a60_74.q90": {"truth": 1.0, "filled": 0.9959601739528496, "gap": -0.004048008188114695, "seed_sd": 0.0009839142641652522, "tolerance": 0.04069424979546094, "passes": true}, "odd.men.a60_74.r1": {"truth": 0.8712782553777247, "filled": 0.8633097014979182, "gap": -0.00796855387980655, "seed_sd": 0.001064453719918088, "tolerance": 0.009750532797445479, "passes": true}, "odd.men.a60_74.r2": {"truth": 0.7888015745113253, "filled": 0.7830136813243864, "gap": -0.005787893186938842, "seed_sd": 0.0018046466076060675, "tolerance": 0.020757725353477027, "passes": true}, "odd.men.a60_74.r3": {"truth": 0.7283746799072385, "filled": 0.7250001781709396, "gap": -0.0033745017362988294, "seed_sd": 0.000782393180230533, "tolerance": 0.02064690204069302, "passes": true}, "odd.men.a60_74.r4": {"truth": 0.6781549627202678, "filled": 0.6692115763481665, "gap": -0.008943386372101236, "seed_sd": 0.0036651414112914395, "tolerance": 0.03820786124425089, "passes": true}, "odd.men.a60_74.wint": {"truth": 0.03084255319148936, "filled": 0.03232170212765957, "gap": 0.04684356352844654, "seed_sd": 0.0008910456258281209}, "odd.men.a60_74.zexit": {"truth": 0.5047452182800409, "filled": 0.5044313038399767, "gap": -0.0006221200024142393, "seed_sd": 0.003042999000388235, "tolerance": 0.04434932710092839, "passes": true}, "odd.men.a60_74.zint": {"truth": 0.027132152458672565, "filled": 0.0301380441207314, "gap": 0.10506883573756909, "seed_sd": 0.0007776869923929043, "tolerance": 0.17758159695027936, "passes": true}, "odd.men.b1936_1940.aime_p10": {"truth": 346.0, "filled": 345.15, "gap": -0.002459669908239981, "seed_sd": 1.2258187382102497, "tolerance": 0.30817736367768905, "passes": true}, "odd.men.b1936_1940.aime_p25": {"truth": 1131.0, "filled": 1129.85, "gap": -0.001017316583746819, "seed_sd": 1.1367080817685316, "tolerance": 0.15436566295506535, "passes": true}, "odd.men.b1936_1940.aime_p50": {"truth": 2430.0, "filled": 2428.85, "gap": -0.00047336304741829593, "seed_sd": 1.1821033884786183, "tolerance": 0.0822816707411384, "passes": true}, "odd.men.b1936_1940.aime_p75": {"truth": 3582.0, "filled": 3580.4, "gap": -0.0004467776238730181, "seed_sd": 1.729009451741235, "tolerance": 0.04915519601813279, "passes": true}, "odd.men.b1936_1940.aime_p90": {"truth": 4382.0, "filled": 4378.0, "gap": -0.0009132420726043478, "seed_sd": 2.2710998958306754, "tolerance": 0.03801502836217192, "passes": true}, "odd.men.b1936_1945.aime_p10": {"truth": 342.8000000000002, "filled": 340.81000000000006, "gap": -0.005822049475948887, "seed_sd": 0.6820248490611044, "tolerance": 0.19387427864829543, "passes": true}, "odd.men.b1936_1945.aime_p25": {"truth": 1152.0, "filled": 1153.625, "gap": 0.0014095963299034509, "seed_sd": 1.190698600778021, "tolerance": 0.09928513681983075, "passes": true}, "odd.men.b1936_1945.aime_p50": {"truth": 2625.0, "filled": 2623.35, "gap": -0.0006287690624144915, "seed_sd": 1.598519051464429, "tolerance": 0.05013200755321697, "passes": true}, "odd.men.b1936_1945.aime_p75": {"truth": 3991.5, "filled": 3994.9, "gap": 0.0008514475121224052, "seed_sd": 2.1860803664140644, "tolerance": 0.030282584009309735, "passes": true}, "odd.men.b1936_1945.aime_p90": {"truth": 5075.0, "filled": 5068.030000000001, "gap": -0.001374342991608657, "seed_sd": 2.1406098491588508, "tolerance": 0.029246187918705275, "passes": true}, "odd.men.b1941_1945.aime_p10": {"truth": 338.0, "filled": 336.46000000000004, "gap": -0.004566624192004376, "seed_sd": 1.8222022534920688, "tolerance": 0.2434370939708618, "passes": true}, "odd.men.b1941_1945.aime_p25": {"truth": 1174.25, "filled": 1176.5, "gap": 0.0019142832603122883, "seed_sd": 1.8009500416812174, "tolerance": 0.15717605214643726, "passes": true}, "odd.men.b1941_1945.aime_p50": {"truth": 2821.5, "filled": 2822.425, "gap": 0.0003277860737984639, "seed_sd": 2.341136700970616, "tolerance": 0.06794319334096698, "passes": true}, "odd.men.b1941_1945.aime_p75": {"truth": 4421.0, "filled": 4422.65, "gap": 0.00037314910000851853, "seed_sd": 2.286113686909224, "tolerance": 0.039611439643930886, "passes": true}, "odd.men.b1941_1945.aime_p90": {"truth": 5528.300000000001, "filled": 5531.785000000001, "gap": 0.0006301940926025651, "seed_sd": 2.464970374541801, "tolerance": 0.024092287379584104, "passes": true}, "odd.men.b1946_1955.paime_p10": {"truth": 308.0, "filled": 308.0, "gap": 0.0, "seed_sd": 0.6488856845230502, "tolerance": 0.11641286886749184, "passes": true}, "odd.men.b1946_1955.paime_p25": {"truth": 1026.0, "filled": 1027.9, "gap": 0.0018501392881615786, "seed_sd": 1.5861240410775606, "tolerance": 0.07923354406873329, "passes": true}, "odd.men.b1946_1955.paime_p50": {"truth": 2614.0, "filled": 2611.8, "gap": -0.0008419763978597672, "seed_sd": 2.375311613906246, "tolerance": 0.03884145918889322, "passes": true}, "odd.men.b1946_1955.paime_p75": {"truth": 4390.0, "filled": 4389.45, "gap": -0.00012529258682825173, "seed_sd": 2.874113135523631, "tolerance": 0.02464882648778213, "passes": true}, "odd.men.b1946_1955.paime_p90": {"truth": 5810.0, "filled": 5808.47, "gap": -0.0002633737503892064, "seed_sd": 3.151123442102177, "tolerance": 0.01935089604023114, "passes": true}, "odd.men.b1946_1980.paime_p10": {"truth": 175.0, "filled": 174.95, "gap": -0.00028575510981720953, "seed_sd": 0.5104177855340405, "tolerance": 0.05432313062129484, "passes": true}, "odd.men.b1946_1980.paime_p25": {"truth": 487.0, "filled": 487.15, "gap": 0.00030796078876083044, "seed_sd": 0.48936048492959283, "tolerance": 0.03064646902486962, "passes": true}, "odd.men.b1946_1980.paime_p50": {"truth": 1245.0, "filled": 1246.425, "gap": 0.0011439237828883009, "seed_sd": 0.5910516942393091, "tolerance": 0.025757396517852603, "passes": true}, "odd.men.b1946_1980.paime_p75": {"truth": 2644.0, "filled": 2645.525, "gap": 0.0005766113374088278, "seed_sd": 1.4439073450372888, "tolerance": 0.0206417529534006, "passes": true}, "odd.men.b1946_1980.paime_p90": {"truth": 4310.0, "filled": 4311.220000000001, "gap": 0.0002830225903398542, "seed_sd": 1.667206929977776, "tolerance": 0.01797004074133569, "passes": true}, "odd.men.b1956_1965.paime_p10": {"truth": 253.0, "filled": 252.76000000000005, "gap": -0.0009490668222653653, "seed_sd": 0.9483503682988224, "tolerance": 0.10325340800222761, "passes": true}, "odd.men.b1956_1965.paime_p25": {"truth": 765.0, "filled": 767.9, "gap": 0.00378368250996175, "seed_sd": 1.252366181526625, "tolerance": 0.06714959239176221, "passes": true}, "odd.men.b1956_1965.paime_p50": {"truth": 1764.0, "filled": 1765.1, "gap": 0.0006233884194966066, "seed_sd": 1.9708400559953587, "tolerance": 0.03783036836459788, "passes": true}, "odd.men.b1956_1965.paime_p75": {"truth": 2958.0, "filled": 2956.9, "gap": -0.00037194204895474314, "seed_sd": 2.125038699338017, "tolerance": 0.02582244416868233, "passes": true}, "odd.men.b1956_1965.paime_p90": {"truth": 4111.0, "filled": 4111.450000000001, "gap": 0.00010945642732984595, "seed_sd": 1.6693837501499207, "tolerance": 0.022664845011086624, "passes": true}, "odd.men.b1966_1980.paime_p10": {"truth": 112.0, "filled": 112.4, "gap": 0.0035650661644970327, "seed_sd": 0.5982430416161189, "tolerance": 0.07763263566160054, "passes": true}, "odd.men.b1966_1980.paime_p25": {"truth": 303.25, "filled": 303.7625, "gap": 0.0016885982472416572, "seed_sd": 0.558869865277946, "tolerance": 0.04145124387911718, "passes": true}, "odd.men.b1966_1980.paime_p50": {"truth": 655.0, "filled": 656.55, "gap": 0.0023636166697622585, "seed_sd": 0.8413648060396307, "tolerance": 0.02859495381248372, "passes": true}, "odd.men.b1966_1980.paime_p75": {"truth": 1225.0, "filled": 1226.95, "gap": 0.0015905711055381744, "seed_sd": 1.0990426455975697, "tolerance": 0.024541637055534683, "passes": true}, "odd.men.b1966_1980.paime_p90": {"truth": 1888.9000000000015, "filled": 1889.7400000000002, "gap": 0.0004446044152581763, "seed_sd": 2.1283549367626273, "tolerance": 0.025554311796701996, "passes": true}, "odd.women.a22_29.atcap": {"truth": 0.005728386357627567, "filled": 0.005840506534186164, "gap": 0.019383650297911004, "seed_sd": 0.00014710922581249458}, "odd.women.a22_29.level": {"truth": 0.1854665923300802, "filled": 0.18547670162135138, "gap": 5.450585811095365e-05, "seed_sd": 0.000224942861263494, "tolerance": 0.018517427407013797, "passes": true}, "odd.women.a22_29.q10": {"truth": 0.028735632183908046, "filled": 0.02897917143511101, "gap": 0.00843945336115448, "seed_sd": 0.0002739223160825707, "tolerance": 0.0684466769976442, "passes": true}, "odd.women.a22_29.q50": {"truth": 0.191131498470948, "filled": 0.19147020675974674, "gap": 0.0017705534118206412, "seed_sd": 0.00042356535871290117, "tolerance": 0.021849929026002995, "passes": true}, "odd.women.a22_29.q90": {"truth": 0.46005509641873277, "filled": 0.46155634393835354, "gap": 0.003257878063804065, "seed_sd": 0.0007612439450409235, "tolerance": 0.020751504651450387, "passes": true}, "odd.women.a22_29.r1": {"truth": 0.7942016511246733, "filled": 0.7811893738584953, "gap": -0.01301227726617804, "seed_sd": 0.0009183067288461948, "tolerance": 0.006445547562698701, "passes": false}, "odd.women.a22_29.r2": {"truth": 0.6731585499668502, "filled": 0.6832118075212995, "gap": 0.010053257554449302, "seed_sd": 0.0015380755525540044, "tolerance": 0.013111971249341917, "passes": true}, "odd.women.a22_29.r3": {"truth": 0.5535614164842262, "filled": 0.5458382326973334, "gap": -0.00772318378689274, "seed_sd": 0.001022663276913233, "tolerance": 0.012054115899361516, "passes": true}, "odd.women.a22_29.r4": {"truth": 0.552312691565311, "filled": 0.5543544199888721, "gap": 0.0020417284235610955, "seed_sd": 0.0022237898651966616, "tolerance": 0.01982762523559322, "passes": true}, "odd.women.a22_29.wint": {"truth": 0.08501885498800137, "filled": 0.09672608844703463, "gap": 0.12901009825016718, "seed_sd": 0.002577622041541449, "tolerance": 0.15037337545812676, "passes": true}, "odd.women.a22_29.zexit": {"truth": 0.41757828810020875, "filled": 0.4201440501043841, "gap": 0.006125585793858135, "seed_sd": 0.0032598959894505576, "tolerance": 0.041287246240518535, "passes": true}, "odd.women.a22_29.zint": {"truth": 0.0315547703180212, "filled": 0.03659054770318021, "gap": 0.14806517134934172, "seed_sd": 0.0004775866131467123, "tolerance": 0.09342121518593281, "passes": false}, "odd.women.a22_74.atcap": {"truth": 0.029398307023564402, "filled": 0.02945745381492562, "gap": 0.002009890295517458, "seed_sd": 0.00011259923212691128, "tolerance": 0.06709882226274566, "passes": true}, "odd.women.a22_74.level": {"truth": 0.2372854094489719, "filled": 0.23753619967326758, "gap": 0.001056355661994468, "seed_sd": 8.268291240265756e-05, "tolerance": 0.010902725750872375, "passes": true}, "odd.women.a22_74.q10": {"truth": 0.03777777777777778, "filled": 0.037823300526436246, "gap": 0.0012042884885090643, "seed_sd": 0.0001235564277909059, "tolerance": 0.031129845660342752, "passes": true}, "odd.women.a22_74.q50": {"truth": 0.24875621890547264, "filled": 0.24890363927672238, "gap": 0.0005924543566777629, "seed_sd": 0.00017148945504414184, "tolerance": 0.012092425838785583, "passes": true}, "odd.women.a22_74.q90": {"truth": 0.6444444444444445, "filled": 0.6455617608911275, "gap": 0.0017322656611419296, "seed_sd": 0.0005288221041851096, "tolerance": 0.013338685039058783, "passes": true}, "odd.women.a22_74.r1": {"truth": 0.8812684347612898, "filled": 0.8785632456320756, "gap": -0.002705189129214247, "seed_sd": 0.00030066051021595305, "tolerance": 0.002475922564645967, "passes": false}, "odd.women.a22_74.r2": {"truth": 0.8056395091372208, "filled": 0.8086345514864973, "gap": 0.0029950423492765, "seed_sd": 0.0005474787133861371, "tolerance": 0.004772836686607259, "passes": true}, "odd.women.a22_74.r3": {"truth": 0.7468316241328095, "filled": 0.7425035732187585, "gap": -0.004328050914050974, "seed_sd": 0.0002996302075456965, "tolerance": 0.005171487109756781, "passes": true}, "odd.women.a22_74.r4": {"truth": 0.7116744894807431, "filled": 0.7070081700012946, "gap": -0.004666319479448511, "seed_sd": 0.0008952863437053086, "tolerance": 0.007666262733486471, "passes": true}, "odd.women.a22_74.wint": {"truth": 0.051544874036178946, "filled": 0.05278242947314182, "gap": 0.023725591419143655, "seed_sd": 0.0007715150904738836, "tolerance": 0.07133033012317215, "passes": true}, "odd.women.a22_74.zexit": {"truth": 0.45845752128015943, "filled": 0.4576647226063578, "gap": -0.0017307709249479997, "seed_sd": 0.0011185395892386197, "tolerance": 0.016039678883419197, "passes": true}, "odd.women.a22_74.zint": {"truth": 0.022201124403524123, "filled": 0.023460056043049, "gap": 0.05515629569622549, "seed_sd": 0.00019832742657188788, "tolerance": 0.04476234327525859, "passes": false}, "odd.women.a30_44.atcap": {"truth": 0.034619662054697874, "filled": 0.03434400495858832, "gap": -0.007994312835381212, "seed_sd": 0.0001719755329787735, "tolerance": 0.08510646139536093, "passes": true}, "odd.women.a30_44.level": {"truth": 0.2579924798566703, "filled": 0.25856952575486103, "gap": 0.002234179563932237, "seed_sd": 0.0001654878317954069, "tolerance": 0.01578879525774269, "passes": true}, "odd.women.a30_44.q10": {"truth": 0.04228855721393035, "filled": 0.04248416876478217, "gap": 0.004614972463625744, "seed_sd": 0.0002927857689515677, "tolerance": 0.04573918824793483, "passes": true}, "odd.women.a30_44.q50": {"truth": 0.26859504132231404, "filled": 0.26862745098039215, "gap": 0.00012065637080271863, "seed_sd": 0.00022554139509321257, "tolerance": 0.018488627942446087, "passes": true}, "odd.women.a30_44.q90": {"truth": 0.6716417910447762, "filled": 0.6719058518348973, "gap": 0.0003930799103709637, "seed_sd": 0.0005591699861576255, "tolerance": 0.018806950089013175, "passes": true}, "odd.women.a30_44.r1": {"truth": 0.888241419643356, "filled": 0.888364648373031, "gap": 0.00012322872967496235, "seed_sd": 0.00037398027635182656, "tolerance": 0.0034876380830913202, "passes": true}, "odd.women.a30_44.r2": {"truth": 0.8242277470219269, "filled": 0.8261260370432127, "gap": 0.0018982900212858311, "seed_sd": 0.0008564417910465683, "tolerance": 0.007078708407960211, "passes": true}, "odd.women.a30_44.r3": {"truth": 0.7685608654079126, "filled": 0.7647437596543007, "gap": -0.003817105753611827, "seed_sd": 0.00045573186883397765, "tolerance": 0.0073196290204859075, "passes": true}, "odd.women.a30_44.r4": {"truth": 0.7490435327097834, "filled": 0.7421544589183829, "gap": -0.0068890737914004685, "seed_sd": 0.001251592453325158, "tolerance": 0.010187042430903364, "passes": true}, "odd.women.a30_44.wint": {"truth": 0.06073485056210584, "filled": 0.06228544008774335, "gap": 0.02521001436438297, "seed_sd": 0.0012284904999587422, "tolerance": 0.10093703169083498, "passes": true}, "odd.women.a30_44.zexit": {"truth": 0.45066799061202384, "filled": 0.45179860985737497, "gap": 0.002505621451870721, "seed_sd": 0.002515913111566333, "tolerance": 0.024719366356404402, "passes": true}, "odd.women.a30_44.zint": {"truth": 0.022222222222222223, "filled": 0.022268645656248635, "gap": 0.002086875490999507, "seed_sd": 0.0003599125241648141, "tolerance": 0.07029190354044747, "passes": true}, "odd.women.a45_59.atcap": {"truth": 0.04053483605515179, "filled": 0.04080148917927712, "gap": 0.006556826330295085, "seed_sd": 0.00019608552018686722, "tolerance": 0.09107631963722376, "passes": true}, "odd.women.a45_59.level": {"truth": 0.28104586423928846, "filled": 0.2810774396033896, "gap": 0.00011234319558894867, "seed_sd": 0.00015511405161526974, "tolerance": 0.01605667435242628, "passes": true}, "odd.women.a45_59.q10": {"truth": 0.053516819571865444, "filled": 0.052449683375295666, "gap": -0.020141690886250174, "seed_sd": 0.00032583523000722845, "tolerance": 0.051646333169991114, "passes": true}, "odd.women.a45_59.q50": {"truth": 0.29338842975206614, "filled": 0.2937277790493629, "gap": 0.0011559869409125678, "seed_sd": 0.0002448197822313442, "tolerance": 0.0182941133196311, "passes": true}, "odd.women.a45_59.q90": {"truth": 0.7300275482093664, "filled": 0.7307142748149844, "gap": 0.0009402437109500283, "seed_sd": 0.0008138649714838994, "tolerance": 0.0208569023153089, "passes": true}, "odd.women.a45_59.r1": {"truth": 0.9101006329634369, "filled": 0.9087391750153332, "gap": -0.001361457948103717, "seed_sd": 0.000419730811063702, "tolerance": 0.0037361334549905496, "passes": true}, "odd.women.a45_59.r2": {"truth": 0.8498173267452248, "filled": 0.8515147642421521, "gap": 0.001697437496927301, "seed_sd": 0.0008879958346065681, "tolerance": 0.0074184353060767995, "passes": true}, "odd.women.a45_59.r3": {"truth": 0.8096130207660378, "filled": 0.8055548329310239, "gap": -0.004058187835013882, "seed_sd": 0.0005770848358180316, "tolerance": 0.007740061426920219, "passes": true}, "odd.women.a45_59.r4": {"truth": 0.7620167777938601, "filled": 0.7581822504292892, "gap": -0.0038345273645709055, "seed_sd": 0.0014198400183539773, "tolerance": 0.012512985825002505, "passes": true}, "odd.women.a45_59.wint": {"truth": 0.050115932427956277, "filled": 0.046954289499834385, "gap": -0.06516440543993252, "seed_sd": 0.001114473359463049, "tolerance": 0.1271041647107106, "passes": true}, "odd.women.a45_59.zexit": {"truth": 0.4693136110029843, "filled": 0.46718080965356173, "gap": -0.004554869713656262, "seed_sd": 0.002121600873384182, "tolerance": 0.030024476078798885, "passes": true}, "odd.women.a45_59.zint": {"truth": 0.01632776600720909, "filled": 0.01605233469994222, "gap": -0.017012791468189903, "seed_sd": 0.0003180980237889735, "tolerance": 0.09017841563442361, "passes": true}, "odd.women.a60_74.atcap": {"truth": 0.01812919896640827, "filled": 0.01870662717455454, "gap": 0.03135401441666019, "seed_sd": 0.00040142732607189924}, "odd.women.a60_74.level": {"truth": 0.12926943691615103, "filled": 0.1293852858038575, "gap": 0.0008957802450684227, "seed_sd": 0.00036005331017849995, "tolerance": 0.04408494999775865, "passes": true}, "odd.women.a60_74.q10": {"truth": 0.017412935323383085, "filled": 0.016837567711909668, "gap": -0.03360077619576973, "seed_sd": 0.0003065216899310214, "tolerance": 0.13013445963696416, "passes": true}, "odd.women.a60_74.q50": {"truth": 0.15517241379310345, "filled": 0.15723735408560308, "gap": 0.0132196274053622, "seed_sd": 0.0008103021569256636, "tolerance": 0.05365060769995596, "passes": true}, "odd.women.a60_74.q90": {"truth": 0.5360696517412935, "filled": 0.5337356374456398, "gap": -0.00436344449308268, "seed_sd": 0.0018056770118273073, "tolerance": 0.04960336282899433, "passes": true}, "odd.women.a60_74.r1": {"truth": 0.8610627385333514, "filled": 0.8496491895289273, "gap": -0.01141354900442404, "seed_sd": 0.001253637642873998, "tolerance": 0.009751654703991025, "passes": false}, "odd.women.a60_74.r2": {"truth": 0.7722738686836694, "filled": 0.7795050897584435, "gap": 0.00723122107477403, "seed_sd": 0.003077248521934921, "tolerance": 0.022174026993056747, "passes": true}, "odd.women.a60_74.r3": {"truth": 0.7141078828415282, "filled": 0.7020798077569574, "gap": -0.01202807508457071, "seed_sd": 0.00218141956853534, "tolerance": 0.018479646240433655, "passes": true}, "odd.women.a60_74.r4": {"truth": 0.6501204938024134, "filled": 0.6374646350510534, "gap": -0.012655858751359994, "seed_sd": 0.005366315338935656, "tolerance": 0.036314307016843086, "passes": true}, "odd.women.a60_74.wint": {"truth": 0.023107482596493787, "filled": 0.023204067500091116, "gap": 0.00417109958221884, "seed_sd": 0.0009728381028477894}, "odd.women.a60_74.zexit": {"truth": 0.5155483759303063, "filled": 0.5075809150175965, "gap": -0.01557500512456278, "seed_sd": 0.004491939020397144, "tolerance": 0.03896610501214408, "passes": true}, "odd.women.a60_74.zint": {"truth": 0.022290698541288734, "filled": 0.026766233443502895, "gap": 0.1829716612979282, "seed_sd": 0.0008912211575906045}, "odd.women.b1936_1940.aime_p10": {"truth": 82.29999999999995, "filled": 83.05, "gap": 0.009071728376279786, "seed_sd": 0.22360679774997896, "tolerance": 0.32371200850504067, "passes": true}, "odd.women.b1936_1940.aime_p25": {"truth": 287.75, "filled": 287.8375, "gap": 0.0003040371817455423, "seed_sd": 0.48851951423609846, "tolerance": 0.20139115911902655, "passes": true}, "odd.women.b1936_1940.aime_p50": {"truth": 786.0, "filled": 785.975, "gap": -3.1807121617433154e-05, "seed_sd": 0.7340407273943894, "tolerance": 0.12524071173214943, "passes": true}, "odd.women.b1936_1940.aime_p75": {"truth": 1566.0, "filled": 1566.675, "gap": 0.00043094161408152587, "seed_sd": 1.8745174817732921, "tolerance": 0.10110924143579125, "passes": true}, "odd.women.b1936_1940.aime_p90": {"truth": 2438.7000000000007, "filled": 2436.605, "gap": -0.0008594334627067823, "seed_sd": 3.0194936835939212, "tolerance": 0.08967006892401257, "passes": true}, "odd.women.b1936_1945.aime_p10": {"truth": 98.0, "filled": 98.13000000000002, "gap": 0.0013256515478303754, "seed_sd": 0.3197038102929258, "tolerance": 0.1995625804481066, "passes": true}, "odd.women.b1936_1945.aime_p25": {"truth": 345.0, "filled": 344.8, "gap": -0.0005798782418224846, "seed_sd": 0.6958523739384594, "tolerance": 0.11108088073804002, "passes": true}, "odd.women.b1936_1945.aime_p50": {"truth": 932.0, "filled": 930.0, "gap": -0.002148228538289665, "seed_sd": 1.2565617248750864, "tolerance": 0.07343389253317807, "passes": true}, "odd.women.b1936_1945.aime_p75": {"truth": 1865.0, "filled": 1863.5, "gap": -0.0008046131586025851, "seed_sd": 1.3178930553209385, "tolerance": 0.06283602704614394, "passes": true}, "odd.women.b1936_1945.aime_p90": {"truth": 2920.0, "filled": 2923.59, "gap": 0.001228696897506154, "seed_sd": 2.6663398922591086, "tolerance": 0.0536209031123137, "passes": true}, "odd.women.b1941_1945.aime_p10": {"truth": 115.0, "filled": 115.56000000000002, "gap": 0.004857747234784604, "seed_sd": 0.5716089940730976, "tolerance": 0.2647761737592696, "passes": true}, "odd.women.b1941_1945.aime_p25": {"truth": 399.0, "filled": 398.65, "gap": -0.0008775779413596752, "seed_sd": 0.8750939799154207, "tolerance": 0.15691889792263147, "passes": true}, "odd.women.b1941_1945.aime_p50": {"truth": 1077.0, "filled": 1074.3, "gap": -0.002510111483892352, "seed_sd": 1.218281792655455, "tolerance": 0.08797933476607916, "passes": true}, "odd.women.b1941_1945.aime_p75": {"truth": 2116.0, "filled": 2117.2, "gap": 0.0005669470056419712, "seed_sd": 2.647739850474263, "tolerance": 0.07205053130731136, "passes": true}, "odd.women.b1941_1945.aime_p90": {"truth": 3268.6000000000004, "filled": 3270.2, "gap": 0.0004893864415294047, "seed_sd": 4.079215610874249, "tolerance": 0.06548294022251817, "passes": true}, "odd.women.b1946_1955.paime_p10": {"truth": 158.0, "filled": 158.1, "gap": 0.0006327111884596448, "seed_sd": 0.6407232755171874, "tolerance": 0.1085059910489472, "passes": true}, "odd.women.b1946_1955.paime_p25": {"truth": 519.0, "filled": 519.5375, "gap": 0.001035109561267511, "seed_sd": 0.7533914548297761, "tolerance": 0.062889445073137, "passes": true}, "odd.women.b1946_1955.paime_p50": {"truth": 1313.0, "filled": 1313.1, "gap": 7.615856216336425e-05, "seed_sd": 1.3337718577107005, "tolerance": 0.04313158587991037, "passes": true}, "odd.women.b1946_1955.paime_p75": {"truth": 2486.0, "filled": 2487.075, "gap": 0.0004323280934812601, "seed_sd": 1.3280872672658541, "tolerance": 0.03158401312013406, "passes": true}, "odd.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3780.7699999999995, "gap": 0.00020368295892225774, "seed_sd": 3.2444446447169453, "tolerance": 0.02998098721497899, "passes": true}, "odd.women.b1946_1980.paime_p10": {"truth": 109.0, "filled": 107.9, "gap": -0.010143009965054794, "seed_sd": 0.30779350562554625, "tolerance": 0.0500438197853722, "passes": true}, "odd.women.b1946_1980.paime_p25": {"truth": 309.0, "filled": 309.2, "gap": 0.0006470398155213886, "seed_sd": 0.523148363780597, "tolerance": 0.030021121307637916, "passes": true}, "odd.women.b1946_1980.paime_p50": {"truth": 758.0, "filled": 757.575, "gap": -0.0005608432590138435, "seed_sd": 0.6742442162583931, "tolerance": 0.02113414008122686, "passes": true}, "odd.women.b1946_1980.paime_p75": {"truth": 1590.0, "filled": 1588.9, "gap": -0.0006920633199563042, "seed_sd": 0.7880689256524124, "tolerance": 0.018434974631993433, "passes": true}, "odd.women.b1946_1980.paime_p90": {"truth": 2696.0, "filled": 2699.1399999999994, "gap": 0.0011640107039063707, "seed_sd": 1.7863223025034836, "tolerance": 0.018162919937725473, "passes": true}, "odd.women.b1956_1965.paime_p10": {"truth": 140.0, "filled": 139.7, "gap": -0.0021451563463878998, "seed_sd": 0.47016234598162726, "tolerance": 0.0987946317785998, "passes": true}, "odd.women.b1956_1965.paime_p25": {"truth": 426.0, "filled": 425.3, "gap": -0.0016445440097827557, "seed_sd": 0.8645047258706176, "tolerance": 0.05702004966656188, "passes": true}, "odd.women.b1956_1965.paime_p50": {"truth": 1000.0, "filled": 999.975, "gap": -2.500031250463053e-05, "seed_sd": 1.2190915513308735, "tolerance": 0.0353321966023362, "passes": true}, "odd.women.b1956_1965.paime_p75": {"truth": 1842.0, "filled": 1845.2, "gap": 0.0017357348684132745, "seed_sd": 1.538112308540638, "tolerance": 0.026005856561860215, "passes": true}, "odd.women.b1956_1965.paime_p90": {"truth": 2793.0, "filled": 2796.055, "gap": 0.0010932081735646193, "seed_sd": 2.1636895876208633, "tolerance": 0.027293077620429543, "passes": true}, "odd.women.b1966_1980.paime_p10": {"truth": 80.0, "filled": 79.5, "gap": -0.006269613013595077, "seed_sd": 0.512989176042577, "tolerance": 0.06283109225080236, "passes": true}, "odd.women.b1966_1980.paime_p25": {"truth": 211.0, "filled": 210.6, "gap": -0.0018975337761917288, "seed_sd": 0.5982430416161189, "tolerance": 0.033157126538046186, "passes": true}, "odd.women.b1966_1980.paime_p50": {"truth": 463.0, "filled": 463.45, "gap": 0.0009714502356077404, "seed_sd": 0.8255779474818965, "tolerance": 0.02601990880890944, "passes": true}, "odd.women.b1966_1980.paime_p75": {"truth": 866.75, "filled": 867.0875, "gap": 0.0003893098450840071, "seed_sd": 0.840093196584384, "tolerance": 0.02735691355887302, "passes": true}, "odd.women.b1966_1980.paime_p90": {"truth": 1387.0, "filled": 1389.2, "gap": 0.0015849005550876427, "seed_sd": 1.5761378513048248, "tolerance": 0.027821321010049294, "passes": true}}, "tier": "improves"}, "alternative": {"passes": false, "n_gating": 183, "n_failing": 46, "cells": {"odd.men.a22_29.atcap": {"truth": 0.014676604027739744, "filled": 0.014777786673743196, "gap": 0.006870489703507232, "seed_sd": 0.0001771275607299105, "tolerance": 0.18023370223284765, "passes": true}, "odd.men.a22_29.level": {"truth": 0.23657185599421368, "filled": 0.23663321104590285, "gap": 0.00025931696977288254, "seed_sd": 0.0002391145887235585, "tolerance": 0.02171146931539837, "passes": true}, "odd.men.a22_29.q10": {"truth": 0.040229885057471264, "filled": 0.04246680662035942, "gap": 0.0541126212544607, "seed_sd": 0.00023223418208336734, "tolerance": 0.07329266358707544, "passes": true}, "odd.men.a22_29.q50": {"truth": 0.24222222222222223, "filled": 0.24479072242975236, "gap": 0.010548072901813477, "seed_sd": 0.00041499148868458826, "tolerance": 0.022980444394274636, "passes": true}, "odd.men.a22_29.q90": {"truth": 0.5661157024793388, "filled": 0.5691991060972214, "gap": 0.005431817106252845, "seed_sd": 0.000776009792142812, "tolerance": 0.02353059621893112, "passes": true}, "odd.men.a22_29.r1": {"truth": 0.8114342592359591, "filled": 0.7956558165756714, "gap": -0.015778442660287717, "seed_sd": 0.000538390871650334, "tolerance": 0.007273490161268873, "passes": false}, "odd.men.a22_29.r2": {"truth": 0.7090594261869381, "filled": 0.6903291482211632, "gap": -0.018730277965774866, "seed_sd": 0.001555470519385471, "tolerance": 0.014356564488812985, "passes": false}, "odd.men.a22_29.r3": {"truth": 0.6020834292042946, "filled": 0.5785067458786555, "gap": -0.023576683325639114, "seed_sd": 0.0007072870652359764, "tolerance": 0.013280307041936976, "passes": false}, "odd.men.a22_29.r4": {"truth": 0.5990582702554191, "filled": 0.5615431784424237, "gap": -0.037515091812995394, "seed_sd": 0.0016355341838879228, "tolerance": 0.02119963449400435, "passes": false}, "odd.men.a22_29.wint": {"truth": 0.08288543140028289, "filled": 0.06141796322489392, "gap": -0.29975695663693624, "seed_sd": 0.0022284434581811073}, "odd.men.a22_29.zexit": {"truth": 0.4099387400566883, "filled": 0.4131343147115296, "gap": 0.00776502326956241, "seed_sd": 0.002840187990001759, "tolerance": 0.052311207438163365, "passes": true}, "odd.men.a22_29.zint": {"truth": 0.027589420573962787, "filled": 0.03422566442780474, "gap": 0.21554339780599507, "seed_sd": 0.0004781100182703536, "tolerance": 0.12140160693630424, "passes": false}, "odd.men.a22_74.atcap": {"truth": 0.10371778137953785, "filled": 0.10473678694403343, "gap": 0.009776841924077129, "seed_sd": 0.00015417691554360955, "tolerance": 0.03690628859850606, "passes": true}, "odd.men.a22_74.level": {"truth": 0.35325148977531445, "filled": 0.3531990207432427, "gap": -0.00014854269729291936, "seed_sd": 0.0001000050086771046, "tolerance": 0.01106822547283481, "passes": true}, "odd.men.a22_74.q10": {"truth": 0.06111111111111111, "filled": 0.06439628414809703, "gap": 0.05236223099161608, "seed_sd": 0.0001374553093986069, "tolerance": 0.03462864626687699, "passes": false}, "odd.men.a22_74.q50": {"truth": 0.37052341597796146, "filled": 0.3743119269609451, "gap": 0.010172835353512988, "seed_sd": 0.0003016483959067802, "tolerance": 0.012463773427145929, "passes": true}, "odd.men.a22_74.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.003451823067723808, "passes": true}, "odd.men.a22_74.r1": {"truth": 0.8935990156167083, "filled": 0.8912007740159659, "gap": -0.0023982416007424234, "seed_sd": 0.00018848733136768257, "tolerance": 0.00258049589094515, "passes": true}, "odd.men.a22_74.r2": {"truth": 0.8291027164330249, "filled": 0.8070684084035232, "gap": -0.02203430802950168, "seed_sd": 0.000390606153237376, "tolerance": 0.004576681103770273, "passes": false}, "odd.men.a22_74.r3": {"truth": 0.7806492619651291, "filled": 0.7637546915867583, "gap": -0.01689457037837072, "seed_sd": 0.00023760130078809586, "tolerance": 0.004722018684826323, "passes": false}, "odd.men.a22_74.r4": {"truth": 0.7414979369803346, "filled": 0.7061540022562445, "gap": -0.03534393472409014, "seed_sd": 0.0007006280998860182, "tolerance": 0.007000982741779954, "passes": false}, "odd.men.a22_74.wint": {"truth": 0.05745341614906832, "filled": 0.027112318840579706, "gap": -0.7509862711587996, "seed_sd": 0.0005542995081891025, "tolerance": 0.07337302886151653, "passes": false}, "odd.men.a22_74.zexit": {"truth": 0.44752843148294114, "filled": 0.44908700221446535, "gap": 0.0034765680865107562, "seed_sd": 0.001594744130375705, "tolerance": 0.01916127382099423, "passes": true}, "odd.men.a22_74.zint": {"truth": 0.019892968610837895, "filled": 0.024561302963886585, "gap": 0.2108058209935062, "seed_sd": 0.00015894269446237608, "tolerance": 0.05481211657295162, "passes": false}, "odd.men.a30_44.atcap": {"truth": 0.10574805846620577, "filled": 0.10662077371152032, "gap": 0.008218909988384926, "seed_sd": 0.0002907042942532108, "tolerance": 0.049095580366705964, "passes": true}, "odd.men.a30_44.level": {"truth": 0.3982899917120792, "filled": 0.39874622791535386, "gap": 0.0011448319207326696, "seed_sd": 0.00016511436265352354, "tolerance": 0.015220021463125325, "passes": true}, "odd.men.a30_44.q10": {"truth": 0.08706467661691543, "filled": 0.08896815888583662, "gap": 0.021627288538648592, "seed_sd": 0.0003609981444106532, "tolerance": 0.047669064141889934, "passes": true}, "odd.men.a30_44.q50": {"truth": 0.41333333333333333, "filled": 0.41762659698724747, "gap": 0.010333354712889542, "seed_sd": 0.00047442460909717036, "tolerance": 0.016660415997544607, "passes": true}, "odd.men.a30_44.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0037405831290436837, "passes": true}, "odd.men.a30_44.r1": {"truth": 0.9000861304840623, "filled": 0.8982106532629379, "gap": -0.0018754772211243553, "seed_sd": 0.00027892984330348577, "tolerance": 0.003556208950302802, "passes": true}, "odd.men.a30_44.r2": {"truth": 0.8479233224100629, "filled": 0.8246521380478662, "gap": -0.023271184362196662, "seed_sd": 0.0006272798022775334, "tolerance": 0.006500650570758907, "passes": false}, "odd.men.a30_44.r3": {"truth": 0.8030903893396322, "filled": 0.7843304370743531, "gap": -0.018759952265279045, "seed_sd": 0.0004185258212879459, "tolerance": 0.006087370619188444, "passes": false}, "odd.men.a30_44.r4": {"truth": 0.7869108820616492, "filled": 0.7484063801783494, "gap": -0.03850450188329979, "seed_sd": 0.0006994124324364131, "tolerance": 0.009218001294372115, "passes": false}, "odd.men.a30_44.wint": {"truth": 0.07239145006330527, "filled": 0.03310866165189543, "gap": -0.782293269893291, "seed_sd": 0.0008974621170717661, "tolerance": 0.12001713872675526, "passes": false}, "odd.men.a30_44.zexit": {"truth": 0.4342024282437196, "filled": 0.43137833603759856, "gap": -0.006525334996968168, "seed_sd": 0.002601768462594161, "tolerance": 0.031953074094816486, "passes": true}, "odd.men.a30_44.zint": {"truth": 0.017933882760031997, "filled": 0.02123248934943166, "gap": 0.1688407098487641, "seed_sd": 0.00022396503563602355, "tolerance": 0.08989768311292742, "passes": false}, "odd.men.a45_59.atcap": {"truth": 0.15826463477931516, "filled": 0.15946433822763031, "gap": 0.007551776839737512, "seed_sd": 0.000344680339561423, "tolerance": 0.04711609868499565, "passes": true}, "odd.men.a45_59.level": {"truth": 0.4279852808128974, "filled": 0.4275779517090131, "gap": -0.0009521894330581926, "seed_sd": 0.0001983681644962252, "tolerance": 0.015834008317210692, "passes": true}, "odd.men.a45_59.q10": {"truth": 0.08868501529051988, "filled": 0.09235107265412808, "gap": 0.04050638360257164, "seed_sd": 0.00043726182355950465, "tolerance": 0.055701610600039544, "passes": true}, "odd.men.a45_59.q50": {"truth": 0.4701492537313433, "filled": 0.47394056916236876, "gap": 0.008031726897409053, "seed_sd": 0.0004982713854334642, "tolerance": 0.020385938992396834, "passes": true}, "odd.men.a45_59.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0}, "odd.men.a45_59.r1": {"truth": 0.9077661818197654, "filled": 0.9062895753124117, "gap": -0.001476606507353706, "seed_sd": 0.0002991092950177627, "tolerance": 0.004032009924506552, "passes": true}, "odd.men.a45_59.r2": {"truth": 0.850421444941529, "filled": 0.8269450270263959, "gap": -0.023476417915133108, "seed_sd": 0.0006042911387057817, "tolerance": 0.0074262368998509335, "passes": false}, "odd.men.a45_59.r3": {"truth": 0.8146289743583418, "filled": 0.7978400340916917, "gap": -0.01678894026665012, "seed_sd": 0.0004995869149426763, "tolerance": 0.007198396861260139, "passes": false}, "odd.men.a45_59.r4": {"truth": 0.7598303885400162, "filled": 0.7277735022660579, "gap": -0.03205688627395831, "seed_sd": 0.0013551668525231687, "tolerance": 0.012107705260459355, "passes": false}, "odd.men.a45_59.wint": {"truth": 0.05825168693530555, "filled": 0.02120773131028173, "gap": -1.010407252645218, "seed_sd": 0.0007690594119414791, "tolerance": 0.13344110142481172, "passes": false}, "odd.men.a45_59.zexit": {"truth": 0.45185706741770815, "filled": 0.4509065305403978, "gap": -0.002105838628690182, "seed_sd": 0.003266441784644715, "tolerance": 0.036314919811016276, "passes": true}, "odd.men.a45_59.zint": {"truth": 0.015835938476928848, "filled": 0.01935900962861073, "gap": 0.20087598076997093, "seed_sd": 0.0002885353100226522, "tolerance": 0.09287476169914739, "passes": false}, "odd.men.a60_74.atcap": {"truth": 0.10024796315975912, "filled": 0.10304852526017169, "gap": 0.02755324794681613, "seed_sd": 0.0004932495167821012, "tolerance": 0.12472307183935011, "passes": true}, "odd.men.a60_74.level": {"truth": 0.20453937198442498, "filled": 0.2035442319076033, "gap": -0.004877147917293545, "seed_sd": 0.0003456308860662253, "tolerance": 0.0522939828282486, "passes": true}, "odd.men.a60_74.q10": {"truth": 0.01529051987767584, "filled": 0.01800191108137369, "gap": 0.16324490292884608, "seed_sd": 0.00021772331220310686, "tolerance": 0.20186846491463123, "passes": true}, "odd.men.a60_74.q50": {"truth": 0.21666666666666667, "filled": 0.22468282133340836, "gap": 0.03632965048242465, "seed_sd": 0.001131712401025331, "tolerance": 0.07446457229432804, "passes": true}, "odd.men.a60_74.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.04069424979546094, "passes": true}, "odd.men.a60_74.r1": {"truth": 0.8712782553777247, "filled": 0.8740493797312835, "gap": 0.0027711243535587515, "seed_sd": 0.0008060871217724458, "tolerance": 0.009750532797445479, "passes": true}, "odd.men.a60_74.r2": {"truth": 0.7888015745113253, "filled": 0.7688138264043076, "gap": -0.019987748107017644, "seed_sd": 0.002381034523972206, "tolerance": 0.020757725353477027, "passes": true}, "odd.men.a60_74.r3": {"truth": 0.7283746799072385, "filled": 0.7151581255769487, "gap": -0.013216554330289787, "seed_sd": 0.0011687895450421084, "tolerance": 0.02064690204069302, "passes": true}, "odd.men.a60_74.r4": {"truth": 0.6781549627202678, "filled": 0.6469288632171912, "gap": -0.031226099503076532, "seed_sd": 0.003228826412160021, "tolerance": 0.03820786124425089, "passes": true}, "odd.men.a60_74.wint": {"truth": 0.03084255319148936, "filled": 0.010389787234042554, "gap": -1.088072006632677, "seed_sd": 0.000692980423590686}, "odd.men.a60_74.zexit": {"truth": 0.5047452182800409, "filled": 0.5156543534335911, "gap": 0.021382899633966224, "seed_sd": 0.0028521576213372778, "tolerance": 0.04434932710092839, "passes": true}, "odd.men.a60_74.zint": {"truth": 0.027132152458672565, "filled": 0.038515072358762184, "gap": 0.3503301923053983, "seed_sd": 0.0007221867926525434, "tolerance": 0.17758159695027936, "passes": false}, "odd.men.b1936_1940.aime_p10": {"truth": 346.0, "filled": 346.45, "gap": 0.0012997330156654385, "seed_sd": 1.1459310165698642, "tolerance": 0.30817736367768905, "passes": true}, "odd.men.b1936_1940.aime_p25": {"truth": 1131.0, "filled": 1130.3, "gap": -0.0006191129194350609, "seed_sd": 0.9233805168766387, "tolerance": 0.15436566295506535, "passes": true}, "odd.men.b1936_1940.aime_p50": {"truth": 2430.0, "filled": 2427.9, "gap": -0.0008645711648282983, "seed_sd": 1.2937094768634554, "tolerance": 0.0822816707411384, "passes": true}, "odd.men.b1936_1940.aime_p75": {"truth": 3582.0, "filled": 3580.8, "gap": -0.0003350645030515409, "seed_sd": 1.3218806379747876, "tolerance": 0.04915519601813279, "passes": true}, "odd.men.b1936_1940.aime_p90": {"truth": 4382.0, "filled": 4376.85, "gap": -0.0011759535997271087, "seed_sd": 2.084403234046921, "tolerance": 0.03801502836217192, "passes": true}, "odd.men.b1936_1945.aime_p10": {"truth": 342.8000000000002, "filled": 341.72, "gap": -0.0031554984402069053, "seed_sd": 0.77974354758472, "tolerance": 0.19387427864829543, "passes": true}, "odd.men.b1936_1945.aime_p25": {"truth": 1152.0, "filled": 1154.125, "gap": 0.0018429188369548655, "seed_sd": 0.8251794063302713, "tolerance": 0.09928513681983075, "passes": true}, "odd.men.b1936_1945.aime_p50": {"truth": 2625.0, "filled": 2623.7, "gap": -0.0004953607661262183, "seed_sd": 1.3803127029389888, "tolerance": 0.05013200755321697, "passes": true}, "odd.men.b1936_1945.aime_p75": {"truth": 3991.5, "filled": 3995.55, "gap": 0.00101414172870129, "seed_sd": 1.6535448364999934, "tolerance": 0.030282584009309735, "passes": true}, "odd.men.b1936_1945.aime_p90": {"truth": 5075.0, "filled": 5069.499999999999, "gap": -0.0010843315173527657, "seed_sd": 3.0803280754866273, "tolerance": 0.029246187918705275, "passes": true}, "odd.men.b1941_1945.aime_p10": {"truth": 338.0, "filled": 337.69, "gap": -0.0009175806116727969, "seed_sd": 1.1247923785022893, "tolerance": 0.2434370939708618, "passes": true}, "odd.men.b1941_1945.aime_p25": {"truth": 1174.25, "filled": 1177.95, "gap": 0.0031459935818887175, "seed_sd": 2.067289096988207, "tolerance": 0.15717605214643726, "passes": true}, "odd.men.b1941_1945.aime_p50": {"truth": 2821.5, "filled": 2824.2, "gap": 0.0009564802259571792, "seed_sd": 2.816305904437304, "tolerance": 0.06794319334096698, "passes": true}, "odd.men.b1941_1945.aime_p75": {"truth": 4421.0, "filled": 4423.125, "gap": 0.00048054500380700915, "seed_sd": 2.5707105304913167, "tolerance": 0.039611439643930886, "passes": true}, "odd.men.b1941_1945.aime_p90": {"truth": 5528.300000000001, "filled": 5530.155000000001, "gap": 0.0003354899065737271, "seed_sd": 3.074080949521515, "tolerance": 0.024092287379584104, "passes": true}, "odd.men.b1946_1955.paime_p10": {"truth": 308.0, "filled": 309.55, "gap": 0.00501984699166691, "seed_sd": 0.998683343734455, "tolerance": 0.11641286886749184, "passes": true}, "odd.men.b1946_1955.paime_p25": {"truth": 1026.0, "filled": 1031.0, "gap": 0.0048614582862454014, "seed_sd": 1.2565617248750864, "tolerance": 0.07923354406873329, "passes": true}, "odd.men.b1946_1955.paime_p50": {"truth": 2614.0, "filled": 2612.45, "gap": -0.0005931368502301027, "seed_sd": 1.8202082009311031, "tolerance": 0.03884145918889322, "passes": true}, "odd.men.b1946_1955.paime_p75": {"truth": 4390.0, "filled": 4390.2, "gap": 4.55570488231416e-05, "seed_sd": 2.912766816330444, "tolerance": 0.02464882648778213, "passes": true}, "odd.men.b1946_1955.paime_p90": {"truth": 5810.0, "filled": 5804.210000000001, "gap": -0.0009970545529416341, "seed_sd": 2.14178970907387, "tolerance": 0.01935089604023114, "passes": true}, "odd.men.b1946_1980.paime_p10": {"truth": 175.0, "filled": 176.85, "gap": 0.01051594172797543, "seed_sd": 0.36634754853252327, "tolerance": 0.05432313062129484, "passes": true}, "odd.men.b1946_1980.paime_p25": {"truth": 487.0, "filled": 487.45, "gap": 0.0009235979926911497, "seed_sd": 0.6048053188292994, "tolerance": 0.03064646902486962, "passes": true}, "odd.men.b1946_1980.paime_p50": {"truth": 1245.0, "filled": 1246.15, "gap": 0.0009232684356144105, "seed_sd": 0.7451598203705947, "tolerance": 0.025757396517852603, "passes": true}, "odd.men.b1946_1980.paime_p75": {"truth": 2644.0, "filled": 2645.6125, "gap": 0.0006096855109705146, "seed_sd": 1.2888055465593347, "tolerance": 0.0206417529534006, "passes": true}, "odd.men.b1946_1980.paime_p90": {"truth": 4310.0, "filled": 4309.555000000002, "gap": -0.00010325359032847814, "seed_sd": 2.004068230795734, "tolerance": 0.01797004074133569, "passes": true}, "odd.men.b1956_1965.paime_p10": {"truth": 253.0, "filled": 255.37000000000006, "gap": 0.009323985168177451, "seed_sd": 0.5777451632719952, "tolerance": 0.10325340800222761, "passes": true}, "odd.men.b1956_1965.paime_p25": {"truth": 765.0, "filled": 768.75, "gap": 0.00488998529419149, "seed_sd": 1.2085223687584246, "tolerance": 0.06714959239176221, "passes": true}, "odd.men.b1956_1965.paime_p50": {"truth": 1764.0, "filled": 1764.85, "gap": 0.0004817433534656246, "seed_sd": 1.496487114615601, "tolerance": 0.03783036836459788, "passes": true}, "odd.men.b1956_1965.paime_p75": {"truth": 2958.0, "filled": 2957.75, "gap": -8.452013697279881e-05, "seed_sd": 1.4823523268955432, "tolerance": 0.02582244416868233, "passes": true}, "odd.men.b1956_1965.paime_p90": {"truth": 4111.0, "filled": 4108.740000000001, "gap": -0.0005498957526519632, "seed_sd": 1.215860102756237, "tolerance": 0.022664845011086624, "passes": true}, "odd.men.b1966_1980.paime_p10": {"truth": 112.0, "filled": 114.15, "gap": 0.019014501680710616, "seed_sd": 0.36634754853252327, "tolerance": 0.07763263566160054, "passes": true}, "odd.men.b1966_1980.paime_p25": {"truth": 303.25, "filled": 304.2125, "gap": 0.0031689225440505453, "seed_sd": 0.6138478724621562, "tolerance": 0.04145124387911718, "passes": true}, "odd.men.b1966_1980.paime_p50": {"truth": 655.0, "filled": 655.25, "gap": 0.0003816065682640257, "seed_sd": 0.8506963092234007, "tolerance": 0.02859495381248372, "passes": true}, "odd.men.b1966_1980.paime_p75": {"truth": 1225.0, "filled": 1225.025, "gap": 2.0407955021894963e-05, "seed_sd": 1.0159905718584514, "tolerance": 0.024541637055534683, "passes": true}, "odd.men.b1966_1980.paime_p90": {"truth": 1888.9000000000015, "filled": 1888.94, "gap": 2.117612180541073e-05, "seed_sd": 1.428064571737837, "tolerance": 0.025554311796701996, "passes": true}, "odd.women.a22_29.atcap": {"truth": 0.005728386357627567, "filled": 0.00547841056691332, "gap": -0.044618861731397175, "seed_sd": 0.0001198351137429579}, "odd.women.a22_29.level": {"truth": 0.1854665923300802, "filled": 0.1848740910687206, "gap": -0.0031997660178675336, "seed_sd": 0.00014689392351929534, "tolerance": 0.018517427407013797, "passes": true}, "odd.women.a22_29.q10": {"truth": 0.028735632183908046, "filled": 0.029796942509710787, "gap": 0.03626789570059907, "seed_sd": 0.00021225931773054583, "tolerance": 0.0684466769976442, "passes": true}, "odd.women.a22_29.q50": {"truth": 0.191131498470948, "filled": 0.19297325015068054, "gap": 0.0095899142162017, "seed_sd": 0.0003559220591245615, "tolerance": 0.021849929026002995, "passes": true}, "odd.women.a22_29.q90": {"truth": 0.46005509641873277, "filled": 0.4622275517880918, "gap": 0.004711049029350156, "seed_sd": 0.0005400193806653005, "tolerance": 0.020751504651450387, "passes": true}, "odd.women.a22_29.r1": {"truth": 0.7942016511246733, "filled": 0.7745630021286282, "gap": -0.01963864899604517, "seed_sd": 0.0007292428154612855, "tolerance": 0.006445547562698701, "passes": false}, "odd.women.a22_29.r2": {"truth": 0.6731585499668502, "filled": 0.6642985372572703, "gap": -0.008860012709579923, "seed_sd": 0.002126151717425611, "tolerance": 0.013111971249341917, "passes": true}, "odd.women.a22_29.r3": {"truth": 0.5535614164842262, "filled": 0.5329081040600008, "gap": -0.020653312424225412, "seed_sd": 0.0009899564896844986, "tolerance": 0.012054115899361516, "passes": false}, "odd.women.a22_29.r4": {"truth": 0.552312691565311, "filled": 0.5278673657012054, "gap": -0.024445325864105638, "seed_sd": 0.002194118197599944, "tolerance": 0.01982762523559322, "passes": false}, "odd.women.a22_29.wint": {"truth": 0.08501885498800137, "filled": 0.06794309221803221, "gap": -0.22420257962872592, "seed_sd": 0.0017381669686975256, "tolerance": 0.15037337545812676, "passes": false}, "odd.women.a22_29.zexit": {"truth": 0.41757828810020875, "filled": 0.42427974947807934, "gap": 0.015920981053845762, "seed_sd": 0.0035450250782922098, "tolerance": 0.041287246240518535, "passes": true}, "odd.women.a22_29.zint": {"truth": 0.0315547703180212, "filled": 0.04034628975265017, "gap": 0.24577466265365944, "seed_sd": 0.0006563413066339625, "tolerance": 0.09342121518593281, "passes": false}, "odd.women.a22_74.atcap": {"truth": 0.029398307023564402, "filled": 0.029927931331664083, "gap": 0.017855114137902195, "seed_sd": 0.00013923057619168755, "tolerance": 0.06709882226274566, "passes": true}, "odd.women.a22_74.level": {"truth": 0.2372854094489719, "filled": 0.2373323106690793, "gap": 0.00019763788106530455, "seed_sd": 6.692992230174344e-05, "tolerance": 0.010902725750872375, "passes": true}, "odd.women.a22_74.q10": {"truth": 0.03777777777777778, "filled": 0.03953723702579737, "gap": 0.045521897075397, "seed_sd": 0.00010489045382469636, "tolerance": 0.031129845660342752, "passes": false}, "odd.women.a22_74.q50": {"truth": 0.24875621890547264, "filled": 0.25114321485161784, "gap": 0.009549977161435352, "seed_sd": 0.00023603789043680378, "tolerance": 0.012092425838785583, "passes": true}, "odd.women.a22_74.q90": {"truth": 0.6444444444444445, "filled": 0.6478378057479859, "gap": 0.005251746052140682, "seed_sd": 0.00023512489065456287, "tolerance": 0.013338685039058783, "passes": true}, "odd.women.a22_74.r1": {"truth": 0.8812684347612898, "filled": 0.8790561126410884, "gap": -0.002212322120201393, "seed_sd": 0.00026366261525432043, "tolerance": 0.002475922564645967, "passes": true}, "odd.women.a22_74.r2": {"truth": 0.8056395091372208, "filled": 0.7906943070261213, "gap": -0.01494520211109951, "seed_sd": 0.0008216520277802765, "tolerance": 0.004772836686607259, "passes": false}, "odd.women.a22_74.r3": {"truth": 0.7468316241328095, "filled": 0.7322134148251858, "gap": -0.014618209307623697, "seed_sd": 0.00042896857078218335, "tolerance": 0.005171487109756781, "passes": false}, "odd.women.a22_74.r4": {"truth": 0.7116744894807431, "filled": 0.6823372442061806, "gap": -0.029337245274562496, "seed_sd": 0.0007616203326649829, "tolerance": 0.007666262733486471, "passes": false}, "odd.women.a22_74.wint": {"truth": 0.051544874036178946, "filled": 0.027846837562797884, "gap": -0.6157333613583016, "seed_sd": 0.0002980240871252773, "tolerance": 0.07133033012317215, "passes": false}, "odd.women.a22_74.zexit": {"truth": 0.45845752128015943, "filled": 0.4576462554649855, "gap": -0.0017711225471119807, "seed_sd": 0.0008638884333265164, "tolerance": 0.016039678883419197, "passes": true}, "odd.women.a22_74.zint": {"truth": 0.022201124403524123, "filled": 0.027159357807590257, "gap": 0.2015787212211242, "seed_sd": 0.0002523189210600705, "tolerance": 0.04476234327525859, "passes": false}, "odd.women.a30_44.atcap": {"truth": 0.034619662054697874, "filled": 0.035343450511844496, "gap": 0.020691311561050973, "seed_sd": 0.0002127099479691062, "tolerance": 0.08510646139536093, "passes": true}, "odd.women.a30_44.level": {"truth": 0.2579924798566703, "filled": 0.25834447971623264, "gap": 0.0013634503886146287, "seed_sd": 0.00012777256821587513, "tolerance": 0.01578879525774269, "passes": true}, "odd.women.a30_44.q10": {"truth": 0.04228855721393035, "filled": 0.0434532780200243, "gap": 0.027169757945184614, "seed_sd": 0.0002946128390438379, "tolerance": 0.04573918824793483, "passes": true}, "odd.women.a30_44.q50": {"truth": 0.26859504132231404, "filled": 0.27048982232809066, "gap": 0.0070296494531243425, "seed_sd": 0.0003572641637749294, "tolerance": 0.018488627942446087, "passes": true}, "odd.women.a30_44.q90": {"truth": 0.6716417910447762, "filled": 0.6776760023832323, "gap": 0.008944151770306275, "seed_sd": 0.0008011122590598357, "tolerance": 0.018806950089013175, "passes": true}, "odd.women.a30_44.r1": {"truth": 0.888241419643356, "filled": 0.8878263947476558, "gap": -0.0004150248957002223, "seed_sd": 0.00039085301367594664, "tolerance": 0.0034876380830913202, "passes": true}, "odd.women.a30_44.r2": {"truth": 0.8242277470219269, "filled": 0.8080048131468264, "gap": -0.016222933875100543, "seed_sd": 0.0013833436447721692, "tolerance": 0.007078708407960211, "passes": false}, "odd.women.a30_44.r3": {"truth": 0.7685608654079126, "filled": 0.7530075711034663, "gap": -0.015553294304446297, "seed_sd": 0.0006708836561040368, "tolerance": 0.0073196290204859075, "passes": false}, "odd.women.a30_44.r4": {"truth": 0.7490435327097834, "filled": 0.7166554517324724, "gap": -0.03238808097731105, "seed_sd": 0.0012795706934268887, "tolerance": 0.010187042430903364, "passes": false}, "odd.women.a30_44.wint": {"truth": 0.06073485056210584, "filled": 0.033955305730737594, "gap": -0.5814725551358548, "seed_sd": 0.0006491314836425788, "tolerance": 0.10093703169083498, "passes": false}, "odd.women.a30_44.zexit": {"truth": 0.45066799061202384, "filled": 0.44936021845098406, "gap": -0.0029060713169714036, "seed_sd": 0.0018066337869472745, "tolerance": 0.024719366356404402, "passes": true}, "odd.women.a30_44.zint": {"truth": 0.022222222222222223, "filled": 0.02548624733010767, "gap": 0.13704619707907684, "seed_sd": 0.0002748908459707359, "tolerance": 0.07029190354044747, "passes": false}, "odd.women.a45_59.atcap": {"truth": 0.04053483605515179, "filled": 0.041239707200188894, "gap": 0.01723980531931124, "seed_sd": 0.00018965501967751098, "tolerance": 0.09107631963722376, "passes": true}, "odd.women.a45_59.level": {"truth": 0.28104586423928846, "filled": 0.28119635395726095, "gap": 0.0005353198913060631, "seed_sd": 0.00014247082170896653, "tolerance": 0.01605667435242628, "passes": true}, "odd.women.a45_59.q10": {"truth": 0.053516819571865444, "filled": 0.05545805092900992, "gap": 0.03563090683069614, "seed_sd": 0.0004748767234848939, "tolerance": 0.051646333169991114, "passes": true}, "odd.women.a45_59.q50": {"truth": 0.29338842975206614, "filled": 0.2964274540543556, "gap": 0.010305084280428867, "seed_sd": 0.0003713143604795996, "tolerance": 0.0182941133196311, "passes": true}, "odd.women.a45_59.q90": {"truth": 0.7300275482093664, "filled": 0.7337269335985186, "gap": 0.005054663622374556, "seed_sd": 0.0005886018592810698, "tolerance": 0.0208569023153089, "passes": true}, "odd.women.a45_59.r1": {"truth": 0.9101006329634369, "filled": 0.9108974471764941, "gap": 0.0007968142130572176, "seed_sd": 0.0003800420009505249, "tolerance": 0.0037361334549905496, "passes": true}, "odd.women.a45_59.r2": {"truth": 0.8498173267452248, "filled": 0.8361682096210815, "gap": -0.013649117124143295, "seed_sd": 0.0008553583208083107, "tolerance": 0.0074184353060767995, "passes": false}, "odd.women.a45_59.r3": {"truth": 0.8096130207660378, "filled": 0.7977960800478768, "gap": -0.011816940718160973, "seed_sd": 0.0005535582972400586, "tolerance": 0.007740061426920219, "passes": false}, "odd.women.a45_59.r4": {"truth": 0.7620167777938601, "filled": 0.7367946054768152, "gap": -0.025222172317044933, "seed_sd": 0.0014331459374298417, "tolerance": 0.012512985825002505, "passes": false}, "odd.women.a45_59.wint": {"truth": 0.050115932427956277, "filled": 0.018719774759854254, "gap": -0.9847585311516158, "seed_sd": 0.00039460658224803785, "tolerance": 0.1271041647107106, "passes": false}, "odd.women.a45_59.zexit": {"truth": 0.4693136110029843, "filled": 0.46855780459322693, "gap": -0.0016117488193031493, "seed_sd": 0.0026195994047660633, "tolerance": 0.030024476078798885, "passes": true}, "odd.women.a45_59.zint": {"truth": 0.01632776600720909, "filled": 0.01969952948298159, "gap": 0.18772765670992042, "seed_sd": 0.0003049152557496679, "tolerance": 0.09017841563442361, "passes": false}, "odd.women.a60_74.atcap": {"truth": 0.01812919896640827, "filled": 0.018630583537738197, "gap": 0.02728066557727482, "seed_sd": 0.00039508547268232713}, "odd.women.a60_74.level": {"truth": 0.12926943691615103, "filled": 0.12904856374827697, "gap": -0.0017100877301055029, "seed_sd": 0.0001935279226056811, "tolerance": 0.04408494999775865, "passes": true}, "odd.women.a60_74.q10": {"truth": 0.017412935323383085, "filled": 0.018677153233438732, "gap": 0.07008768527188503, "seed_sd": 0.00034170082521718935, "tolerance": 0.13013445963696416, "passes": true}, "odd.women.a60_74.q50": {"truth": 0.15517241379310345, "filled": 0.1576036237180233, "gap": 0.015546324522169197, "seed_sd": 0.0005376073055390707, "tolerance": 0.05365060769995596, "passes": true}, "odd.women.a60_74.q90": {"truth": 0.5360696517412935, "filled": 0.5359236469864845, "gap": -0.0002723986351127472, "seed_sd": 0.0013064682810661346, "tolerance": 0.04960336282899433, "passes": true}, "odd.women.a60_74.r1": {"truth": 0.8610627385333514, "filled": 0.8641934680473181, "gap": 0.003130729513966757, "seed_sd": 0.0009278283970724514, "tolerance": 0.009751654703991025, "passes": true}, "odd.women.a60_74.r2": {"truth": 0.7722738686836694, "filled": 0.7442124191252916, "gap": -0.02806144955837786, "seed_sd": 0.0023964065139079997, "tolerance": 0.022174026993056747, "passes": false}, "odd.women.a60_74.r3": {"truth": 0.7141078828415282, "filled": 0.6932584179481365, "gap": -0.020849464893391678, "seed_sd": 0.001709358820841769, "tolerance": 0.018479646240433655, "passes": false}, "odd.women.a60_74.r4": {"truth": 0.6501204938024134, "filled": 0.6102725160155678, "gap": -0.03984797778684568, "seed_sd": 0.004143724959812022, "tolerance": 0.036314307016843086, "passes": false}, "odd.women.a60_74.wint": {"truth": 0.023107482596493787, "filled": 0.008455734956445676, "gap": -1.0053115827709154, "seed_sd": 0.000566243984687916}, "odd.women.a60_74.zexit": {"truth": 0.5155483759303063, "filled": 0.5055270293659493, "gap": -0.019629634203051416, "seed_sd": 0.003325690297144084, "tolerance": 0.03896610501214408, "passes": true}, "odd.women.a60_74.zint": {"truth": 0.022290698541288734, "filled": 0.033288188663303596, "gap": 0.40103315359017433, "seed_sd": 0.0006465836607581495}, "odd.women.b1936_1940.aime_p10": {"truth": 82.29999999999995, "filled": 83.05, "gap": 0.009071728376279786, "seed_sd": 0.22360679774997896, "tolerance": 0.32371200850504067, "passes": true}, "odd.women.b1936_1940.aime_p25": {"truth": 287.75, "filled": 288.15, "gap": 0.0013891302806836592, "seed_sd": 0.36634754853252327, "tolerance": 0.20139115911902655, "passes": true}, "odd.women.b1936_1940.aime_p50": {"truth": 786.0, "filled": 785.9, "gap": -0.0001272345570777489, "seed_sd": 0.7181848464596079, "tolerance": 0.12524071173214943, "passes": true}, "odd.women.b1936_1940.aime_p75": {"truth": 1566.0, "filled": 1566.225, "gap": 0.00014366784020136691, "seed_sd": 1.2822164198638313, "tolerance": 0.10110924143579125, "passes": true}, "odd.women.b1936_1940.aime_p90": {"truth": 2438.7000000000007, "filled": 2438.2400000000007, "gap": -0.00018864287908559874, "seed_sd": 2.977353116267416, "tolerance": 0.08967006892401257, "passes": true}, "odd.women.b1936_1945.aime_p10": {"truth": 98.0, "filled": 98.54, "gap": 0.005495078445244772, "seed_sd": 0.5030433694811536, "tolerance": 0.1995625804481066, "passes": true}, "odd.women.b1936_1945.aime_p25": {"truth": 345.0, "filled": 344.55, "gap": -0.0013051992281427616, "seed_sd": 0.6048053188292994, "tolerance": 0.11108088073804002, "passes": true}, "odd.women.b1936_1945.aime_p50": {"truth": 932.0, "filled": 930.85, "gap": -0.001234667467684858, "seed_sd": 0.9880869341680844, "tolerance": 0.07343389253317807, "passes": true}, "odd.women.b1936_1945.aime_p75": {"truth": 1865.0, "filled": 1865.0, "gap": 0.0, "seed_sd": 1.3764944032233706, "tolerance": 0.06283602704614394, "passes": true}, "odd.women.b1936_1945.aime_p90": {"truth": 2920.0, "filled": 2925.88, "gap": 0.002011673856783247, "seed_sd": 1.7355721034622027, "tolerance": 0.0536209031123137, "passes": true}, "odd.women.b1941_1945.aime_p10": {"truth": 115.0, "filled": 115.96000000000001, "gap": 0.008313175690204844, "seed_sd": 0.6003507746572602, "tolerance": 0.2647761737592696, "passes": true}, "odd.women.b1941_1945.aime_p25": {"truth": 399.0, "filled": 399.05, "gap": 0.00012530543215394374, "seed_sd": 0.9445132413883327, "tolerance": 0.15691889792263147, "passes": true}, "odd.women.b1941_1945.aime_p50": {"truth": 1077.0, "filled": 1074.3, "gap": -0.002510111483892352, "seed_sd": 1.0809352675491621, "tolerance": 0.08797933476607916, "passes": true}, "odd.women.b1941_1945.aime_p75": {"truth": 2116.0, "filled": 2119.05, "gap": 0.0014403610475932638, "seed_sd": 2.0124611797498106, "tolerance": 0.07205053130731136, "passes": true}, "odd.women.b1941_1945.aime_p90": {"truth": 3268.6000000000004, "filled": 3272.19, "gap": 0.0010977268374308125, "seed_sd": 3.0741537132407077, "tolerance": 0.06548294022251817, "passes": true}, "odd.women.b1946_1955.paime_p10": {"truth": 158.0, "filled": 158.7, "gap": 0.004420594505396558, "seed_sd": 0.6569466853317862, "tolerance": 0.1085059910489472, "passes": true}, "odd.women.b1946_1955.paime_p25": {"truth": 519.0, "filled": 518.7375, "gap": -0.0005059082968452699, "seed_sd": 0.9370832406995656, "tolerance": 0.062889445073137, "passes": true}, "odd.women.b1946_1955.paime_p50": {"truth": 1313.0, "filled": 1313.25, "gap": 0.00019038553127526114, "seed_sd": 1.019545822516343, "tolerance": 0.04313158587991037, "passes": true}, "odd.women.b1946_1955.paime_p75": {"truth": 2486.0, "filled": 2486.425, "gap": 0.0001709427496789928, "seed_sd": 1.709301180692212, "tolerance": 0.03158401312013406, "passes": true}, "odd.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3782.2200000000003, "gap": 0.0005871291932191269, "seed_sd": 2.638699839334961, "tolerance": 0.02998098721497899, "passes": true}, "odd.women.b1946_1980.paime_p10": {"truth": 109.0, "filled": 108.15500000000002, "gap": -0.007782498813678096, "seed_sd": 0.36487200351269256, "tolerance": 0.0500438197853722, "passes": true}, "odd.women.b1946_1980.paime_p25": {"truth": 309.0, "filled": 308.55, "gap": -0.001457372130669654, "seed_sd": 0.5104177855340405, "tolerance": 0.030021121307637916, "passes": true}, "odd.women.b1946_1980.paime_p50": {"truth": 758.0, "filled": 757.325, "gap": -0.0008908980511055375, "seed_sd": 0.46665100224336475, "tolerance": 0.02113414008122686, "passes": true}, "odd.women.b1946_1980.paime_p75": {"truth": 1590.0, "filled": 1589.4, "gap": -0.000377429708198207, "seed_sd": 0.7539370349250519, "tolerance": 0.018434974631993433, "passes": true}, "odd.women.b1946_1980.paime_p90": {"truth": 2696.0, "filled": 2698.4, "gap": 0.0008898117152424945, "seed_sd": 1.729009451741235, "tolerance": 0.018162919937725473, "passes": true}, "odd.women.b1956_1965.paime_p10": {"truth": 140.0, "filled": 139.79500000000002, "gap": -0.0014653588283035646, "seed_sd": 0.6056531924077901, "tolerance": 0.0987946317785998, "passes": true}, "odd.women.b1956_1965.paime_p25": {"truth": 426.0, "filled": 425.075, "gap": -0.002173722325821359, "seed_sd": 1.0390405493328425, "tolerance": 0.05702004966656188, "passes": true}, "odd.women.b1956_1965.paime_p50": {"truth": 1000.0, "filled": 999.575, "gap": -0.0004250903380960125, "seed_sd": 1.0166379058341997, "tolerance": 0.0353321966023362, "passes": true}, "odd.women.b1956_1965.paime_p75": {"truth": 1842.0, "filled": 1844.7625, "gap": 0.0014986050861729439, "seed_sd": 1.298974798183471, "tolerance": 0.026005856561860215, "passes": true}, "odd.women.b1956_1965.paime_p90": {"truth": 2793.0, "filled": 2794.3599999999997, "gap": 0.00048681310202169925, "seed_sd": 2.08286240291426, "tolerance": 0.027293077620429543, "passes": true}, "odd.women.b1966_1980.paime_p10": {"truth": 80.0, "filled": 80.05, "gap": 0.0006248047688428571, "seed_sd": 0.22360679774997896, "tolerance": 0.06283109225080236, "passes": true}, "odd.women.b1966_1980.paime_p25": {"truth": 211.0, "filled": 210.35, "gap": -0.0030853234395369356, "seed_sd": 0.5871429486123998, "tolerance": 0.033157126538046186, "passes": true}, "odd.women.b1966_1980.paime_p50": {"truth": 463.0, "filled": 462.8, "gap": -0.0004320587667123732, "seed_sd": 0.7677718959499145, "tolerance": 0.02601990880890944, "passes": true}, "odd.women.b1966_1980.paime_p75": {"truth": 866.75, "filled": 867.1, "gap": 0.0004037258179820924, "seed_sd": 0.7181848464596079, "tolerance": 0.02735691355887302, "passes": true}, "odd.women.b1966_1980.paime_p90": {"truth": 1387.0, "filled": 1389.1649999999997, "gap": 0.0015597058812399922, "seed_sd": 1.3819418069857066, "tolerance": 0.027821321010049294, "passes": true}}, "tier": "not_adopted"}, "adopted": "primary"}, "pre": {"dropped_undefined_truth": [], "truth": {"pre.men.b1930_1934.aime_p10": {"value": 319.29999999999995, "events": 12005}, "pre.men.b1930_1934.aime_p25": {"value": 961.0, "events": 12005}, "pre.men.b1930_1934.aime_p50": {"value": 1871.0, "events": 12005}, "pre.men.b1930_1934.aime_p75": {"value": 2604.0, "events": 12005}, "pre.men.b1930_1934.aime_p90": {"value": 3048.7000000000007, "events": 12005}, "pre.men.b1930_1934.pzero": {"value": 0.23849477516714046, "events": 48408}, "pre.men.b1930_1934.plevel": {"value": 0.5292246059138269, "events": 154565}, "pre.men.b1930_1934.pr_in": {"value": 0.6408121492387024, "events": 9749}, "pre.men.b1930_1934.pr_cross": {"value": 0.6005766012618865, "events": 9749}, "pre.men.b1935_1939.aime_p10": {"value": 342.0, "events": 12618}, "pre.men.b1935_1939.aime_p25": {"value": 1099.5, "events": 12618}, "pre.men.b1935_1939.aime_p50": {"value": 2295.0, "events": 12618}, "pre.men.b1935_1939.aime_p75": {"value": 3360.0, "events": 12618}, "pre.men.b1935_1939.aime_p90": {"value": 4087.0, "events": 12618}, "pre.men.b1935_1939.pzero": {"value": 0.21315539421109053, "events": 34995}, "pre.men.b1935_1939.plevel": {"value": 0.47885857925887765, "events": 129181}, "pre.men.b1935_1939.pr_in": {"value": 0.5082293220171369, "events": 9949}, "pre.men.b1935_1939.pr_cross": {"value": 0.5480805317128244, "events": 10010}, "pre.men.b1940_1945.aime_p10": {"value": 343.0, "events": 18739}, "pre.men.b1940_1945.aime_p25": {"value": 1178.0, "events": 18739}, "pre.men.b1940_1945.aime_p50": {"value": 2795.0, "events": 18739}, "pre.men.b1940_1945.aime_p75": {"value": 4361.5, "events": 18739}, "pre.men.b1940_1945.aime_p90": {"value": 5432.800000000003, "events": 18739}, "pre.men.b1940_1945.pzero": {"value": 0.21918652571223796, "events": 30582}, "pre.men.b1940_1945.plevel": {"value": 0.3760217560197671, "events": 108943}, "pre.men.b1940_1945.pr_in": {"value": 0.36011521796896784, "events": 12582}, "pre.men.b1940_1945.pr_cross": {"value": 0.376062044153939, "events": 14181}, "pre.men.b1930_1945.aime_p10": {"value": 335.0, "events": 43362}, "pre.men.b1930_1945.aime_p25": {"value": 1080.0, "events": 43362}, "pre.men.b1930_1945.aime_p50": {"value": 2266.0, "events": 43362}, "pre.men.b1930_1945.aime_p75": {"value": 3412.75, "events": 43362}, "pre.men.b1930_1945.aime_p90": {"value": 4641.0, "events": 43362}, "pre.men.b1930_1945.pzero": {"value": 0.22496713863351978, "events": 113985}, "pre.men.b1930_1945.plevel": {"value": 0.47071653085260085, "events": 392689}, "pre.men.b1930_1945.pr_in": {"value": 0.5486408320203577, "events": 32280}, "pre.men.b1930_1945.pr_cross": {"value": 0.5031824436185776, "events": 33940}, "pre.men.b1946_1955.yzero": {"value": 0.41447106660067734, "events": 128130}, "pre.men.b1946_1955.ylevel": {"value": 0.1479729717947154, "events": 181011}, "pre.men.b1946_1955.yr_cross": {"value": 0.4294012571477206, "events": 32399}, "pre.men.b1946_1955.paime_p10": {"value": 305.0, "events": 44105}, "pre.men.b1946_1955.paime_p25": {"value": 1022.0, "events": 44105}, "pre.men.b1946_1955.paime_p50": {"value": 2610.0, "events": 44105}, "pre.men.b1946_1955.paime_p75": {"value": 4389.0, "events": 44105}, "pre.men.b1946_1955.paime_p90": {"value": 5809.0, "events": 44105}, "pre.men.b1956_1965.yzero": {"value": 0.4166095146652036, "events": 145790}, "pre.men.b1956_1965.ylevel": {"value": 0.09639046670103465, "events": 204154}, "pre.men.b1956_1965.yr_cross": {"value": 0.40595517510382484, "events": 35165}, "pre.men.b1956_1965.paime_p10": {"value": 249.0, "events": 49926}, "pre.men.b1956_1965.paime_p25": {"value": 760.0, "events": 49926}, "pre.men.b1956_1965.paime_p50": {"value": 1761.0, "events": 49926}, "pre.men.b1956_1965.paime_p75": {"value": 2956.0, "events": 49926}, "pre.men.b1956_1965.paime_p90": {"value": 4109.0, "events": 49926}, "pre.men.b1966_1980.yzero": {"value": 0.41574858870643483, "events": 180950}, "pre.men.b1966_1980.ylevel": {"value": 0.055118361796504756, "events": 254289}, "pre.men.b1966_1980.yr_cross": {"value": 0.4093896749871457, "events": 45353}, "pre.men.b1966_1980.paime_p10": {"value": 104.0, "events": 62071}, "pre.men.b1966_1980.paime_p25": {"value": 297.0, "events": 62071}, "pre.men.b1966_1980.paime_p50": {"value": 648.0, "events": 62071}, "pre.men.b1966_1980.paime_p75": {"value": 1218.0, "events": 62071}, "pre.men.b1966_1980.paime_p90": {"value": 1883.0, "events": 62071}, "pre.men.b1946_1980.yzero": {"value": 0.415663002913214, "events": 454870}, "pre.men.b1946_1980.ylevel": {"value": 0.09454735400371912, "events": 639454}, "pre.men.b1946_1980.yr_cross": {"value": 0.49823745301361005, "events": 112917}, "pre.men.b1946_1980.paime_p10": {"value": 168.0, "events": 156102}, "pre.men.b1946_1980.paime_p25": {"value": 480.0, "events": 156102}, "pre.men.b1946_1980.paime_p50": {"value": 1237.0, "events": 156102}, "pre.men.b1946_1980.paime_p75": {"value": 2636.0, "events": 156102}, "pre.men.b1946_1980.paime_p90": {"value": 4302.0, "events": 156102}, "pre.women.b1930_1934.aime_p10": {"value": 51.40000000000009, "events": 10532}, "pre.women.b1930_1934.aime_p25": {"value": 192.0, "events": 10532}, "pre.women.b1930_1934.aime_p50": {"value": 526.0, "events": 10532}, "pre.women.b1930_1934.aime_p75": {"value": 1055.0, "events": 10532}, "pre.women.b1930_1934.aime_p90": {"value": 1663.0, "events": 10532}, "pre.women.b1930_1934.pzero": {"value": 0.5514059876498446, "events": 80419}, "pre.women.b1930_1934.plevel": {"value": 0.17987838861322633, "events": 80419}, "pre.women.b1930_1934.pr_in": {"value": 0.5486860492872828, "events": 3125}, "pre.women.b1930_1934.pr_cross": {"value": 0.5714979296198577, "events": 3542}, "pre.women.b1935_1939.aime_p10": {"value": 75.0, "events": 11626}, "pre.women.b1935_1939.aime_p25": {"value": 267.0, "events": 11626}, "pre.women.b1935_1939.aime_p50": {"value": 726.0, "events": 11626}, "pre.women.b1935_1939.aime_p75": {"value": 1448.0, "events": 11626}, "pre.women.b1935_1939.aime_p90": {"value": 2293.8999999999996, "events": 11626}, "pre.women.b1935_1939.pzero": {"value": 0.5152445437812253, "events": 73741}, "pre.women.b1935_1939.plevel": {"value": 0.18459645468504016, "events": 73741}, "pre.women.b1935_1939.pr_in": {"value": 0.48588537175161917, "events": 3458}, "pre.women.b1935_1939.pr_cross": {"value": 0.5177687037806673, "events": 3675}, "pre.women.b1940_1945.aime_p10": {"value": 110.0, "events": 17687}, "pre.women.b1940_1945.aime_p25": {"value": 386.0, "events": 17687}, "pre.women.b1940_1945.aime_p50": {"value": 1041.0, "events": 17687}, "pre.women.b1940_1945.aime_p75": {"value": 2058.0, "events": 17687}, "pre.women.b1940_1945.aime_p90": {"value": 3179.7000000000007, "events": 17687}, "pre.women.b1940_1945.pzero": {"value": 0.45477910425062734, "events": 59808}, "pre.women.b1940_1945.plevel": {"value": 0.19012308454067162, "events": 71702}, "pre.women.b1940_1945.pr_in": {"value": 0.19634726483554435, "events": 5880}, "pre.women.b1940_1945.pr_cross": {"value": 0.3145537439002571, "events": 6220}, "pre.women.b1930_1945.aime_p10": {"value": 79.0, "events": 39845}, "pre.women.b1930_1945.aime_p25": {"value": 280.0, "events": 39845}, "pre.women.b1930_1945.aime_p50": {"value": 771.0, "events": 39845}, "pre.women.b1930_1945.aime_p75": {"value": 1574.0, "events": 39845}, "pre.women.b1930_1945.aime_p90": {"value": 2560.0, "events": 39845}, "pre.women.b1930_1945.pzero": {"value": 0.5120706676834471, "events": 225862}, "pre.women.b1930_1945.plevel": {"value": 0.18433938803699407, "events": 225862}, "pre.women.b1930_1945.pr_in": {"value": 0.372715069734525, "events": 12463}, "pre.women.b1930_1945.pr_cross": {"value": 0.4395941263612976, "events": 13437}, "pre.women.b1946_1955.yzero": {"value": 0.5347643649882455, "events": 136154}, "pre.women.b1946_1955.ylevel": {"value": 0.09059452101103667, "events": 136154}, "pre.women.b1946_1955.yr_cross": {"value": 0.34158903474392094, "events": 22200}, "pre.women.b1946_1955.paime_p10": {"value": 156.0, "events": 41718}, "pre.women.b1946_1955.paime_p25": {"value": 516.0, "events": 41718}, "pre.women.b1946_1955.paime_p50": {"value": 1311.0, "events": 41718}, "pre.women.b1946_1955.paime_p75": {"value": 2484.0, "events": 41718}, "pre.women.b1946_1955.paime_p90": {"value": 3780.0, "events": 41718}, "pre.women.b1956_1965.yzero": {"value": 0.47511142887567126, "events": 156162}, "pre.women.b1956_1965.ylevel": {"value": 0.06347343380872424, "events": 172523}, "pre.women.b1956_1965.yr_cross": {"value": 0.3538451043271927, "events": 28194}, "pre.women.b1956_1965.paime_p10": {"value": 136.0, "events": 46845}, "pre.women.b1956_1965.paime_p25": {"value": 422.0, "events": 46845}, "pre.women.b1956_1965.paime_p50": {"value": 996.0, "events": 46845}, "pre.women.b1956_1965.paime_p75": {"value": 1839.0, "events": 46845}, "pre.women.b1956_1965.paime_p90": {"value": 2790.5999999999985, "events": 46845}, "pre.women.b1966_1980.yzero": {"value": 0.42453709903219916, "events": 174368}, "pre.women.b1966_1980.ylevel": {"value": 0.04398552907354234, "events": 236357}, "pre.women.b1966_1980.yr_cross": {"value": 0.318767197977468, "events": 40457}, "pre.women.b1966_1980.paime_p10": {"value": 73.0, "events": 58514}, "pre.women.b1966_1980.paime_p25": {"value": 205.0, "events": 58514}, "pre.women.b1966_1980.paime_p50": {"value": 458.0, "events": 58514}, "pre.women.b1966_1980.paime_p75": {"value": 861.0, "events": 58514}, "pre.women.b1966_1980.paime_p90": {"value": 1382.5999999999985, "events": 58514}, "pre.women.b1946_1980.yzero": {"value": 0.4719000529035934, "events": 487032}, "pre.women.b1946_1980.ylevel": {"value": 0.06340849534928693, "events": 545034}, "pre.women.b1946_1980.yr_cross": {"value": 0.43636860215681333, "events": 90851}, "pre.women.b1946_1980.paime_p10": {"value": 103.0, "events": 147077}, "pre.women.b1946_1980.paime_p25": {"value": 304.0, "events": 147077}, "pre.women.b1946_1980.paime_p50": {"value": 752.0, "events": 147077}, "pre.women.b1946_1980.paime_p75": {"value": 1583.75, "events": 147077}, "pre.women.b1946_1980.paime_p90": {"value": 2690.0, "events": 147077}}, "current_rule": {"fallback": {"passes": false, "n_gating": 136, "n_failing": 131, "cells": {"pre.men.b1930_1934.aime_p10": {"truth": 319.29999999999995, "filled": 103.29999999999995, "gap": -1.1284937235951435, "seed_sd": 0.0, "tolerance": 0.3785445178728429, "passes": false}, "pre.men.b1930_1934.aime_p25": {"truth": 961.0, "filled": 509.75, "gap": -0.6340539995157277, "seed_sd": 0.0, "tolerance": 0.1748310883030068, "passes": false}, "pre.men.b1930_1934.aime_p50": {"truth": 1871.0, "filled": 1274.0, "gap": -0.3843114901419806, "seed_sd": 0.0, "tolerance": 0.08482500842259981, "passes": false}, "pre.men.b1930_1934.aime_p75": {"truth": 2604.0, "filled": 1990.0, "gap": -0.26891408560992147, "seed_sd": 0.0, "tolerance": 0.0526345782788754, "passes": false}, "pre.men.b1930_1934.aime_p90": {"truth": 3048.7000000000007, "filled": 2479.0, "gap": -0.20686001719645475, "seed_sd": 0.0, "tolerance": 0.032507815796954, "passes": false}, "pre.men.b1930_1934.plevel": {"truth": 0.5292246059138269, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.05399925993701722, "passes": false}, "pre.men.b1930_1934.pr_cross": {"truth": 0.6005766012618865, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.10805017504629746, "passes": false}, "pre.men.b1930_1934.pr_in": {"truth": 0.6408121492387024, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.08754712042010183, "passes": false}, "pre.men.b1930_1934.pzero": {"truth": 0.23849477516714046, "filled": 1.0, "gap": 1.433407875949719, "seed_sd": 0.0, "tolerance": 0.12848404727056906, "passes": false}, "pre.men.b1930_1945.aime_p10": {"truth": 335.0, "filled": 155.0, "gap": -0.7707054149058195, "seed_sd": 0.0, "tolerance": 0.17154652244393, "passes": false}, "pre.men.b1930_1945.aime_p25": {"truth": 1080.0, "filled": 716.0, "gap": -0.4110361531576201, "seed_sd": 0.0, "tolerance": 0.08803344717643322, "passes": false}, "pre.men.b1930_1945.aime_p50": {"truth": 2266.0, "filled": 1813.0, "gap": -0.22303323083310111, "seed_sd": 0.0, "tolerance": 0.042293553277777104, "passes": false}, "pre.men.b1930_1945.aime_p75": {"truth": 3412.75, "filled": 3075.0, "gap": -0.10421351664246892, "seed_sd": 0.0, "tolerance": 0.0339810042777025, "passes": false}, "pre.men.b1930_1945.aime_p90": {"truth": 4641.0, "filled": 4481.0, "gap": -0.03508362445503366, "seed_sd": 0.0, "tolerance": 0.03097131693794459, "passes": false}, "pre.men.b1930_1945.plevel": {"truth": 0.47071653085260085, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.029173903026433003, "passes": false}, "pre.men.b1930_1945.pr_cross": {"truth": 0.5031824436185776, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.04396339753603818, "passes": false}, "pre.men.b1930_1945.pr_in": {"truth": 0.5486408320203577, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.04095810136062007, "passes": false}, "pre.men.b1930_1945.pzero": {"truth": 0.22496713863351978, "filled": 1.0, "gap": 1.4918009379618222, "seed_sd": 0.0, "tolerance": 0.06285985273036603, "passes": false}, "pre.men.b1935_1939.aime_p10": {"truth": 342.0, "filled": 150.0, "gap": -0.8241754429663493, "seed_sd": 0.0, "tolerance": 0.35458282617348863, "passes": false}, "pre.men.b1935_1939.aime_p25": {"truth": 1099.5, "filled": 706.0, "gap": -0.44299557250157395, "seed_sd": 0.0, "tolerance": 0.17611744524582296, "passes": false}, "pre.men.b1935_1939.aime_p50": {"truth": 2295.0, "filled": 1845.0, "gap": -0.21825356602001822, "seed_sd": 0.0, "tolerance": 0.08805365646562159, "passes": false}, "pre.men.b1935_1939.aime_p75": {"truth": 3360.0, "filled": 2934.0, "gap": -0.13557429425432233, "seed_sd": 0.0, "tolerance": 0.047075663502220186, "passes": false}, "pre.men.b1935_1939.aime_p90": {"truth": 4087.0, "filled": 3747.0, "gap": -0.08685568477059036, "seed_sd": 0.0, "tolerance": 0.03677617093870473, "passes": false}, "pre.men.b1935_1939.plevel": {"truth": 0.47885857925887765, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.0505548909117149, "passes": false}, "pre.men.b1935_1939.pr_cross": {"truth": 0.5480805317128244, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.08965478073758665, "passes": false}, "pre.men.b1935_1939.pr_in": {"truth": 0.5082293220171369, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.08171176478999366, "passes": false}, "pre.men.b1935_1939.pzero": {"truth": 0.21315539421109053, "filled": 1.0, "gap": 1.5457338289783504, "seed_sd": 0.0, "tolerance": 0.12406583731865686, "passes": false}, "pre.men.b1940_1945.aime_p10": {"truth": 343.0, "filled": 211.0, "gap": -0.4858723136898728, "seed_sd": 0.0, "tolerance": 0.2436998893100363, "passes": false}, "pre.men.b1940_1945.aime_p25": {"truth": 1178.0, "filled": 965.0, "gap": -0.19944526287254583, "seed_sd": 0.0, "tolerance": 0.14802915955392618, "passes": false}, "pre.men.b1940_1945.aime_p50": {"truth": 2795.0, "filled": 2571.0, "gap": -0.08353717832331053, "seed_sd": 0.0, "tolerance": 0.07497681768693197, "passes": false}, "pre.men.b1940_1945.aime_p75": {"truth": 4361.5, "filled": 4219.0, "gap": -0.033217901748935574, "seed_sd": 0.0, "tolerance": 0.04190766386026753, "passes": true}, "pre.men.b1940_1945.aime_p90": {"truth": 5432.800000000003, "filled": 5366.0, "gap": -0.01237190281401368, "seed_sd": 0.0, "tolerance": 0.023132300299791037, "passes": true}, "pre.men.b1940_1945.plevel": {"truth": 0.3760217560197671, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04021166929479513, "passes": false}, "pre.men.b1940_1945.pr_cross": {"truth": 0.376062044153939, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.05919050745324543, "passes": false}, "pre.men.b1940_1945.pr_in": {"truth": 0.36011521796896784, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.06500235072966114, "passes": false}, "pre.men.b1940_1945.pzero": {"truth": 0.21918652571223796, "filled": 1.0, "gap": 1.5178321960885375, "seed_sd": 0.0, "tolerance": 0.09503409717716321, "passes": false}, "pre.men.b1946_1955.paime_p10": {"truth": 305.0, "filled": 230.0, "gap": -0.2822324676842163, "seed_sd": 0.0, "tolerance": 0.11504556883383053, "passes": false}, "pre.men.b1946_1955.paime_p25": {"truth": 1022.0, "filled": 925.5, "gap": -0.09918263875009803, "seed_sd": 0.0, "tolerance": 0.08011617890879942, "passes": false}, "pre.men.b1946_1955.paime_p50": {"truth": 2610.0, "filled": 2482.0, "gap": -0.050285534552186206, "seed_sd": 0.0, "tolerance": 0.038612530439093365, "passes": false}, "pre.men.b1946_1955.paime_p75": {"truth": 4389.0, "filled": 4277.0, "gap": -0.025849581461324433, "seed_sd": 0.0, "tolerance": 0.022937547887772285, "passes": false}, "pre.men.b1946_1955.paime_p90": {"truth": 5809.0, "filled": 5730.0, "gap": -0.013692908283747585, "seed_sd": 0.0, "tolerance": 0.018317056116449435, "passes": true}, "pre.men.b1946_1955.ylevel": {"truth": 0.1479729717947154, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.02562210127460557, "passes": false}, "pre.men.b1946_1955.yr_cross": {"truth": 0.4294012571477206, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.03213928274663533, "passes": false}, "pre.men.b1946_1955.yzero": {"truth": 0.41447106660067734, "filled": 1.0, "gap": 0.8807521099778143, "seed_sd": 0.0, "tolerance": 0.021706485686521018, "passes": false}, "pre.men.b1946_1980.paime_p10": {"truth": 168.0, "filled": 121.0, "gap": -0.3281734338065174, "seed_sd": 0.0, "tolerance": 0.055691835388134804, "passes": false}, "pre.men.b1946_1980.paime_p25": {"truth": 480.0, "filled": 400.0, "gap": -0.1823215567939549, "seed_sd": 0.0, "tolerance": 0.03034406597699225, "passes": false}, "pre.men.b1946_1980.paime_p50": {"truth": 1237.0, "filled": 1127.0, "gap": -0.09312985835271093, "seed_sd": 0.0, "tolerance": 0.02474263079924987, "passes": false}, "pre.men.b1946_1980.paime_p75": {"truth": 2636.0, "filled": 2493.0, "gap": -0.055775812098840305, "seed_sd": 0.0, "tolerance": 0.02024689542128028, "passes": false}, "pre.men.b1946_1980.paime_p90": {"truth": 4302.0, "filled": 4169.899999999994, "gap": -0.031187976137720952, "seed_sd": 0.0, "tolerance": 0.01712820101125616, "passes": false}, "pre.men.b1946_1980.ylevel": {"truth": 0.09454735400371912, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.01520534060461241, "passes": false}, "pre.men.b1946_1980.yr_cross": {"truth": 0.49823745301361005, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.014632893176957621, "passes": false}, "pre.men.b1946_1980.yzero": {"truth": 0.415663002913214, "filled": 1.0, "gap": 0.877880436171331, "seed_sd": 0.0, "tolerance": 0.011384493587046594, "passes": false}, "pre.men.b1956_1965.paime_p10": {"truth": 249.0, "filled": 194.0, "gap": -0.24959473740137916, "seed_sd": 0.0, "tolerance": 0.11340392915219548, "passes": false}, "pre.men.b1956_1965.paime_p25": {"truth": 760.0, "filled": 679.0, "gap": -0.11269730572168069, "seed_sd": 0.0, "tolerance": 0.06443128496619711, "passes": false}, "pre.men.b1956_1965.paime_p50": {"truth": 1761.0, "filled": 1630.0, "gap": -0.07730181469539765, "seed_sd": 0.0, "tolerance": 0.03738718984350382, "passes": false}, "pre.men.b1956_1965.paime_p75": {"truth": 2956.0, "filled": 2793.0, "gap": -0.05672071612291507, "seed_sd": 0.0, "tolerance": 0.02599848784066106, "passes": false}, "pre.men.b1956_1965.paime_p90": {"truth": 4109.0, "filled": 3938.0, "gap": -0.04250670968433923, "seed_sd": 0.0, "tolerance": 0.022251423931640178, "passes": false}, "pre.men.b1956_1965.ylevel": {"truth": 0.09639046670103465, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.027201666962537053, "passes": false}, "pre.men.b1956_1965.yr_cross": {"truth": 0.40595517510382484, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.033463015007033206, "passes": false}, "pre.men.b1956_1965.yzero": {"truth": 0.4166095146652036, "filled": 1.0, "gap": 0.8756059115653633, "seed_sd": 0.0, "tolerance": 0.022693736955041323, "passes": false}, "pre.men.b1966_1980.paime_p10": {"truth": 104.0, "filled": 70.0, "gap": -0.39589565709201313, "seed_sd": 0.0, "tolerance": 0.07746758178850374, "passes": false}, "pre.men.b1966_1980.paime_p25": {"truth": 297.0, "filled": 233.0, "gap": -0.24269368523699963, "seed_sd": 0.0, "tolerance": 0.03917957142016487, "passes": false}, "pre.men.b1966_1980.paime_p50": {"truth": 648.0, "filled": 553.0, "gap": -0.15853269482993948, "seed_sd": 0.0, "tolerance": 0.029131134938009468, "passes": false}, "pre.men.b1966_1980.paime_p75": {"truth": 1218.0, "filled": 1107.0, "gap": -0.09555651556120548, "seed_sd": 0.0, "tolerance": 0.028246877693415603, "passes": false}, "pre.men.b1966_1980.paime_p90": {"truth": 1883.0, "filled": 1758.4000000000015, "gap": -0.06846194500779479, "seed_sd": 0.0, "tolerance": 0.02461284698314315, "passes": false}, "pre.men.b1966_1980.ylevel": {"truth": 0.055118361796504756, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.021382204828984532, "passes": false}, "pre.men.b1966_1980.yr_cross": {"truth": 0.4093896749871457, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.02246113667973179, "passes": false}, "pre.men.b1966_1980.yzero": {"truth": 0.41574858870643483, "filled": 1.0, "gap": 0.8776745554874777, "seed_sd": 0.0, "tolerance": 0.01675184764087755, "passes": false}, "pre.women.b1930_1934.aime_p10": {"truth": 51.40000000000009, "filled": 11.0, "gap": -1.5417428996627507, "seed_sd": 0.0, "tolerance": 0.3914251373153828, "passes": false}, "pre.women.b1930_1934.aime_p25": {"truth": 192.0, "filled": 83.0, "gap": -0.8386547642311832, "seed_sd": 0.0, "tolerance": 0.17419751750648213, "passes": false}, "pre.women.b1930_1934.aime_p50": {"truth": 526.0, "filled": 348.0, "gap": -0.4130987329632356, "seed_sd": 0.0, "tolerance": 0.10499905152764143, "passes": false}, "pre.women.b1930_1934.aime_p75": {"truth": 1055.0, "filled": 819.0, "gap": -0.2532119620570974, "seed_sd": 0.0, "tolerance": 0.0942504614966514, "passes": false}, "pre.women.b1930_1934.aime_p90": {"truth": 1663.0, "filled": 1328.6000000000004, "gap": -0.22449744396178684, "seed_sd": 0.0, "tolerance": 0.09146081269677721, "passes": false}, "pre.women.b1930_1934.plevel": {"truth": 0.17987838861322633, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.0866710956881841, "passes": false}, "pre.women.b1930_1934.pr_cross": {"truth": 0.5714979296198577, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.10849315671129696, "passes": false}, "pre.women.b1930_1934.pr_in": {"truth": 0.5486860492872828, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.1270201928993476, "passes": false}, "pre.women.b1930_1934.pzero": {"truth": 0.5514059876498446, "filled": 1.0, "gap": 0.5952839214563962, "seed_sd": 0.0, "tolerance": 0.04703504786975244, "passes": false}, "pre.women.b1930_1945.aime_p10": {"truth": 79.0, "filled": 27.0, "gap": -1.0736109864626924, "seed_sd": 0.0, "tolerance": 0.16621909484652816, "passes": false}, "pre.women.b1930_1945.aime_p25": {"truth": 280.0, "filled": 164.0, "gap": -0.5349231753450505, "seed_sd": 0.0, "tolerance": 0.09520232121757725, "passes": false}, "pre.women.b1930_1945.aime_p50": {"truth": 771.0, "filled": 601.0, "gap": -0.24909343902812164, "seed_sd": 0.0, "tolerance": 0.06218129686271377, "passes": false}, "pre.women.b1930_1945.aime_p75": {"truth": 1574.0, "filled": 1367.0, "gap": -0.14100159225339937, "seed_sd": 0.0, "tolerance": 0.0540210895745143, "passes": false}, "pre.women.b1930_1945.aime_p90": {"truth": 2560.0, "filled": 2353.0, "gap": -0.08431614874624582, "seed_sd": 0.0, "tolerance": 0.04855891958559449, "passes": false}, "pre.women.b1930_1945.plevel": {"truth": 0.18433938803699407, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.050000040503342155, "passes": false}, "pre.women.b1930_1945.pr_cross": {"truth": 0.4395941263612976, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.0677006808491353, "passes": false}, "pre.women.b1930_1945.pr_in": {"truth": 0.372715069734525, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.07412154624977427, "passes": false}, "pre.women.b1930_1945.pzero": {"truth": 0.5120706676834471, "filled": 1.0, "gap": 0.6692926406476696, "seed_sd": 0.0, "tolerance": 0.02913422225552515, "passes": false}, "pre.women.b1935_1939.aime_p10": {"truth": 75.0, "filled": 22.0, "gap": -1.226445660177994, "seed_sd": 0.0, "tolerance": 0.3512121935176347, "passes": false}, "pre.women.b1935_1939.aime_p25": {"truth": 267.0, "filled": 146.0, "gap": -0.6036420366919133, "seed_sd": 0.0, "tolerance": 0.180762984617868, "passes": false}, "pre.women.b1935_1939.aime_p50": {"truth": 726.0, "filled": 548.0, "gap": -0.28127472787678, "seed_sd": 0.0, "tolerance": 0.11655697880988479, "passes": false}, "pre.women.b1935_1939.aime_p75": {"truth": 1448.0, "filled": 1217.75, "gap": -0.1731784002590091, "seed_sd": 0.0, "tolerance": 0.08830721415636139, "passes": false}, "pre.women.b1935_1939.aime_p90": {"truth": 2293.8999999999996, "filled": 2012.8999999999996, "gap": -0.1306769574530957, "seed_sd": 0.0, "tolerance": 0.08570712841284106, "passes": false}, "pre.women.b1935_1939.plevel": {"truth": 0.18459645468504016, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09407530912273358, "passes": false}, "pre.women.b1935_1939.pr_cross": {"truth": 0.5177687037806673, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.1216792156910919, "passes": false}, "pre.women.b1935_1939.pr_in": {"truth": 0.48588537175161917, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.13268326765003252, "passes": false}, "pre.women.b1935_1939.pzero": {"truth": 0.5152445437812253, "filled": 1.0, "gap": 0.6631136487266858, "seed_sd": 0.0, "tolerance": 0.0510656392812699, "passes": false}, "pre.women.b1940_1945.aime_p10": {"truth": 110.0, "filled": 57.0, "gap": -0.6574290979578663, "seed_sd": 0.0, "tolerance": 0.26939110963753704, "passes": false}, "pre.women.b1940_1945.aime_p25": {"truth": 386.0, "filled": 283.0, "gap": -0.3103904718215933, "seed_sd": 0.0, "tolerance": 0.14001979176307694, "passes": false}, "pre.women.b1940_1945.aime_p50": {"truth": 1041.0, "filled": 903.0, "gap": -0.1422145151979839, "seed_sd": 0.0, "tolerance": 0.08949151188066314, "passes": false}, "pre.women.b1940_1945.aime_p75": {"truth": 2058.0, "filled": 1914.0, "gap": -0.0725393443810951, "seed_sd": 0.0, "tolerance": 0.06627004271644896, "passes": false}, "pre.women.b1940_1945.aime_p90": {"truth": 3179.7000000000007, "filled": 3045.0, "gap": -0.04328595155732273, "seed_sd": 0.0, "tolerance": 0.06421030653828377, "passes": true}, "pre.women.b1940_1945.plevel": {"truth": 0.19012308454067162, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.06996499056922599, "passes": false}, "pre.women.b1940_1945.pr_cross": {"truth": 0.3145537439002571, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.10514700123361043, "passes": false}, "pre.women.b1940_1945.pr_in": {"truth": 0.19634726483554435, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.10970200704526734, "passes": false}, "pre.women.b1940_1945.pzero": {"truth": 0.45477910425062734, "filled": 1.0, "gap": 0.7879434630807212, "seed_sd": 0.0, "tolerance": 0.04701699812231318, "passes": false}, "pre.women.b1946_1955.paime_p10": {"truth": 156.0, "filled": 117.0, "gap": -0.2876820724517808, "seed_sd": 0.0, "tolerance": 0.1092839592303832, "passes": false}, "pre.women.b1946_1955.paime_p25": {"truth": 516.0, "filled": 457.0, "gap": -0.12142337458735764, "seed_sd": 0.0, "tolerance": 0.06230795442282464, "passes": false}, "pre.women.b1946_1955.paime_p50": {"truth": 1311.0, "filled": 1228.0, "gap": -0.06540337505661231, "seed_sd": 0.0, "tolerance": 0.0404504662154056, "passes": false}, "pre.women.b1946_1955.paime_p75": {"truth": 2484.0, "filled": 2391.0, "gap": -0.03815847559504526, "seed_sd": 0.0, "tolerance": 0.0286716465849327, "passes": false}, "pre.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3699.300000000003, "gap": -0.02158039706903736, "seed_sd": 0.0, "tolerance": 0.03102321311868942, "passes": true}, "pre.women.b1946_1955.ylevel": {"truth": 0.09059452101103667, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.029569090149685853, "passes": false}, "pre.women.b1946_1955.yr_cross": {"truth": 0.34158903474392094, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.03669253843323441, "passes": false}, "pre.women.b1946_1955.yzero": {"truth": 0.5347643649882455, "filled": 1.0, "gap": 0.6259290683823043, "seed_sd": 0.0, "tolerance": 0.015053483015288157, "passes": false}, "pre.women.b1946_1980.paime_p10": {"truth": 103.0, "filled": 71.0, "gap": -0.3720491111883204, "seed_sd": 0.0, "tolerance": 0.05217708504283977, "passes": false}, "pre.women.b1946_1980.paime_p25": {"truth": 304.0, "filled": 248.0, "gap": -0.2035989552412394, "seed_sd": 0.0, "tolerance": 0.031237246580028546, "passes": false}, "pre.women.b1946_1980.paime_p50": {"truth": 752.0, "filled": 672.0, "gap": -0.11247798342669046, "seed_sd": 0.0, "tolerance": 0.021471785135136603, "passes": false}, "pre.women.b1946_1980.paime_p75": {"truth": 1583.75, "filled": 1484.0, "gap": -0.06505430790802347, "seed_sd": 0.0, "tolerance": 0.021247962622687265, "passes": false}, "pre.women.b1946_1980.paime_p90": {"truth": 2690.0, "filled": 2588.0, "gap": -0.038655816975094126, "seed_sd": 0.0, "tolerance": 0.019286406429246394, "passes": false}, "pre.women.b1946_1980.ylevel": {"truth": 0.06340849534928693, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.015450091379218515, "passes": false}, "pre.women.b1946_1980.yr_cross": {"truth": 0.43636860215681333, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.015684351176846988, "passes": false}, "pre.women.b1946_1980.yzero": {"truth": 0.4719000529035934, "filled": 1.0, "gap": 0.7509880681421656, "seed_sd": 0.0, "tolerance": 0.008556191699814752, "passes": false}, "pre.women.b1956_1965.paime_p10": {"truth": 136.0, "filled": 106.0, "gap": -0.24921579162398544, "seed_sd": 0.0, "tolerance": 0.10309020846799896, "passes": false}, "pre.women.b1956_1965.paime_p25": {"truth": 422.0, "filled": 367.0, "gap": -0.1396434659814414, "seed_sd": 0.0, "tolerance": 0.05400647590482275, "passes": false}, "pre.women.b1956_1965.paime_p50": {"truth": 996.0, "filled": 912.0, "gap": -0.0881072675102672, "seed_sd": 0.0, "tolerance": 0.03402127173688933, "passes": false}, "pre.women.b1956_1965.paime_p75": {"truth": 1839.0, "filled": 1727.0, "gap": -0.06283614645764235, "seed_sd": 0.0, "tolerance": 0.026722598015179132, "passes": false}, "pre.women.b1956_1965.paime_p90": {"truth": 2790.5999999999985, "filled": 2662.0, "gap": -0.047178906503049234, "seed_sd": 0.0, "tolerance": 0.026506809111477153, "passes": false}, "pre.women.b1956_1965.ylevel": {"truth": 0.06347343380872424, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.026809792541125404, "passes": false}, "pre.women.b1956_1965.yr_cross": {"truth": 0.3538451043271927, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.03047360671843986, "passes": false}, "pre.women.b1956_1965.yzero": {"truth": 0.47511142887567126, "filled": 1.0, "gap": 0.7442059153520724, "seed_sd": 0.0, "tolerance": 0.017102901274216892, "passes": false}, "pre.women.b1966_1980.paime_p10": {"truth": 73.0, "filled": 47.0, "gap": -0.4403118394383325, "seed_sd": 0.0, "tolerance": 0.0627268236453548, "passes": false}, "pre.women.b1966_1980.paime_p25": {"truth": 205.0, "filled": 156.0, "gap": -0.27315397188887136, "seed_sd": 0.0, "tolerance": 0.039676687855380324, "passes": false}, "pre.women.b1966_1980.paime_p50": {"truth": 458.0, "filled": 384.0, "gap": -0.17622663152645845, "seed_sd": 0.0, "tolerance": 0.02838346665574009, "passes": false}, "pre.women.b1966_1980.paime_p75": {"truth": 861.0, "filled": 772.0, "gap": -0.10910995440295412, "seed_sd": 0.0, "tolerance": 0.023922104075604994, "passes": false}, "pre.women.b1966_1980.paime_p90": {"truth": 1382.5999999999985, "filled": 1281.5999999999985, "gap": -0.07585648719707017, "seed_sd": 0.0, "tolerance": 0.02489092426663599, "passes": false}, "pre.women.b1966_1980.ylevel": {"truth": 0.04398552907354234, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.017769920520911725, "passes": false}, "pre.women.b1966_1980.yr_cross": {"truth": 0.318767197977468, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.02158530789673962, "passes": false}, "pre.women.b1966_1980.yzero": {"truth": 0.42453709903219916, "filled": 1.0, "gap": 0.8567558823917126, "seed_sd": 0.0, "tolerance": 0.014833297517636257, "passes": false}}}}, "alternative": {"passes": false, "n_gating": 136, "n_failing": 50, "cells": {"pre.men.b1930_1934.aime_p10": {"truth": 319.29999999999995, "filled": 333.45500000000004, "gap": 0.04337682399699627, "seed_sd": 7.198280130844114, "tolerance": 0.3785445178728429, "passes": true}, "pre.men.b1930_1934.aime_p25": {"truth": 961.0, "filled": 915.0, "gap": -0.04905034369477157, "seed_sd": 4.867994291394828, "tolerance": 0.1748310883030068, "passes": true}, "pre.men.b1930_1934.aime_p50": {"truth": 1871.0, "filled": 1756.225, "gap": -0.06330642816879539, "seed_sd": 4.537780004409762, "tolerance": 0.08482500842259981, "passes": true}, "pre.men.b1930_1934.aime_p75": {"truth": 2604.0, "filled": 2493.6375, "gap": -0.04330623648985288, "seed_sd": 4.3607663196664, "tolerance": 0.0526345782788754, "passes": true}, "pre.men.b1930_1934.aime_p90": {"truth": 3048.7000000000007, "filled": 2965.025, "gap": -0.027829806132161572, "seed_sd": 3.0473241065771446, "tolerance": 0.032507815796954, "passes": true}, "pre.men.b1930_1934.plevel": {"truth": 0.5292246059138269, "filled": 0.4512559722229019, "gap": -0.15937818319317099, "seed_sd": 0.00167539442003081, "tolerance": 0.05399925993701722, "passes": false}, "pre.men.b1930_1934.pr_cross": {"truth": 0.6005766012618865, "filled": 0.47070779788851286, "gap": -0.12986880337337364, "seed_sd": 0.00949471493032198, "tolerance": 0.10805017504629746, "passes": false}, "pre.men.b1930_1934.pr_in": {"truth": 0.6408121492387024, "filled": 0.5053907062038611, "gap": -0.13542144303484138, "seed_sd": 0.009558967224263285, "tolerance": 0.08754712042010183, "passes": false}, "pre.men.b1930_1934.pzero": {"truth": 0.23849477516714046, "filled": 0.19747749700699108, "gap": -0.18872276438791413, "seed_sd": 0.001329875025424285, "tolerance": 0.12848404727056906, "passes": false}, "pre.men.b1930_1945.aime_p10": {"truth": 335.0, "filled": 331.65500000000003, "gap": -0.010035259832665844, "seed_sd": 3.554311688491496, "tolerance": 0.17154652244393, "passes": true}, "pre.men.b1930_1945.aime_p25": {"truth": 1080.0, "filled": 1031.675, "gap": -0.0457773461558757, "seed_sd": 2.610832696367397, "tolerance": 0.08803344717643322, "passes": true}, "pre.men.b1930_1945.aime_p50": {"truth": 2266.0, "filled": 2180.175, "gap": -0.03861101379734322, "seed_sd": 2.852860983201545, "tolerance": 0.042293553277777104, "passes": true}, "pre.men.b1930_1945.aime_p75": {"truth": 3412.75, "filled": 3355.925, "gap": -0.016790977579084654, "seed_sd": 2.1644556870152436, "tolerance": 0.0339810042777025, "passes": true}, "pre.men.b1930_1945.aime_p90": {"truth": 4641.0, "filled": 4621.67, "gap": -0.004173748619134443, "seed_sd": 3.457120797791944, "tolerance": 0.03097131693794459, "passes": true}, "pre.men.b1930_1945.plevel": {"truth": 0.47071653085260085, "filled": 0.39204916712598037, "gap": -0.18286880924423354, "seed_sd": 0.0011927952381529446, "tolerance": 0.029173903026433003, "passes": false}, "pre.men.b1930_1945.pr_cross": {"truth": 0.5031824436185776, "filled": 0.4377212528977751, "gap": -0.06546119072080248, "seed_sd": 0.004959927945707018, "tolerance": 0.04396339753603818, "passes": false}, "pre.men.b1930_1945.pr_in": {"truth": 0.5486408320203577, "filled": 0.5051279148500414, "gap": -0.043512917170316356, "seed_sd": 0.004839756413913183, "tolerance": 0.04095810136062007, "passes": false}, "pre.men.b1930_1945.pzero": {"truth": 0.22496713863351978, "filled": 0.20559610321429558, "gap": -0.0900407608567817, "seed_sd": 0.0010776916227014178, "tolerance": 0.06285985273036603, "passes": false}, "pre.men.b1935_1939.aime_p10": {"truth": 342.0, "filled": 338.3, "gap": -0.010877661275942252, "seed_sd": 6.018392861186361, "tolerance": 0.35458282617348863, "passes": true}, "pre.men.b1935_1939.aime_p25": {"truth": 1099.5, "filled": 1038.55, "gap": -0.05703002147269842, "seed_sd": 6.176994670379337, "tolerance": 0.17611744524582296, "passes": true}, "pre.men.b1935_1939.aime_p50": {"truth": 2295.0, "filled": 2208.45, "gap": -0.03844193151511899, "seed_sd": 4.784569496115392, "tolerance": 0.08805365646562159, "passes": true}, "pre.men.b1935_1939.aime_p75": {"truth": 3360.0, "filled": 3280.05, "gap": -0.024082307792808066, "seed_sd": 4.784569496115392, "tolerance": 0.047075663502220186, "passes": true}, "pre.men.b1935_1939.aime_p90": {"truth": 4087.0, "filled": 4046.85, "gap": -0.009872403867532853, "seed_sd": 3.030980386815982, "tolerance": 0.03677617093870473, "passes": true}, "pre.men.b1935_1939.plevel": {"truth": 0.47885857925887765, "filled": 0.39017091820212124, "gap": -0.20482041726805789, "seed_sd": 0.0017982338283550564, "tolerance": 0.0505548909117149, "passes": false}, "pre.men.b1935_1939.pr_cross": {"truth": 0.5480805317128244, "filled": 0.4469978607767592, "gap": -0.10108267093606527, "seed_sd": 0.0067716623452240094, "tolerance": 0.08965478073758665, "passes": false}, "pre.men.b1935_1939.pr_in": {"truth": 0.5082293220171369, "filled": 0.45769779075131967, "gap": -0.05053153126581722, "seed_sd": 0.007181281967088367, "tolerance": 0.08171176478999366, "passes": true}, "pre.men.b1935_1939.pzero": {"truth": 0.21315539421109053, "filled": 0.2055349137510964, "gap": -0.036405533899280806, "seed_sd": 0.0014509009835175304, "tolerance": 0.12406583731865686, "passes": true}, "pre.men.b1940_1945.aime_p10": {"truth": 343.0, "filled": 326.0300000000001, "gap": -0.050741045493352566, "seed_sd": 2.2750592820500675, "tolerance": 0.2436998893100363, "passes": true}, "pre.men.b1940_1945.aime_p25": {"truth": 1178.0, "filled": 1148.925, "gap": -0.02499136264466273, "seed_sd": 4.693991119674061, "tolerance": 0.14802915955392618, "passes": true}, "pre.men.b1940_1945.aime_p50": {"truth": 2795.0, "filled": 2747.4, "gap": -0.017177096698095085, "seed_sd": 4.546832327074126, "tolerance": 0.07497681768693197, "passes": true}, "pre.men.b1940_1945.aime_p75": {"truth": 4361.5, "filled": 4334.75, "gap": -0.006152096448492017, "seed_sd": 2.4468024246479643, "tolerance": 0.04190766386026753, "passes": true}, "pre.men.b1940_1945.aime_p90": {"truth": 5432.800000000003, "filled": 5423.68, "gap": -0.0016801029698925163, "seed_sd": 2.50170468196981, "tolerance": 0.023132300299791037, "passes": true}, "pre.men.b1940_1945.plevel": {"truth": 0.3760217560197671, "filled": 0.3081285833193937, "gap": -0.1991298293071967, "seed_sd": 0.001180427121821751, "tolerance": 0.04021166929479513, "passes": false}, "pre.men.b1940_1945.pr_cross": {"truth": 0.376062044153939, "filled": 0.3566365358232604, "gap": -0.01942550833067863, "seed_sd": 0.007971024138862238, "tolerance": 0.05919050745324543, "passes": true}, "pre.men.b1940_1945.pr_in": {"truth": 0.36011521796896784, "filled": 0.3780255602140736, "gap": 0.017910342245105737, "seed_sd": 0.006428697396647492, "tolerance": 0.06500235072966114, "passes": true}, "pre.men.b1940_1945.pzero": {"truth": 0.21918652571223796, "filled": 0.2174785880666547, "gap": -0.007822682881506893, "seed_sd": 0.0016370070296969266, "tolerance": 0.09503409717716321, "passes": true}, "pre.men.b1946_1955.paime_p10": {"truth": 305.0, "filled": 297.55, "gap": -0.024729498516550485, "seed_sd": 1.7614288458371994, "tolerance": 0.11504556883383053, "passes": true}, "pre.men.b1946_1955.paime_p25": {"truth": 1022.0, "filled": 1012.25, "gap": -0.009585915851379134, "seed_sd": 1.7206180040292134, "tolerance": 0.08011617890879942, "passes": true}, "pre.men.b1946_1955.paime_p50": {"truth": 2610.0, "filled": 2589.5, "gap": -0.007885414452792894, "seed_sd": 3.137212977016048, "tolerance": 0.038612530439093365, "passes": true}, "pre.men.b1946_1955.paime_p75": {"truth": 4389.0, "filled": 4372.0, "gap": -0.0038808403918046963, "seed_sd": 2.724160904127902, "tolerance": 0.022937547887772285, "passes": true}, "pre.men.b1946_1955.paime_p90": {"truth": 5809.0, "filled": 5803.150000000001, "gap": -0.0010075654370478304, "seed_sd": 2.4639506146796553, "tolerance": 0.018317056116449435, "passes": true}, "pre.men.b1946_1955.ylevel": {"truth": 0.1479729717947154, "filled": 0.12961802758851015, "gap": -0.13243775807020652, "seed_sd": 0.00048550237036035763, "tolerance": 0.02562210127460557, "passes": false}, "pre.men.b1946_1955.yr_cross": {"truth": 0.4294012571477206, "filled": 0.42755734896351827, "gap": -0.0018439081842023253, "seed_sd": 0.003862677810953365, "tolerance": 0.03213928274663533, "passes": true}, "pre.men.b1946_1955.yzero": {"truth": 0.41447106660067734, "filled": 0.410343014999628, "gap": -0.010009737128729324, "seed_sd": 0.0015004624993788292, "tolerance": 0.021706485686521018, "passes": true}, "pre.men.b1946_1980.paime_p10": {"truth": 168.0, "filled": 185.3, "gap": 0.09801215388805495, "seed_sd": 0.6569466853317862, "tolerance": 0.055691835388134804, "passes": false}, "pre.men.b1946_1980.paime_p25": {"truth": 480.0, "filled": 505.15, "gap": 0.05106931097180478, "seed_sd": 0.7451598203705946, "tolerance": 0.03034406597699225, "passes": false}, "pre.men.b1946_1980.paime_p50": {"truth": 1237.0, "filled": 1256.15, "gap": 0.015362394256442258, "seed_sd": 0.9880869341680844, "tolerance": 0.02474263079924987, "passes": true}, "pre.men.b1946_1980.paime_p75": {"truth": 2636.0, "filled": 2630.7, "gap": -0.002012646168978449, "seed_sd": 2.4083189157584592, "tolerance": 0.02024689542128028, "passes": true}, "pre.men.b1946_1980.paime_p90": {"truth": 4302.0, "filled": 4293.74, "gap": -0.001921882826248833, "seed_sd": 2.0808399113924523, "tolerance": 0.01712820101125616, "passes": true}, "pre.men.b1946_1980.ylevel": {"truth": 0.09454735400371912, "filled": 0.09735423430573367, "gap": 0.029255417006314843, "seed_sd": 0.00024988348955974567, "tolerance": 0.01520534060461241, "passes": false}, "pre.men.b1946_1980.yr_cross": {"truth": 0.49823745301361005, "filled": 0.48664760466065066, "gap": -0.011589848352959398, "seed_sd": 0.002057777438744933, "tolerance": 0.014632893176957621, "passes": true}, "pre.men.b1946_1980.yzero": {"truth": 0.415663002913214, "filled": 0.44072148650673837, "gap": 0.058538283253072976, "seed_sd": 0.0006898565225956868, "tolerance": 0.011384493587046594, "passes": false}, "pre.men.b1956_1965.paime_p10": {"truth": 249.0, "filled": 260.71500000000003, "gap": 0.04597496021884595, "seed_sd": 1.8411309453416889, "tolerance": 0.11340392915219548, "passes": true}, "pre.men.b1956_1965.paime_p25": {"truth": 760.0, "filled": 769.0625, "gap": 0.011853807304998298, "seed_sd": 1.7638083050156288, "tolerance": 0.06443128496619711, "passes": true}, "pre.men.b1956_1965.paime_p50": {"truth": 1761.0, "filled": 1757.275, "gap": -0.002117515766600242, "seed_sd": 2.2211601994405865, "tolerance": 0.03738718984350382, "passes": true}, "pre.men.b1956_1965.paime_p75": {"truth": 2956.0, "filled": 2938.2625, "gap": -0.006018582831216257, "seed_sd": 2.4781611922425144, "tolerance": 0.02599848784066106, "passes": true}, "pre.men.b1956_1965.paime_p90": {"truth": 4109.0, "filled": 4099.719999999999, "gap": -0.0022610112059897602, "seed_sd": 3.206014085401028, "tolerance": 0.022251423931640178, "passes": true}, "pre.men.b1956_1965.ylevel": {"truth": 0.09639046670103465, "filled": 0.09251066477986337, "gap": -0.041083370793583374, "seed_sd": 0.00026376763707092367, "tolerance": 0.027201666962537053, "passes": false}, "pre.men.b1956_1965.yr_cross": {"truth": 0.40595517510382484, "filled": 0.3794224340718494, "gap": -0.02653274103197545, "seed_sd": 0.0031287878272003673, "tolerance": 0.033463015007033206, "passes": true}, "pre.men.b1956_1965.yzero": {"truth": 0.4166095146652036, "filled": 0.44938318702420965, "gap": 0.07572657958344886, "seed_sd": 0.0010074550971289882, "tolerance": 0.022693736955041323, "passes": false}, "pre.men.b1966_1980.paime_p10": {"truth": 104.0, "filled": 126.23000000000002, "gap": 0.1937147406233981, "seed_sd": 0.9205833390902217, "tolerance": 0.07746758178850374, "passes": false}, "pre.men.b1966_1980.paime_p25": {"truth": 297.0, "filled": 327.25, "gap": 0.0969922659873097, "seed_sd": 1.164157703189193, "tolerance": 0.03917957142016487, "passes": false}, "pre.men.b1966_1980.paime_p50": {"truth": 648.0, "filled": 691.7, "gap": 0.06526163925426509, "seed_sd": 1.2607433062326867, "tolerance": 0.029131134938009468, "passes": false}, "pre.men.b1966_1980.paime_p75": {"truth": 1218.0, "filled": 1261.4, "gap": 0.03501204595970808, "seed_sd": 1.353358395757909, "tolerance": 0.028246877693415603, "passes": false}, "pre.men.b1966_1980.paime_p90": {"truth": 1883.0, "filled": 1922.9300000000003, "gap": 0.02098381481301992, "seed_sd": 3.638550897209153, "tolerance": 0.02461284698314315, "passes": true}, "pre.men.b1966_1980.ylevel": {"truth": 0.055118361796504756, "filled": 0.07833231020180534, "gap": 0.35147725854645673, "seed_sd": 0.0003552933419435695, "tolerance": 0.021382204828984532, "passes": false}, "pre.men.b1966_1980.yr_cross": {"truth": 0.4093896749871457, "filled": 0.3745708705928209, "gap": -0.034818804394324776, "seed_sd": 0.0034799046915726293, "tolerance": 0.02246113667973179, "passes": false}, "pre.men.b1966_1980.yzero": {"truth": 0.41574858870643483, "filled": 0.45533442545360137, "gap": 0.09095142647226717, "seed_sd": 0.0007615661255219844, "tolerance": 0.01675184764087755, "passes": false}, "pre.women.b1930_1934.aime_p10": {"truth": 51.40000000000009, "filled": 69.41000000000001, "gap": 0.30039277689037114, "seed_sd": 1.6045658537272247, "tolerance": 0.3914251373153828, "passes": true}, "pre.women.b1930_1934.aime_p25": {"truth": 192.0, "filled": 202.675, "gap": 0.054108338845989756, "seed_sd": 2.0537193474023505, "tolerance": 0.17419751750648213, "passes": true}, "pre.women.b1930_1934.aime_p50": {"truth": 526.0, "filled": 520.6, "gap": -0.010319220177245292, "seed_sd": 3.0847673289381503, "tolerance": 0.10499905152764143, "passes": true}, "pre.women.b1930_1934.aime_p75": {"truth": 1055.0, "filled": 1030.425, "gap": -0.02356942843204557, "seed_sd": 3.7250044153983013, "tolerance": 0.0942504614966514, "passes": true}, "pre.women.b1930_1934.aime_p90": {"truth": 1663.0, "filled": 1599.4800000000002, "gap": -0.038944623789000765, "seed_sd": 6.872799169265325, "tolerance": 0.09146081269677721, "passes": true}, "pre.women.b1930_1934.plevel": {"truth": 0.17987838861322633, "filled": 0.166275116265377, "gap": -0.07863726033585361, "seed_sd": 0.0012300386992293107, "tolerance": 0.0866710956881841, "passes": true}, "pre.women.b1930_1934.pr_cross": {"truth": 0.5714979296198577, "filled": 0.4748597988631834, "gap": -0.09663813075667427, "seed_sd": 0.01059299078241411, "tolerance": 0.10849315671129696, "passes": true}, "pre.women.b1930_1934.pr_in": {"truth": 0.5486860492872828, "filled": 0.4506119091531285, "gap": -0.09807414013415433, "seed_sd": 0.010727888439518692, "tolerance": 0.1270201928993476, "passes": true}, "pre.women.b1930_1934.pzero": {"truth": 0.5514059876498446, "filled": 0.4412112523637662, "gap": -0.22294756647395808, "seed_sd": 0.0023214886803956196, "tolerance": 0.04703504786975244, "passes": false}, "pre.women.b1930_1945.aime_p10": {"truth": 79.0, "filled": 89.05, "gap": 0.11975015726864946, "seed_sd": 1.1459310165698642, "tolerance": 0.16621909484652816, "passes": true}, "pre.women.b1930_1945.aime_p25": {"truth": 280.0, "filled": 276.85, "gap": -0.011313759900272835, "seed_sd": 1.7252002172135514, "tolerance": 0.09520232121757725, "passes": true}, "pre.women.b1930_1945.aime_p50": {"truth": 771.0, "filled": 742.05, "gap": -0.038271747221502395, "seed_sd": 1.8202082009311031, "tolerance": 0.06218129686271377, "passes": true}, "pre.women.b1930_1945.aime_p75": {"truth": 1574.0, "filled": 1527.0, "gap": -0.030315123758716034, "seed_sd": 2.9019050004400464, "tolerance": 0.0540210895745143, "passes": true}, "pre.women.b1930_1945.aime_p90": {"truth": 2560.0, "filled": 2502.3, "gap": -0.022796949557932322, "seed_sd": 3.388836161031659, "tolerance": 0.04855891958559449, "passes": true}, "pre.women.b1930_1945.plevel": {"truth": 0.18433938803699407, "filled": 0.15443381522319435, "gap": -0.17701293468413826, "seed_sd": 0.0007451874725392632, "tolerance": 0.050000040503342155, "passes": false}, "pre.women.b1930_1945.pr_cross": {"truth": 0.4395941263612976, "filled": 0.38852744136167233, "gap": -0.051066684999625245, "seed_sd": 0.005151871755167338, "tolerance": 0.0677006808491353, "passes": true}, "pre.women.b1930_1945.pr_in": {"truth": 0.372715069734525, "filled": 0.34945624096577543, "gap": -0.023258828768749573, "seed_sd": 0.006480298607911677, "tolerance": 0.07412154624977427, "passes": true}, "pre.women.b1930_1945.pzero": {"truth": 0.5120706676834471, "filled": 0.42671014627380915, "gap": -0.18235766995166947, "seed_sd": 0.0016892997635133166, "tolerance": 0.02913422225552515, "passes": false}, "pre.women.b1935_1939.aime_p10": {"truth": 75.0, "filled": 87.8, "gap": 0.15757338710476088, "seed_sd": 1.7044832524535805, "tolerance": 0.3512121935176347, "passes": true}, "pre.women.b1935_1939.aime_p25": {"truth": 267.0, "filled": 264.3625, "gap": -0.00992739104138085, "seed_sd": 2.5150533634602334, "tolerance": 0.180762984617868, "passes": true}, "pre.women.b1935_1939.aime_p50": {"truth": 726.0, "filled": 693.825, "gap": -0.04533024749930359, "seed_sd": 2.749521489469146, "tolerance": 0.11655697880988479, "passes": true}, "pre.women.b1935_1939.aime_p75": {"truth": 1448.0, "filled": 1392.45, "gap": -0.03911850843160991, "seed_sd": 4.525948577515634, "tolerance": 0.08830721415636139, "passes": true}, "pre.women.b1935_1939.aime_p90": {"truth": 2293.8999999999996, "filled": 2210.7949999999996, "gap": -0.03690124642828341, "seed_sd": 7.21989028929614, "tolerance": 0.08570712841284106, "passes": true}, "pre.women.b1935_1939.plevel": {"truth": 0.18459645468504016, "filled": 0.15058972619927874, "gap": -0.20361302258404357, "seed_sd": 0.0011346805174700786, "tolerance": 0.09407530912273358, "passes": false}, "pre.women.b1935_1939.pr_cross": {"truth": 0.5177687037806673, "filled": 0.4330974292581584, "gap": -0.08467127452250889, "seed_sd": 0.008445900072306159, "tolerance": 0.1216792156910919, "passes": true}, "pre.women.b1935_1939.pr_in": {"truth": 0.48588537175161917, "filled": 0.3844627635977695, "gap": -0.10142260815384968, "seed_sd": 0.012464065495063868, "tolerance": 0.13268326765003252, "passes": true}, "pre.women.b1935_1939.pzero": {"truth": 0.5152445437812253, "filled": 0.43101465947935835, "gap": -0.1784995280169943, "seed_sd": 0.0024295006714432823, "tolerance": 0.0510656392812699, "passes": false}, "pre.women.b1940_1945.aime_p10": {"truth": 110.0, "filled": 109.88000000000002, "gap": -0.0010915045653430155, "seed_sd": 1.3820655784577995, "tolerance": 0.26939110963753704, "passes": true}, "pre.women.b1940_1945.aime_p25": {"truth": 386.0, "filled": 366.6375, "gap": -0.051463747964931805, "seed_sd": 2.6251253102922236, "tolerance": 0.14001979176307694, "passes": true}, "pre.women.b1940_1945.aime_p50": {"truth": 1041.0, "filled": 1006.05, "gap": -0.034150017401113786, "seed_sd": 2.6102026539594285, "tolerance": 0.08949151188066314, "passes": true}, "pre.women.b1940_1945.aime_p75": {"truth": 2058.0, "filled": 2005.95, "gap": -0.02561687340707941, "seed_sd": 3.0753690407562644, "tolerance": 0.06627004271644896, "passes": true}, "pre.women.b1940_1945.aime_p90": {"truth": 3179.7000000000007, "filled": 3143.15, "gap": -0.011561370939153548, "seed_sd": 4.8068481850049665, "tolerance": 0.06421030653828377, "passes": true}, "pre.women.b1940_1945.plevel": {"truth": 0.19012308454067162, "filled": 0.14273877017557055, "gap": -0.2866554981221232, "seed_sd": 0.000855197366426852, "tolerance": 0.06996499056922599, "passes": false}, "pre.women.b1940_1945.pr_cross": {"truth": 0.3145537439002571, "filled": 0.31177953224975996, "gap": -0.002774211650497127, "seed_sd": 0.010577622662610095, "tolerance": 0.10514700123361043, "passes": true}, "pre.women.b1940_1945.pr_in": {"truth": 0.19634726483554435, "filled": 0.2748579888301229, "gap": 0.07851072399457856, "seed_sd": 0.00870178568609491, "tolerance": 0.10970200704526734, "passes": true}, "pre.women.b1940_1945.pzero": {"truth": 0.45477910425062734, "filled": 0.40196372899399285, "gap": -0.12344995773614209, "seed_sd": 0.0026140225069639154, "tolerance": 0.04701699812231318, "passes": false}, "pre.women.b1946_1955.paime_p10": {"truth": 156.0, "filled": 155.46999999999997, "gap": -0.003403220287887976, "seed_sd": 0.8933084573650976, "tolerance": 0.1092839592303832, "passes": true}, "pre.women.b1946_1955.paime_p25": {"truth": 516.0, "filled": 513.2, "gap": -0.0054411327400814, "seed_sd": 1.1516578439248717, "tolerance": 0.06230795442282464, "passes": true}, "pre.women.b1946_1955.paime_p50": {"truth": 1311.0, "filled": 1302.675, "gap": -0.006370362155514009, "seed_sd": 1.5241477342401797, "tolerance": 0.0404504662154056, "passes": true}, "pre.women.b1946_1955.paime_p75": {"truth": 2484.0, "filled": 2469.95, "gap": -0.005672256551217281, "seed_sd": 1.8771478925557026, "tolerance": 0.0286716465849327, "passes": true}, "pre.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3774.0200000000013, "gap": -0.0015832632821144443, "seed_sd": 2.8162311508004154, "tolerance": 0.03102321311868942, "passes": true}, "pre.women.b1946_1955.ylevel": {"truth": 0.09059452101103667, "filled": 0.08651283195139477, "gap": -0.046100987533894244, "seed_sd": 0.00039502547096333595, "tolerance": 0.029569090149685853, "passes": false}, "pre.women.b1946_1955.yr_cross": {"truth": 0.34158903474392094, "filled": 0.38173946433515016, "gap": 0.04015042959122922, "seed_sd": 0.00490792684904385, "tolerance": 0.03669253843323441, "passes": false}, "pre.women.b1946_1955.yzero": {"truth": 0.5347643649882455, "filled": 0.49023495161554864, "gap": -0.08694144132228299, "seed_sd": 0.0011450982333459034, "tolerance": 0.015053483015288157, "passes": false}, "pre.women.b1946_1980.paime_p10": {"truth": 103.0, "filled": 114.75, "gap": 0.10802686071101864, "seed_sd": 0.5501196042201808, "tolerance": 0.05217708504283977, "passes": false}, "pre.women.b1946_1980.paime_p25": {"truth": 304.0, "filled": 319.85, "gap": 0.050824434489924464, "seed_sd": 0.6708203932499369, "tolerance": 0.031237246580028546, "passes": false}, "pre.women.b1946_1980.paime_p50": {"truth": 752.0, "filled": 771.15, "gap": 0.025146583219783025, "seed_sd": 0.8127277008872491, "tolerance": 0.021471785135136603, "passes": false}, "pre.women.b1946_1980.paime_p75": {"truth": 1583.75, "filled": 1589.05, "gap": 0.0033409007373377264, "seed_sd": 1.2763022245616642, "tolerance": 0.021247962622687265, "passes": true}, "pre.women.b1946_1980.paime_p90": {"truth": 2690.0, "filled": 2691.95, "gap": 0.0007246444449799938, "seed_sd": 1.9049796241485244, "tolerance": 0.019286406429246394, "passes": true}, "pre.women.b1946_1980.ylevel": {"truth": 0.06340849534928693, "filled": 0.07062478439851178, "gap": 0.10778328818546301, "seed_sd": 0.00021066219000565875, "tolerance": 0.015450091379218515, "passes": false}, "pre.women.b1946_1980.yr_cross": {"truth": 0.43636860215681333, "filled": 0.45372177216986065, "gap": 0.017353170013047314, "seed_sd": 0.002315286295492016, "tolerance": 0.015684351176846988, "passes": false}, "pre.women.b1946_1980.yzero": {"truth": 0.4719000529035934, "filled": 0.4926267796826947, "gap": 0.042984637320677144, "seed_sd": 0.0006742279130238006, "tolerance": 0.008556191699814752, "passes": false}, "pre.women.b1956_1965.paime_p10": {"truth": 136.0, "filled": 147.4, "gap": 0.08049509401918353, "seed_sd": 1.2732056517228265, "tolerance": 0.10309020846799896, "passes": true}, "pre.women.b1956_1965.paime_p25": {"truth": 422.0, "filled": 429.75, "gap": 0.01819833022694617, "seed_sd": 1.0699237552766379, "tolerance": 0.05400647590482275, "passes": true}, "pre.women.b1956_1965.paime_p50": {"truth": 996.0, "filled": 1001.6, "gap": 0.00560674276123585, "seed_sd": 1.6982963599783725, "tolerance": 0.03402127173688933, "passes": true}, "pre.women.b1956_1965.paime_p75": {"truth": 1839.0, "filled": 1839.85, "gap": 0.000462100936502452, "seed_sd": 1.8144159564878983, "tolerance": 0.026722598015179132, "passes": true}, "pre.women.b1956_1965.paime_p90": {"truth": 2790.5999999999985, "filled": 2792.7999999999997, "gap": 0.0007880503327202248, "seed_sd": 3.231424649153998, "tolerance": 0.026506809111477153, "passes": true}, "pre.women.b1956_1965.ylevel": {"truth": 0.06347343380872424, "filled": 0.06691935721416442, "gap": 0.05286681764248957, "seed_sd": 0.0003901465800188738, "tolerance": 0.026809792541125404, "passes": false}, "pre.women.b1956_1965.yr_cross": {"truth": 0.3538451043271927, "filled": 0.3495167466131729, "gap": -0.004328357714019793, "seed_sd": 0.003718149045814111, "tolerance": 0.03047360671843986, "passes": true}, "pre.women.b1956_1965.yzero": {"truth": 0.47511142887567126, "filled": 0.4996823706588375, "gap": 0.05042327424751103, "seed_sd": 0.0011326514137311617, "tolerance": 0.017102901274216892, "passes": false}, "pre.women.b1966_1980.paime_p10": {"truth": 73.0, "filled": 87.52000000000002, "gap": 0.18140789752527997, "seed_sd": 0.6786984447804205, "tolerance": 0.0627268236453548, "passes": false}, "pre.women.b1966_1980.paime_p25": {"truth": 205.0, "filled": 224.55, "gap": 0.09108842039533904, "seed_sd": 0.8870412083230169, "tolerance": 0.039676687855380324, "passes": false}, "pre.women.b1966_1980.paime_p50": {"truth": 458.0, "filled": 486.35, "gap": 0.06005934520126388, "seed_sd": 1.4964871146156007, "tolerance": 0.02838346665574009, "passes": false}, "pre.women.b1966_1980.paime_p75": {"truth": 861.0, "filled": 901.2, "gap": 0.0456327041303588, "seed_sd": 1.9695043458751373, "tolerance": 0.023922104075604994, "passes": false}, "pre.women.b1966_1980.paime_p90": {"truth": 1382.5999999999985, "filled": 1421.5699999999997, "gap": 0.027796110124051587, "seed_sd": 2.5250534126710935, "tolerance": 0.02489092426663599, "passes": false}, "pre.women.b1966_1980.ylevel": {"truth": 0.04398552907354234, "filled": 0.06226928105552239, "gap": 0.3476075280866513, "seed_sd": 0.0003490420029782692, "tolerance": 0.017769920520911725, "passes": false}, "pre.women.b1966_1980.yr_cross": {"truth": 0.318767197977468, "filled": 0.3300755287219385, "gap": 0.011308330744470518, "seed_sd": 0.004440189459906143, "tolerance": 0.02158530789673962, "passes": true}, "pre.women.b1966_1980.yzero": {"truth": 0.42453709903219916, "filled": 0.4886847647452675, "gap": 0.14071823213377666, "seed_sd": 0.0010570800354365248, "tolerance": 0.014833297517636257, "passes": false}}, "tier": "not_adopted"}, "primary": {"passes": true, "n_gating": 136, "n_failing": 0, "cells": {"pre.men.b1930_1934.aime_p10": {"truth": 319.29999999999995, "filled": 319.01000000000005, "gap": -0.0009086494648462562, "seed_sd": 3.4908753237999974, "tolerance": 0.3785445178728429, "passes": true}, "pre.men.b1930_1934.aime_p25": {"truth": 961.0, "filled": 955.5125, "gap": -0.0057265632195964145, "seed_sd": 5.253742400472859, "tolerance": 0.1748310883030068, "passes": true}, "pre.men.b1930_1934.aime_p50": {"truth": 1871.0, "filled": 1864.975, "gap": -0.003225399111755678, "seed_sd": 3.4659053651246743, "tolerance": 0.08482500842259981, "passes": true}, "pre.men.b1930_1934.aime_p75": {"truth": 2604.0, "filled": 2607.125, "gap": 0.0011993572883390868, "seed_sd": 2.5860201081971503, "tolerance": 0.0526345782788754, "passes": true}, "pre.men.b1930_1934.aime_p90": {"truth": 3048.7000000000007, "filled": 3048.425, "gap": -9.02064498227162e-05, "seed_sd": 2.036735003303978, "tolerance": 0.032507815796954, "passes": true}, "pre.men.b1930_1934.plevel": {"truth": 0.5292246059138269, "filled": 0.5291854236691471, "gap": -7.403982124609687e-05, "seed_sd": 0.0013655795254600492, "tolerance": 0.05399925993701722, "passes": true}, "pre.men.b1930_1934.pr_cross": {"truth": 0.6005766012618865, "filled": 0.6132148513475164, "gap": 0.01263825008562991, "seed_sd": 0.0042711947827688895, "tolerance": 0.10805017504629746, "passes": true}, "pre.men.b1930_1934.pr_in": {"truth": 0.6408121492387024, "filled": 0.6569132735924044, "gap": 0.016101124353701923, "seed_sd": 0.005878717774556174, "tolerance": 0.08754712042010183, "passes": true}, "pre.men.b1930_1934.pzero": {"truth": 0.23849477516714046, "filled": 0.2343592497524301, "gap": -0.017492209622133048, "seed_sd": 0.0017178638387153479, "tolerance": 0.12848404727056906, "passes": true}, "pre.men.b1930_1945.aime_p10": {"truth": 335.0, "filled": 334.91, "gap": -0.00026869281109842547, "seed_sd": 1.7480515468733664, "tolerance": 0.17154652244393, "passes": true}, "pre.men.b1930_1945.aime_p25": {"truth": 1080.0, "filled": 1080.075, "gap": 6.9442033289846e-05, "seed_sd": 2.3114417920195907, "tolerance": 0.08803344717643322, "passes": true}, "pre.men.b1930_1945.aime_p50": {"truth": 2266.0, "filled": 2269.95, "gap": 0.00174164221319284, "seed_sd": 1.4039418191573851, "tolerance": 0.042293553277777104, "passes": true}, "pre.men.b1930_1945.aime_p75": {"truth": 3412.75, "filled": 3411.5875, "gap": -0.00034069241483081214, "seed_sd": 2.272945862693795, "tolerance": 0.0339810042777025, "passes": true}, "pre.men.b1930_1945.aime_p90": {"truth": 4641.0, "filled": 4645.5, "gap": 0.0009691488401912807, "seed_sd": 3.063365881888381, "tolerance": 0.03097131693794459, "passes": true}, "pre.men.b1930_1945.plevel": {"truth": 0.47071653085260085, "filled": 0.4709284946535627, "gap": 0.00045019895778231067, "seed_sd": 0.0007883479291290621, "tolerance": 0.029173903026433003, "passes": true}, "pre.men.b1930_1945.pr_cross": {"truth": 0.5031824436185776, "filled": 0.5142593494210748, "gap": 0.011076905802497206, "seed_sd": 0.0033007110869924168, "tolerance": 0.04396339753603818, "passes": true}, "pre.men.b1930_1945.pr_in": {"truth": 0.5486408320203577, "filled": 0.5533205928140609, "gap": 0.004679760793703136, "seed_sd": 0.0028791732273558686, "tolerance": 0.04095810136062007, "passes": true}, "pre.men.b1930_1945.pzero": {"truth": 0.22496713863351978, "filled": 0.2234473251045051, "gap": -0.00677863660306266, "seed_sd": 0.0010310919431983968, "tolerance": 0.06285985273036603, "passes": true}, "pre.men.b1935_1939.aime_p10": {"truth": 342.0, "filled": 351.05, "gap": 0.026117926400653246, "seed_sd": 3.9132030224276426, "tolerance": 0.35458282617348863, "passes": true}, "pre.men.b1935_1939.aime_p25": {"truth": 1099.5, "filled": 1097.875, "gap": -0.0014790377575346625, "seed_sd": 3.1366886663285034, "tolerance": 0.17611744524582296, "passes": true}, "pre.men.b1935_1939.aime_p50": {"truth": 2295.0, "filled": 2303.65, "gap": 0.0037619780594635444, "seed_sd": 3.2971279367927098, "tolerance": 0.08805365646562159, "passes": true}, "pre.men.b1935_1939.aime_p75": {"truth": 3360.0, "filled": 3361.725, "gap": 0.0005132611161187128, "seed_sd": 3.0714089000943767, "tolerance": 0.047075663502220186, "passes": true}, "pre.men.b1935_1939.aime_p90": {"truth": 4087.0, "filled": 4089.05, "gap": 0.0005014646541940948, "seed_sd": 4.148239956516752, "tolerance": 0.03677617093870473, "passes": true}, "pre.men.b1935_1939.plevel": {"truth": 0.47885857925887765, "filled": 0.480522922686636, "gap": 0.0034696209881766027, "seed_sd": 0.0014865531284624595, "tolerance": 0.0505548909117149, "passes": true}, "pre.men.b1935_1939.pr_cross": {"truth": 0.5480805317128244, "filled": 0.575558051296075, "gap": 0.02747751958325051, "seed_sd": 0.004077817588025134, "tolerance": 0.08965478073758665, "passes": true}, "pre.men.b1935_1939.pr_in": {"truth": 0.5082293220171369, "filled": 0.5135920240451662, "gap": 0.005362702028029354, "seed_sd": 0.006155107199129598, "tolerance": 0.08171176478999366, "passes": true}, "pre.men.b1935_1939.pzero": {"truth": 0.21315539421109053, "filled": 0.21182816002338956, "gap": -0.006246069945369914, "seed_sd": 0.0015124780729217227, "tolerance": 0.12406583731865686, "passes": true}, "pre.men.b1940_1945.aime_p10": {"truth": 343.0, "filled": 334.09000000000003, "gap": -0.026320029409510504, "seed_sd": 2.4721394952976294, "tolerance": 0.2436998893100363, "passes": true}, "pre.men.b1940_1945.aime_p25": {"truth": 1178.0, "filled": 1176.775, "gap": -0.0010404392016276631, "seed_sd": 3.614572115610279, "tolerance": 0.14802915955392618, "passes": true}, "pre.men.b1940_1945.aime_p50": {"truth": 2795.0, "filled": 2792.95, "gap": -0.000733721701864809, "seed_sd": 3.677456214630749, "tolerance": 0.07497681768693197, "passes": true}, "pre.men.b1940_1945.aime_p75": {"truth": 4361.5, "filled": 4359.025, "gap": -0.0005676263909464296, "seed_sd": 3.290636636531632, "tolerance": 0.04190766386026753, "passes": true}, "pre.men.b1940_1945.aime_p90": {"truth": 5432.800000000003, "filled": 5432.7, "gap": -1.8406884176869198e-05, "seed_sd": 1.2422390650503299, "tolerance": 0.023132300299791037, "passes": true}, "pre.men.b1940_1945.plevel": {"truth": 0.3760217560197671, "filled": 0.37489008956602926, "gap": -0.0030141149514301135, "seed_sd": 0.0009568302876987612, "tolerance": 0.04021166929479513, "passes": true}, "pre.men.b1940_1945.pr_cross": {"truth": 0.376062044153939, "filled": 0.37480361977220633, "gap": -0.0012584243817326812, "seed_sd": 0.0061954412945916595, "tolerance": 0.05919050745324543, "passes": true}, "pre.men.b1940_1945.pr_in": {"truth": 0.36011521796896784, "filled": 0.3591838645046824, "gap": -0.0009313534642854115, "seed_sd": 0.006786464340961721, "tolerance": 0.06500235072966114, "passes": true}, "pre.men.b1940_1945.pzero": {"truth": 0.21918652571223796, "filled": 0.2212452965418384, "gap": 0.009348942200510635, "seed_sd": 0.0012275780068450736, "tolerance": 0.09503409717716321, "passes": true}, "pre.men.b1946_1955.paime_p10": {"truth": 305.0, "filled": 303.13, "gap": -0.006150020206342255, "seed_sd": 1.4542496708829384, "tolerance": 0.11504556883383053, "passes": true}, "pre.men.b1946_1955.paime_p25": {"truth": 1022.0, "filled": 1023.3, "gap": 0.0012712073290597203, "seed_sd": 2.0862961388420795, "tolerance": 0.08011617890879942, "passes": true}, "pre.men.b1946_1955.paime_p50": {"truth": 2610.0, "filled": 2607.7, "gap": -0.0008816145615773152, "seed_sd": 1.8381913307436342, "tolerance": 0.038612530439093365, "passes": true}, "pre.men.b1946_1955.paime_p75": {"truth": 4389.0, "filled": 4387.375, "gap": -0.0003703123484513071, "seed_sd": 1.5293187337745144, "tolerance": 0.022937547887772285, "passes": true}, "pre.men.b1946_1955.paime_p90": {"truth": 5809.0, "filled": 5810.430000000001, "gap": 0.00024613944181695047, "seed_sd": 3.429454460222852, "tolerance": 0.018317056116449435, "passes": true}, "pre.men.b1946_1955.ylevel": {"truth": 0.1479729717947154, "filled": 0.14659159386435963, "gap": -0.009379186894413527, "seed_sd": 0.0002957004696019843, "tolerance": 0.02562210127460557, "passes": true}, "pre.men.b1946_1955.yr_cross": {"truth": 0.4294012571477206, "filled": 0.4399009712087537, "gap": 0.010499714061033116, "seed_sd": 0.0031621441318267795, "tolerance": 0.03213928274663533, "passes": true}, "pre.men.b1946_1955.yzero": {"truth": 0.41447106660067734, "filled": 0.4158327429878276, "gap": 0.0032799502836826644, "seed_sd": 0.0006758004614627337, "tolerance": 0.021706485686521018, "passes": true}, "pre.men.b1946_1980.paime_p10": {"truth": 168.0, "filled": 168.15, "gap": 0.0008924587830199116, "seed_sd": 0.36634754853252327, "tolerance": 0.055691835388134804, "passes": true}, "pre.men.b1946_1980.paime_p25": {"truth": 480.0, "filled": 479.2, "gap": -0.0016680571006970624, "seed_sd": 0.523148363780597, "tolerance": 0.03034406597699225, "passes": true}, "pre.men.b1946_1980.paime_p50": {"truth": 1237.0, "filled": 1234.35, "gap": -0.0021445776726558563, "seed_sd": 0.7451598203705947, "tolerance": 0.02474263079924987, "passes": true}, "pre.men.b1946_1980.paime_p75": {"truth": 2636.0, "filled": 2636.7625, "gap": 0.00028922220764382445, "seed_sd": 1.2016299237636396, "tolerance": 0.02024689542128028, "passes": true}, "pre.men.b1946_1980.paime_p90": {"truth": 4302.0, "filled": 4304.3949999999995, "gap": 0.0005565628958059676, "seed_sd": 1.633361081038826, "tolerance": 0.01712820101125616, "passes": true}, "pre.men.b1946_1980.ylevel": {"truth": 0.09454735400371912, "filled": 0.09367892426336906, "gap": -0.009227573433769454, "seed_sd": 9.960753662665636e-05, "tolerance": 0.01520534060461241, "passes": true}, "pre.men.b1946_1980.yr_cross": {"truth": 0.49823745301361005, "filled": 0.5047654750530509, "gap": 0.006528022039440862, "seed_sd": 0.0008904235198013295, "tolerance": 0.014632893176957621, "passes": true}, "pre.men.b1946_1980.yzero": {"truth": 0.415663002913214, "filled": 0.41931923269525295, "gap": 0.008757678893166476, "seed_sd": 0.0005288771136011216, "tolerance": 0.011384493587046594, "passes": true}, "pre.men.b1956_1965.paime_p10": {"truth": 249.0, "filled": 249.91500000000005, "gap": 0.0036679635844345526, "seed_sd": 0.9702278625473394, "tolerance": 0.11340392915219548, "passes": true}, "pre.men.b1956_1965.paime_p25": {"truth": 760.0, "filled": 760.2125, "gap": 0.0002795661808914218, "seed_sd": 1.6824852388497704, "tolerance": 0.06443128496619711, "passes": true}, "pre.men.b1956_1965.paime_p50": {"truth": 1761.0, "filled": 1759.05, "gap": -0.0011079389210228996, "seed_sd": 1.6928642809405043, "tolerance": 0.03738718984350382, "passes": true}, "pre.men.b1956_1965.paime_p75": {"truth": 2956.0, "filled": 2952.125, "gap": -0.001311753070776689, "seed_sd": 1.6984900414935118, "tolerance": 0.02599848784066106, "passes": true}, "pre.men.b1956_1965.paime_p90": {"truth": 4109.0, "filled": 4109.58, "gap": 0.00014114360411632276, "seed_sd": 2.4312872916908015, "tolerance": 0.022251423931640178, "passes": true}, "pre.men.b1956_1965.ylevel": {"truth": 0.09639046670103465, "filled": 0.09526375922291459, "gap": -0.011757846225209256, "seed_sd": 0.0002008564053301557, "tolerance": 0.027201666962537053, "passes": true}, "pre.men.b1956_1965.yr_cross": {"truth": 0.40595517510382484, "filled": 0.4183289514486261, "gap": 0.012373776344801246, "seed_sd": 0.0029119168385594732, "tolerance": 0.033463015007033206, "passes": true}, "pre.men.b1956_1965.yzero": {"truth": 0.4166095146652036, "filled": 0.42287151658551086, "gap": 0.014919022193501164, "seed_sd": 0.0009039797761472841, "tolerance": 0.022693736955041323, "passes": true}, "pre.men.b1966_1980.paime_p10": {"truth": 104.0, "filled": 104.58000000000001, "gap": 0.005561429618611946, "seed_sd": 0.5908067633151707, "tolerance": 0.07746758178850374, "passes": true}, "pre.men.b1966_1980.paime_p25": {"truth": 297.0, "filled": 295.5, "gap": -0.005063301956546695, "seed_sd": 0.6882472016116853, "tolerance": 0.03917957142016487, "passes": true}, "pre.men.b1966_1980.paime_p50": {"truth": 648.0, "filled": 650.0, "gap": 0.0030816665374082675, "seed_sd": 0.8583950752789521, "tolerance": 0.029131134938009468, "passes": true}, "pre.men.b1966_1980.paime_p75": {"truth": 1218.0, "filled": 1217.95, "gap": -4.105174573165726e-05, "seed_sd": 0.7591546545162483, "tolerance": 0.028246877693415603, "passes": true}, "pre.men.b1966_1980.paime_p90": {"truth": 1883.0, "filled": 1882.02, "gap": -0.0005205815757323151, "seed_sd": 1.0092206477224783, "tolerance": 0.02461284698314315, "passes": true}, "pre.men.b1966_1980.ylevel": {"truth": 0.055118361796504756, "filled": 0.05482193056978226, "gap": -0.005392598813348748, "seed_sd": 0.00010597279122111407, "tolerance": 0.021382204828984532, "passes": true}, "pre.men.b1966_1980.yr_cross": {"truth": 0.4093896749871457, "filled": 0.4115686971693808, "gap": 0.0021790221822351463, "seed_sd": 0.0019117001884655493, "tolerance": 0.02246113667973179, "passes": true}, "pre.men.b1966_1980.yzero": {"truth": 0.41574858870643483, "filled": 0.4189394792286537, "gap": 0.007645745013141858, "seed_sd": 0.0007869981860754034, "tolerance": 0.01675184764087755, "passes": true}, "pre.women.b1930_1934.aime_p10": {"truth": 51.40000000000009, "filled": 55.19000000000001, "gap": 0.07114360499028916, "seed_sd": 1.0104194022390238, "tolerance": 0.3914251373153828, "passes": true}, "pre.women.b1930_1934.aime_p25": {"truth": 192.0, "filled": 189.6, "gap": -0.012578782206859707, "seed_sd": 2.588435821108957, "tolerance": 0.17419751750648213, "passes": true}, "pre.women.b1930_1934.aime_p50": {"truth": 526.0, "filled": 520.85, "gap": -0.009839120307252536, "seed_sd": 3.0482954684803594, "tolerance": 0.10499905152764143, "passes": true}, "pre.women.b1930_1934.aime_p75": {"truth": 1055.0, "filled": 1054.4, "gap": -0.0005688821619243001, "seed_sd": 3.8784153030790676, "tolerance": 0.0942504614966514, "passes": true}, "pre.women.b1930_1934.aime_p90": {"truth": 1663.0, "filled": 1662.6100000000001, "gap": -0.00023454343821960322, "seed_sd": 4.867280663022405, "tolerance": 0.09146081269677721, "passes": true}, "pre.women.b1930_1934.plevel": {"truth": 0.17987838861322633, "filled": 0.177081021329544, "gap": -0.015673628277671492, "seed_sd": 0.0011236798865642981, "tolerance": 0.0866710956881841, "passes": true}, "pre.women.b1930_1934.pr_cross": {"truth": 0.5714979296198577, "filled": 0.6067825253454213, "gap": 0.03528459572556364, "seed_sd": 0.008863825559077805, "tolerance": 0.10849315671129696, "passes": true}, "pre.women.b1930_1934.pr_in": {"truth": 0.5486860492872828, "filled": 0.5817695222067873, "gap": 0.03308347291950453, "seed_sd": 0.013200978751289429, "tolerance": 0.1270201928993476, "passes": true}, "pre.women.b1930_1934.pzero": {"truth": 0.5514059876498446, "filled": 0.5527799005963105, "gap": 0.002488554998091752, "seed_sd": 0.0013713513850248907, "tolerance": 0.04703504786975244, "passes": true}, "pre.women.b1930_1945.aime_p10": {"truth": 79.0, "filled": 78.0, "gap": -0.012739025777429802, "seed_sd": 0.7254762501100116, "tolerance": 0.16621909484652816, "passes": true}, "pre.women.b1930_1945.aime_p25": {"truth": 280.0, "filled": 279.0, "gap": -0.0035778213478838694, "seed_sd": 1.3377121081198773, "tolerance": 0.09520232121757725, "passes": true}, "pre.women.b1930_1945.aime_p50": {"truth": 771.0, "filled": 761.95, "gap": -0.01180743682745966, "seed_sd": 1.669383750149485, "tolerance": 0.06218129686271377, "passes": true}, "pre.women.b1930_1945.aime_p75": {"truth": 1574.0, "filled": 1574.6, "gap": 0.0003811217730183003, "seed_sd": 2.891002375793231, "tolerance": 0.0540210895745143, "passes": true}, "pre.women.b1930_1945.aime_p90": {"truth": 2560.0, "filled": 2559.55, "gap": -0.00017579670133471836, "seed_sd": 3.677456214630749, "tolerance": 0.04855891958559449, "passes": true}, "pre.women.b1930_1945.plevel": {"truth": 0.18433938803699407, "filled": 0.17995218343257352, "gap": -0.024087390805605846, "seed_sd": 0.0007199009581150091, "tolerance": 0.050000040503342155, "passes": true}, "pre.women.b1930_1945.pr_cross": {"truth": 0.4395941263612976, "filled": 0.47329830691293084, "gap": 0.03370418055163327, "seed_sd": 0.006410383721397189, "tolerance": 0.0677006808491353, "passes": true}, "pre.women.b1930_1945.pr_in": {"truth": 0.372715069734525, "filled": 0.40355800362085176, "gap": 0.03084293388632675, "seed_sd": 0.005884207173703561, "tolerance": 0.07412154624977427, "passes": true}, "pre.women.b1930_1945.pzero": {"truth": 0.5120706676834471, "filled": 0.517344820360381, "gap": 0.01024697779893513, "seed_sd": 0.0010493346222991984, "tolerance": 0.02913422225552515, "passes": true}, "pre.women.b1935_1939.aime_p10": {"truth": 75.0, "filled": 73.26000000000002, "gap": -0.023473356185641947, "seed_sd": 1.8590886278808187, "tolerance": 0.3512121935176347, "passes": true}, "pre.women.b1935_1939.aime_p25": {"truth": 267.0, "filled": 267.725, "gap": 0.002711675886689413, "seed_sd": 3.2372218433647535, "tolerance": 0.180762984617868, "passes": true}, "pre.women.b1935_1939.aime_p50": {"truth": 726.0, "filled": 716.075, "gap": -0.01376510474655035, "seed_sd": 3.5919317482672923, "tolerance": 0.11655697880988479, "passes": true}, "pre.women.b1935_1939.aime_p75": {"truth": 1448.0, "filled": 1437.2875, "gap": -0.007425637288465126, "seed_sd": 3.934726466785341, "tolerance": 0.08830721415636139, "passes": true}, "pre.women.b1935_1939.aime_p90": {"truth": 2293.8999999999996, "filled": 2280.29, "gap": -0.005950797917480877, "seed_sd": 5.162914313692055, "tolerance": 0.08570712841284106, "passes": true}, "pre.women.b1935_1939.plevel": {"truth": 0.18459645468504016, "filled": 0.17846485916570784, "gap": -0.03378040207475497, "seed_sd": 0.0010822452970810686, "tolerance": 0.09407530912273358, "passes": true}, "pre.women.b1935_1939.pr_cross": {"truth": 0.5177687037806673, "filled": 0.5511541429766346, "gap": 0.03338543919596726, "seed_sd": 0.010162657342750367, "tolerance": 0.1216792156910919, "passes": true}, "pre.women.b1935_1939.pr_in": {"truth": 0.48588537175161917, "filled": 0.5109930990815804, "gap": 0.02510772732996125, "seed_sd": 0.011878561501692732, "tolerance": 0.13268326765003252, "passes": true}, "pre.women.b1935_1939.pzero": {"truth": 0.5152445437812253, "filled": 0.5244790297133843, "gap": 0.017763815300941732, "seed_sd": 0.0019054062085465567, "tolerance": 0.0510656392812699, "passes": true}, "pre.women.b1940_1945.aime_p10": {"truth": 110.0, "filled": 107.55, "gap": -0.02252451007867773, "seed_sd": 1.5719582155957412, "tolerance": 0.26939110963753704, "passes": true}, "pre.women.b1940_1945.aime_p25": {"truth": 386.0, "filled": 386.2125, "gap": 0.0005503666551991415, "seed_sd": 1.9504975748443485, "tolerance": 0.14001979176307694, "passes": true}, "pre.women.b1940_1945.aime_p50": {"truth": 1041.0, "filled": 1037.5, "gap": -0.003367816510116306, "seed_sd": 2.417044737516849, "tolerance": 0.08949151188066314, "passes": true}, "pre.women.b1940_1945.aime_p75": {"truth": 2058.0, "filled": 2049.5125, "gap": -0.004132677419653064, "seed_sd": 3.882717272871233, "tolerance": 0.06627004271644896, "passes": true}, "pre.women.b1940_1945.aime_p90": {"truth": 3179.7000000000007, "filled": 3176.3300000000004, "gap": -0.0010604104498526112, "seed_sd": 3.293869904438844, "tolerance": 0.06421030653828377, "passes": true}, "pre.women.b1940_1945.plevel": {"truth": 0.19012308454067162, "filled": 0.1855864479487595, "gap": -0.024150875623204504, "seed_sd": 0.0008659010649369567, "tolerance": 0.06996499056922599, "passes": true}, "pre.women.b1940_1945.pr_cross": {"truth": 0.3145537439002571, "filled": 0.34696255632716466, "gap": 0.032408812426907574, "seed_sd": 0.008928812485620915, "tolerance": 0.10514700123361043, "passes": true}, "pre.women.b1940_1945.pr_in": {"truth": 0.19634726483554435, "filled": 0.23404289705626513, "gap": 0.03769563222072078, "seed_sd": 0.007665655829315212, "tolerance": 0.10970200704526734, "passes": true}, "pre.women.b1940_1945.pzero": {"truth": 0.45477910425062734, "filled": 0.46078891339061673, "gap": 0.013128233708663006, "seed_sd": 0.0015635862218243877, "tolerance": 0.04701699812231318, "passes": true}, "pre.women.b1946_1955.paime_p10": {"truth": 156.0, "filled": 155.08499999999998, "gap": -0.005882653542770733, "seed_sd": 0.7727428931716918, "tolerance": 0.1092839592303832, "passes": true}, "pre.women.b1946_1955.paime_p25": {"truth": 516.0, "filled": 515.0875, "gap": -0.0017699763370702115, "seed_sd": 1.2441309585832447, "tolerance": 0.06230795442282464, "passes": true}, "pre.women.b1946_1955.paime_p50": {"truth": 1311.0, "filled": 1311.425, "gap": 0.00032412748026811045, "seed_sd": 1.2276957449243124, "tolerance": 0.0404504662154056, "passes": true}, "pre.women.b1946_1955.paime_p75": {"truth": 2484.0, "filled": 2480.5, "gap": -0.001410011312265702, "seed_sd": 1.4001879573076872, "tolerance": 0.0286716465849327, "passes": true}, "pre.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3776.390000000001, "gap": -0.0009554827833504476, "seed_sd": 1.7873532328411794, "tolerance": 0.03102321311868942, "passes": true}, "pre.women.b1946_1955.ylevel": {"truth": 0.09059452101103667, "filled": 0.09040111213730556, "gap": -0.002137167002129292, "seed_sd": 0.00024301583562582729, "tolerance": 0.029569090149685853, "passes": true}, "pre.women.b1946_1955.yr_cross": {"truth": 0.34158903474392094, "filled": 0.35976623713880124, "gap": 0.0181772023948803, "seed_sd": 0.005037967725535398, "tolerance": 0.03669253843323441, "passes": true}, "pre.women.b1946_1955.yzero": {"truth": 0.5347643649882455, "filled": 0.5381167650757204, "gap": 0.006249361469893522, "seed_sd": 0.000702873211849729, "tolerance": 0.015053483015288157, "passes": true}, "pre.women.b1946_1980.paime_p10": {"truth": 103.0, "filled": 102.9, "gap": -0.0009713453896322832, "seed_sd": 0.30779350562554625, "tolerance": 0.05217708504283977, "passes": true}, "pre.women.b1946_1980.paime_p25": {"truth": 304.0, "filled": 303.35, "gap": -0.002140447017914937, "seed_sd": 0.4893604849295929, "tolerance": 0.031237246580028546, "passes": true}, "pre.women.b1946_1980.paime_p50": {"truth": 752.0, "filled": 752.55, "gap": 0.0007311156485316772, "seed_sd": 0.6048053188292994, "tolerance": 0.021471785135136603, "passes": true}, "pre.women.b1946_1980.paime_p75": {"truth": 1583.75, "filled": 1580.6375, "gap": -0.0019672059782536166, "seed_sd": 0.871609974339682, "tolerance": 0.021247962622687265, "passes": true}, "pre.women.b1946_1980.paime_p90": {"truth": 2690.0, "filled": 2690.230000000002, "gap": 8.549820366088312e-05, "seed_sd": 1.5444637972601931, "tolerance": 0.019286406429246394, "passes": true}, "pre.women.b1946_1980.ylevel": {"truth": 0.06340849534928693, "filled": 0.06327556972848844, "gap": -0.0020985381152560656, "seed_sd": 9.26175686774783e-05, "tolerance": 0.015450091379218515, "passes": true}, "pre.women.b1946_1980.yr_cross": {"truth": 0.43636860215681333, "filled": 0.4451774460795388, "gap": 0.008808843922725462, "seed_sd": 0.0018089186135343722, "tolerance": 0.015684351176846988, "passes": true}, "pre.women.b1946_1980.yzero": {"truth": 0.4719000529035934, "filled": 0.47480921762755485, "gap": 0.0061458654129843415, "seed_sd": 0.00043829093555835673, "tolerance": 0.008556191699814752, "passes": true}, "pre.women.b1956_1965.paime_p10": {"truth": 136.0, "filled": 134.12000000000003, "gap": -0.013919964138024099, "seed_sd": 0.6436736669102601, "tolerance": 0.10309020846799896, "passes": true}, "pre.women.b1956_1965.paime_p25": {"truth": 422.0, "filled": 419.85, "gap": -0.005107809406439401, "seed_sd": 1.0894228312566054, "tolerance": 0.05400647590482275, "passes": true}, "pre.women.b1956_1965.paime_p50": {"truth": 996.0, "filled": 996.6, "gap": 0.0006022282627062836, "seed_sd": 0.7539370349250519, "tolerance": 0.03402127173688933, "passes": true}, "pre.women.b1956_1965.paime_p75": {"truth": 1839.0, "filled": 1838.425, "gap": -0.0003127188207425746, "seed_sd": 1.2594380534692113, "tolerance": 0.026722598015179132, "passes": true}, "pre.women.b1956_1965.paime_p90": {"truth": 2790.5999999999985, "filled": 2790.0499999999997, "gap": -0.0001971096563231356, "seed_sd": 2.580799549710988, "tolerance": 0.026506809111477153, "passes": true}, "pre.women.b1956_1965.ylevel": {"truth": 0.06347343380872424, "filled": 0.06311257468924407, "gap": -0.005701421527938955, "seed_sd": 0.00015504667222653533, "tolerance": 0.026809792541125404, "passes": true}, "pre.women.b1956_1965.yr_cross": {"truth": 0.3538451043271927, "filled": 0.3758782599407146, "gap": 0.022033155613521926, "seed_sd": 0.00382221620303007, "tolerance": 0.03047360671843986, "passes": true}, "pre.women.b1956_1965.yzero": {"truth": 0.47511142887567126, "filled": 0.47814290277925675, "gap": 0.006360283971058256, "seed_sd": 0.0008973301455629317, "tolerance": 0.017102901274216892, "passes": true}, "pre.women.b1966_1980.paime_p10": {"truth": 73.0, "filled": 74.35, "gap": 0.018324231757772758, "seed_sd": 0.4893604849295929, "tolerance": 0.0627268236453548, "passes": true}, "pre.women.b1966_1980.paime_p25": {"truth": 205.0, "filled": 204.3, "gap": -0.003420477314832304, "seed_sd": 0.5712405705774795, "tolerance": 0.039676687855380324, "passes": true}, "pre.women.b1966_1980.paime_p50": {"truth": 458.0, "filled": 456.9, "gap": -0.002404635544959177, "seed_sd": 0.6407232755171874, "tolerance": 0.02838346665574009, "passes": true}, "pre.women.b1966_1980.paime_p75": {"truth": 861.0, "filled": 862.275, "gap": 0.0014797408801827672, "seed_sd": 1.0447235846964749, "tolerance": 0.023922104075604994, "passes": true}, "pre.women.b1966_1980.paime_p90": {"truth": 1382.5999999999985, "filled": 1381.23, "gap": -0.0009913779879404672, "seed_sd": 1.0488590292113351, "tolerance": 0.02489092426663599, "passes": true}, "pre.women.b1966_1980.ylevel": {"truth": 0.04398552907354234, "filled": 0.04407810496564042, "gap": 0.002102477991286822, "seed_sd": 8.952444715162168e-05, "tolerance": 0.017769920520911725, "passes": true}, "pre.women.b1966_1980.yr_cross": {"truth": 0.318767197977468, "filled": 0.3245024764728558, "gap": 0.005735278495387797, "seed_sd": 0.003385430928445152, "tolerance": 0.02158530789673962, "passes": true}, "pre.women.b1966_1980.yzero": {"truth": 0.42453709903219916, "filled": 0.427032564367886, "gap": 0.005860876886955135, "seed_sd": 0.0006091590835569396, "tolerance": 0.014833297517636257, "passes": true}}, "tier": "certified"}, "adopted": "primary"}}} \ No newline at end of file diff --git a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl index 7ffa31bd..c9cc6fbe 100644 --- a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl +++ b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl @@ -15,3 +15,4 @@ {"utc": "2026-10-04T08:27:20+00:00", "candidate": "odd_knn2", "family": "odd", "artifact_sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 47, "n_gating": 183, "tier": "not_adopted", "worst": [["odd.men.a22_74.wint", -0.7579019412918284, 0.07337302886151653], ["odd.women.a22_74.wint", -0.6178668454511387, 0.07133033012317215], ["odd.men.a45_59.wint", -1.0410721850282765, 0.13344110142481172], ["odd.women.a45_59.wint", -0.9735852305534904, 0.1271041647107106], ["odd.men.a30_44.wint", -0.7720545590267562, 0.12001713872675526], ["odd.women.a30_44.wint", -0.5937411523100029, 0.10093703169083498]]} {"utc": "2026-10-04T08:27:45+00:00", "candidate": "pre_chain3", "family": "pre", "artifact_sha256": "74c62142e1de4bfed543de3b1d8c88c0d70c674a8a2fa44129bfe5943d4ff0ea", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 50, "n_gating": 136, "tier": "not_adopted", "worst": [["pre.women.b1966_1980.ylevel", 0.35125489825189504, 0.017769920520911725], ["pre.men.b1966_1980.ylevel", 0.35211188653405623, 0.021382204828984532], ["pre.women.b1966_1980.yzero", 0.13965247274046733, 0.014833297517636257], ["pre.women.b1946_1980.ylevel", 0.1097696357515745, 0.015450091379218515], ["pre.men.b1930_1945.plevel", -0.18375090914832892, 0.029173903026433003], ["pre.women.b1930_1945.pzero", -0.18200182661313102, 0.02913422225552515]]} {"utc": "2026-10-04T08:52:00+00:00", "candidate": "registered set (manifest runs/epuf_fill_candidates_v1.json)", "family": "odd+pre", "description": "DEV dry run of the registered TEST procedure: epuf_fill_scoring.score_registered with the DEV matrix, 20 draw seeds, candidates loaded by SHA-256, both readings of the current odd rule", "manifest_sha256": "8d421536351d884184af441889333db2361fb0f50326e3a4481e6fa4f07aeecf", "summary": {"odd": {"adopted": "primary", "dropped": [], "primary": {"n_failing": 7, "n_gating": 183, "tier": "improves", "worst": [[2.690259040347596, "odd.men.a22_29.r1"], [2.0188009070760624, "odd.women.a22_29.r1"], [1.5849202031321585, "odd.women.a22_29.zint"], [1.3654563419043049, "odd.men.a22_29.zint"], [1.2322030452483468, "odd.women.a22_74.zint"]]}, "alternative": {"n_failing": 46, "n_gating": 183, "tier": "not_adopted", "worst": [[10.235181548470665, "odd.men.a22_74.wint"], [8.632139516178636, "odd.women.a22_74.wint"], [7.7476496021426895, "odd.women.a45_59.wint"], [7.571934298028397, "odd.men.a45_59.wint"], [6.518179638279406, "odd.men.a30_44.wint"]]}, "current": {"fallback": {"n_failing": 100}, "two_sided": {"n_failing": 99}}}, "pre": {"adopted": "primary", "dropped": [], "primary": {"n_failing": 0, "n_gating": 136, "tier": "certified", "worst": [[0.769263808372738, "pre.men.b1946_1980.yzero"], [0.7230242162372418, "pre.women.b1956_1965.yr_cross"], [0.7182944969684824, "pre.women.b1946_1980.yzero"], [0.6574070292194412, "pre.men.b1956_1965.yzero"], [0.6068639745544633, "pre.men.b1946_1980.ylevel"]]}, "alternative": {"n_failing": 50, "n_gating": 136, "tier": "not_adopted", "worst": [[19.56156909523513, "pre.women.b1966_1980.ylevel"], [16.437839846619262, "pre.men.b1966_1980.ylevel"], [9.486645296938711, "pre.women.b1966_1980.yzero"], [6.976223346512972, "pre.women.b1946_1980.ylevel"], [6.268232573425137, "pre.men.b1930_1945.plevel"]]}, "current": {"fallback": {"n_failing": 131}}}}} +{"utc": "2026-10-04T19:10:00+00:00", "candidate": "none: K sweep of the registered dry run", "family": "odd+pre", "description": "DEV-only, post hoc sweep of K (tolerance multiplier) over the registered TEST-procedure dry run on DEV, re-scored with the repository's own score/adoption_tier/adopt/combined_current; shown to Max before he ruled on d927 (ratify K = 1). K = 1 was registered at 14045be4, before any DEV score. Script, input and output: docs/amendments/gate_epuf_fill_dev_k_sweep.py, gate_epuf_fill_dev_registered_dryrun.json, gate_epuf_fill_dev_k_sweep.txt.", "dryrun_sha256": "086378aa9c6cd4277edfcd88fbff4f1b3907d6619d82f554d64520db88ef49a6", "summary": {"k1_reproduces_record": true, "odd": {"primary_not_adopted_below_K": 0.9, "primary_certified_from_K": 2.6903, "adopted_primary_for_K": [0.9, 4.0]}, "pre": {"primary_certified_from_K": 0.77, "primary_improves_from_K": 0.26, "adopted_primary_for_K": [0.26, 4.0]}, "failing_at_K": {"odd_primary": {"1.0": 7, "1.5": 3, "2.0": 2, "3.0": 0}, "pre_primary": {"1.0": 0}}}} From 80eacd419a3e5dfaf748c419cbd8e574d7b362ab Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Sun, 4 Oct 2026 15:15:09 -0400 Subject: [PATCH 13/13] gate_epuf_fill: exact K-sweep thresholds in the DEV log Review of 06349367 (APPROVE, non-blocking): the log line gave grid-rounded thresholds as if exact. It now carries the exact breakpoints (odd 0.8968 and 2.6903, odd alternative 3.4117; pre 0.2564 and 0.7693), each a cell's |gap|/sigma or |gap|/(3 sigma), and the paragraph says the sweep used a 0.005 grid with every breakpoint checked exactly. Co-Authored-By: Claude Opus 5.5 --- .../gate_epuf_fill_candidates_registration.md | 11 ++++++----- .../gate_epuf_fill_dev_scores_after_round_2.jsonl | 2 +- 2 files changed, 7 insertions(+), 6 deletions(-) diff --git a/docs/amendments/gate_epuf_fill_candidates_registration.md b/docs/amendments/gate_epuf_fill_candidates_registration.md index fd41fad8..03ff4652 100644 --- a/docs/amendments/gate_epuf_fill_candidates_registration.md +++ b/docs/amendments/gate_epuf_fill_candidates_registration.md @@ -120,11 +120,12 @@ zero shares over ages 15-21 (`ylevel`, `yzero`) and one pre-career level (`plevel`); the log records only the five worst cells. **A sweep of K, shown to the ratifier.** Before Max ruled on d927 (ratify -`K = 1`), the dry run below was re-scored at every `K` from 0.25 to 4 with -the repository's own scoring and adoption functions. `K = 1` reproduces the -record exactly. The same primaries are adopted for every `K` from about 0.9 -to 4. The odd primary is certified from `K = 2.69` and not adopted below -about 0.9. The pre primary is certified from `K = 0.77`. `K = 1` was +`K = 1`), the dry run below was re-scored on a 0.005 grid of `K` from 0.25 +to 4 with the repository's own scoring and adoption functions, and every +breakpoint was checked exactly. `K = 1` reproduces the record exactly. The +same primaries are adopted for every `K` from 0.8968 to 4. The odd primary +is certified from `K = 2.6903` and not adopted below 0.8968. The pre +primary is certified from `K = 0.7693`. `K = 1` was registered at `14045be4`, before any DEV score; the sweep is disclosed so the ratification is read with it in view. Script, input and output: `gate_epuf_fill_dev_k_sweep.py`, `gate_epuf_fill_dev_registered_dryrun.json` diff --git a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl index c9cc6fbe..af53bb85 100644 --- a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl +++ b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl @@ -15,4 +15,4 @@ {"utc": "2026-10-04T08:27:20+00:00", "candidate": "odd_knn2", "family": "odd", "artifact_sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 47, "n_gating": 183, "tier": "not_adopted", "worst": [["odd.men.a22_74.wint", -0.7579019412918284, 0.07337302886151653], ["odd.women.a22_74.wint", -0.6178668454511387, 0.07133033012317215], ["odd.men.a45_59.wint", -1.0410721850282765, 0.13344110142481172], ["odd.women.a45_59.wint", -0.9735852305534904, 0.1271041647107106], ["odd.men.a30_44.wint", -0.7720545590267562, 0.12001713872675526], ["odd.women.a30_44.wint", -0.5937411523100029, 0.10093703169083498]]} {"utc": "2026-10-04T08:27:45+00:00", "candidate": "pre_chain3", "family": "pre", "artifact_sha256": "74c62142e1de4bfed543de3b1d8c88c0d70c674a8a2fa44129bfe5943d4ff0ea", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 50, "n_gating": 136, "tier": "not_adopted", "worst": [["pre.women.b1966_1980.ylevel", 0.35125489825189504, 0.017769920520911725], ["pre.men.b1966_1980.ylevel", 0.35211188653405623, 0.021382204828984532], ["pre.women.b1966_1980.yzero", 0.13965247274046733, 0.014833297517636257], ["pre.women.b1946_1980.ylevel", 0.1097696357515745, 0.015450091379218515], ["pre.men.b1930_1945.plevel", -0.18375090914832892, 0.029173903026433003], ["pre.women.b1930_1945.pzero", -0.18200182661313102, 0.02913422225552515]]} {"utc": "2026-10-04T08:52:00+00:00", "candidate": "registered set (manifest runs/epuf_fill_candidates_v1.json)", "family": "odd+pre", "description": "DEV dry run of the registered TEST procedure: epuf_fill_scoring.score_registered with the DEV matrix, 20 draw seeds, candidates loaded by SHA-256, both readings of the current odd rule", "manifest_sha256": "8d421536351d884184af441889333db2361fb0f50326e3a4481e6fa4f07aeecf", "summary": {"odd": {"adopted": "primary", "dropped": [], "primary": {"n_failing": 7, "n_gating": 183, "tier": "improves", "worst": [[2.690259040347596, "odd.men.a22_29.r1"], [2.0188009070760624, "odd.women.a22_29.r1"], [1.5849202031321585, "odd.women.a22_29.zint"], [1.3654563419043049, "odd.men.a22_29.zint"], [1.2322030452483468, "odd.women.a22_74.zint"]]}, "alternative": {"n_failing": 46, "n_gating": 183, "tier": "not_adopted", "worst": [[10.235181548470665, "odd.men.a22_74.wint"], [8.632139516178636, "odd.women.a22_74.wint"], [7.7476496021426895, "odd.women.a45_59.wint"], [7.571934298028397, "odd.men.a45_59.wint"], [6.518179638279406, "odd.men.a30_44.wint"]]}, "current": {"fallback": {"n_failing": 100}, "two_sided": {"n_failing": 99}}}, "pre": {"adopted": "primary", "dropped": [], "primary": {"n_failing": 0, "n_gating": 136, "tier": "certified", "worst": [[0.769263808372738, "pre.men.b1946_1980.yzero"], [0.7230242162372418, "pre.women.b1956_1965.yr_cross"], [0.7182944969684824, "pre.women.b1946_1980.yzero"], [0.6574070292194412, "pre.men.b1956_1965.yzero"], [0.6068639745544633, "pre.men.b1946_1980.ylevel"]]}, "alternative": {"n_failing": 50, "n_gating": 136, "tier": "not_adopted", "worst": [[19.56156909523513, "pre.women.b1966_1980.ylevel"], [16.437839846619262, "pre.men.b1966_1980.ylevel"], [9.486645296938711, "pre.women.b1966_1980.yzero"], [6.976223346512972, "pre.women.b1946_1980.ylevel"], [6.268232573425137, "pre.men.b1930_1945.plevel"]]}, "current": {"fallback": {"n_failing": 131}}}}} -{"utc": "2026-10-04T19:10:00+00:00", "candidate": "none: K sweep of the registered dry run", "family": "odd+pre", "description": "DEV-only, post hoc sweep of K (tolerance multiplier) over the registered TEST-procedure dry run on DEV, re-scored with the repository's own score/adoption_tier/adopt/combined_current; shown to Max before he ruled on d927 (ratify K = 1). K = 1 was registered at 14045be4, before any DEV score. Script, input and output: docs/amendments/gate_epuf_fill_dev_k_sweep.py, gate_epuf_fill_dev_registered_dryrun.json, gate_epuf_fill_dev_k_sweep.txt.", "dryrun_sha256": "086378aa9c6cd4277edfcd88fbff4f1b3907d6619d82f554d64520db88ef49a6", "summary": {"k1_reproduces_record": true, "odd": {"primary_not_adopted_below_K": 0.9, "primary_certified_from_K": 2.6903, "adopted_primary_for_K": [0.9, 4.0]}, "pre": {"primary_certified_from_K": 0.77, "primary_improves_from_K": 0.26, "adopted_primary_for_K": [0.26, 4.0]}, "failing_at_K": {"odd_primary": {"1.0": 7, "1.5": 3, "2.0": 2, "3.0": 0}, "pre_primary": {"1.0": 0}}}} +{"utc": "2026-10-04T19:10:00+00:00", "candidate": "none: K sweep of the registered dry run", "family": "odd+pre", "description": "DEV-only, post hoc sweep of K (tolerance multiplier) over the registered TEST-procedure dry run on DEV, re-scored with the repository's own score/adoption_tier/adopt/combined_current; shown to Max before he ruled on d927 (ratify K = 1). K = 1 was registered at 14045be4, before any DEV score. Script, input and output: docs/amendments/gate_epuf_fill_dev_k_sweep.py, gate_epuf_fill_dev_registered_dryrun.json, gate_epuf_fill_dev_k_sweep.txt.", "dryrun_sha256": "086378aa9c6cd4277edfcd88fbff4f1b3907d6619d82f554d64520db88ef49a6", "summary": {"k1_reproduces_record": true, "odd": {"primary_improves_from_K": 0.8968, "primary_certified_from_K": 2.6903, "alternative_improves_from_K": 3.4117, "adopted_primary_for_K": [0.8968, 4.0]}, "pre": {"primary_improves_from_K": 0.2564, "primary_certified_from_K": 0.7693, "adopted_primary_for_K": [0.2564, 4.0]}, "failing_at_K": {"odd_primary": {"1.0": 7, "1.5": 3, "2.0": 2, "3.0": 0}, "pre_primary": {"1.0": 0}}, "thresholds": "exact: each is a cell's |gap|/sigma or |gap|/(3 sigma); the script's 0.005 grid finds no other transition"}}