diff --git a/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_36e045bdc946.py b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_36e045bdc946.py new file mode 100644 index 00000000..bece8a62 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_36e045bdc946.py @@ -0,0 +1,1942 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddQuantileFill` (odd years, primary): a two-part conditional + draw. The probability of a zero year and the conditional quantiles of a + positive share, relative to the neighbours' level, by sex, age, the + shares at ``t-1`` and ``t+1`` and the context at ``t-3``, ``t+3`` and + further; a Gaussian AR(1) copula correlates a person's draws across + masked years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "OddQuantileFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _invert(table: np.ndarray, rows: np.ndarray, value: np.ndarray): + """The level at which each row's quantile function reaches ``value``. + + Where the function is flat at ``value`` (a run of equal quantiles), the + middle of the run's levels. + """ + + points = table.shape[1] + levels = np.linspace(0.0, 1.0, points) + out = np.empty(len(rows)) + for start in range(0, len(rows), 200_000): + block = slice(start, start + 200_000) + curve = table[rows[block]].astype(np.float64) + target = np.asarray(value[block], dtype=np.float64)[:, None] + below = (curve < target).sum(axis=1) + above = (curve <= target).sum(axis=1) + flat = below < above + result = np.empty(len(curve)) + result[above == 0] = 0.0 + result[below >= points] = 1.0 + middle = flat & (above > 0) & (below < points) + result[middle] = 0.5 * ( + levels[below[middle]] + levels[above[middle] - 1] + ) + between = ~flat & (below > 0) & (below < points) + index = np.flatnonzero(between) + left = curve[index, below[index] - 1] + right = curve[index, below[index]] + share = np.where( + right > left, + (target[index, 0] - left) + / np.where(right > left, right - left, 1), + 0.5, + ) + result[index] = levels[below[index] - 1] + share * ( + levels[below[index]] - levels[below[index] - 1] + ) + out[block] = result + return out + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + buffer = io.BytesIO() + np.savez_compressed(buffer, **arrays) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: the two-part conditional draw +# -------------------------------------------------------------------------- +_N_SHARE_BINS = 20 + + +def _share_bin(value: np.ndarray, edges: np.ndarray) -> np.ndarray: + """0 zero, 1..20 quantile bins of a positive share below the cap, 21 cap.""" + + bins = np.searchsorted(edges, value, side="right") + 1 + bins = np.where(value <= 0, 0, bins) + return np.where(value >= 1.0, _N_SHARE_BINS + 1, bins) + + +def _coarse_age(band: np.ndarray) -> np.ndarray: + """Age bands grouped: under 30, 30-44, 45-59, 60 and over.""" + + return np.digitize(band, [4, 7, 10]) + + +@dataclass(frozen=True) +class OddQuantileFill: + """Two-part conditional draw for masked odd years, with an AR(1) copula. + + A unit's reference level ``m`` is the geometric mean of its positive + neighbours' shares (the one positive neighbour's share if only one is, + 1 if neither is). Its cell is the finest of seven nested keys with at + least ``MIN_CELL`` TRAIN units, built from sex, five-year age band, the + bins of the shares at ``t-1`` and ``t+1`` (zero, 20 quantile bins of a + positive share below the cap, at the cap), the context at ``t-3`` and + ``t+3`` (missing, zero, below or above the median positive share), and + whether any share at offsets 5, 7 or 9 is positive. In the cell: ``p0`` + the share of zero years, and 65 quantiles of ``log(x_t / m)`` among + positive years. A uniform ``u`` maps to zero if ``u < p0``, else to + ``min(m * exp(Q((u - p0) / (1 - p0))), 1)``. The uniforms of a person's + consecutive masked years (two years apart) are joined by a Gaussian + AR(1) copula with correlation ``rho`` by sex and age band, learned on + TRAIN from the probability integral transforms of consecutive units. + """ + + share_edges: np.ndarray + context_median: float + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + rho: np.ndarray + stream: str = "epuf_fill.odd_quantile.v1" + name: str = "odd_quantile" + + # -- keys --------------------------------------------------------------- + @staticmethod + def _parts(context: OddContext, share_edges, median): + left = np.nan_to_num(context.left, nan=-1.0) + right = np.nan_to_num(context.right, nan=-1.0) + # A missing neighbour takes the other's value (the PSID fallback). + left = np.where(left < 0, right, left) + right = np.where(right < 0, left, right) + positive_left = np.where(left > 0, left, 1.0) + positive_right = np.where(right > 0, right, 1.0) + level = np.where( + (left > 0) & (right > 0), + np.sqrt(positive_left * positive_right), + np.where(left > 0, positive_left, positive_right), + ) + + def context_code(value): + return np.where( + np.isnan(value), + 0, + np.where(value <= 0, 1, np.where(value < median, 2, 3)), + ) + + return { + "sex": context.sex, + "age": _age_band(context.age), + "left_bin": _share_bin(left, share_edges), + "right_bin": _share_bin(right, share_edges), + "context3": 4 * context_code(context.left3) + + context_code(context.right3), + "wide": context.wide, + "level": level, + "valid": ~(np.isnan(context.left) & np.isnan(context.right)), + } + + @staticmethod + def _keys(parts) -> list[np.ndarray]: + sex = parts["sex"] + age = parts["age"] + coarse = _coarse_age(age) + left = parts["left_bin"] + right = parts["right_bin"] + context3 = parts["context3"] + wide = parts["wide"] + + # Nested keys from finest to coarsest; a dropped component is held + # at a sentinel (age 16-20 marks the coarse bands, 21 none). + def key(s, a, lb, rb, c3, w): + return ((((s * 22 + a) * 23 + lb) * 23 + rb) * 17 + c3) * 3 + w + + return [ + key(sex, age, left, right, context3, wide), + key(sex, age, left, right, context3, 2), + key(sex, age, left, right, 16, 2), + key(sex, 16 + coarse, left, right, 16, 2), + key(sex, 21, left, right, 16, 2), + key(0 * sex, 21, left, right, 16, 2), + key(0 * sex, 21, np.minimum(left, 1), np.minimum(right, 1), 16, 2), + ] + + def _cells(self, parts) -> tuple[np.ndarray, np.ndarray]: + """(level, row) of each unit's finest populated cell.""" + + keys = self._keys(parts) + level = np.full(len(keys[0]), -1, dtype=np.int64) + row = np.full(len(keys[0]), -1, dtype=np.int64) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + if (level < 0).any(): + raise ValueError("a unit has no populated cell at any level") + return level, row + + def _quantile(self, parts, u: np.ndarray) -> np.ndarray: + level, row = self._cells(parts) + out = np.zeros(len(u)) + for index in np.unique(level): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate(self.level_quantiles[index], row[take], v) + share = np.minimum(parts["level"][take] * np.exp(residual), 1.0) + out[take] = np.where(positive, share, 0.0) + return out + + # -- fitting -------------------------------------------------------------- + @classmethod + def fit( + cls, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + unit_years: tuple[int, ...], + rho_seed: int = 0, + ) -> tuple[OddQuantileFill, dict[str, object]]: + """Fit on complete TRAIN shares; every year of ``unit_years`` a unit. + + Every person-year of ``unit_years`` whose two neighbours are inside + the matrix is a training unit (all years are recorded on TRAIN). + """ + + shares = np.asarray(shares, dtype=np.float64) + known = np.isfinite(shares) + n = len(shares) + rows_list, years_list = [], [] + for year in unit_years: + rows_list.append(np.arange(n)) + years_list.append(np.full(n, year)) + rows = np.concatenate(rows_list) + unit_year = np.concatenate(years_list) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + target = _take(shares, rows, _column_of(years, unit_year)) + neighbours = np.concatenate([context.left, context.right]) + inside = neighbours[(neighbours > 0) & (neighbours < 1.0)] + share_edges = np.quantile( + inside, np.linspace(0, 1, _N_SHARE_BINS + 1)[1:-1] + ) + median = float(np.median(inside)) + parts = cls._parts(context, share_edges, median) + keys = cls._keys(parts) + positive = target > 0 + residual = np.log(np.where(positive, target, 1.0)) - np.log( + parts["level"] + ) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zeros_unique, zeros = np.unique( + key[in_cells & ~positive], return_counts=True + ) + totals = count[count >= MIN_CELL] + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zeros_unique)] = zeros + p0 = p0 / totals + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + # A populated cell with no positive unit draws only zeros. + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0.astype(np.float64)) + level_quantiles.append(full) + provisional = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=np.zeros((4, 16)), + ) + rho, rho_diagnostics = provisional._fit_rho( + parts, target, rows, unit_year, rho_seed + ) + fill = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=rho, + ) + cells = [len(k) for k in level_keys] + return fill, { + "n_units": int(len(target)), + "cells_per_level": cells, + **rho_diagnostics, + } + + def _pit(self, parts, target, rows, unit_year, seed) -> np.ndarray: + """Randomised probability integral transforms of true shares.""" + + level, row = self._cells(parts) + jitter = hash_uniform( + "epuf_fill.odd_quantile.pit", seed, rows, unit_year + ) + out = np.empty(len(target)) + for index in np.unique(level): + take = np.flatnonzero(level == index) + p0 = self.level_p0[index][row[take]] + zero = target[take] <= 0 + out[take[zero]] = jitter[take[zero]] * p0[zero] + positive = take[~zero] + residual = np.log(target[positive]) - np.log( + parts["level"][positive] + ) + # At the cap the residual is censored: spread it over the mass + # the quantile function puts at or above the cap. + at_cap = target[positive] >= 1.0 + v = _invert(self.level_quantiles[index], row[positive], residual) + cap_v = v.copy() + cap_v[at_cap] = v[at_cap] + jitter[positive][at_cap] * ( + 1.0 - v[at_cap] + ) + p0_positive = p0[~zero] + out[positive] = p0_positive + (1.0 - p0_positive) * cap_v + return np.clip(out, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, parts, target, rows, unit_year, seed): + """AR(1) correlation of consecutive units' normal scores (t, t+2). + + On a 5 percent sample of persons (by seed): every unit's + probability integral transform under the fitted cells, its normal + score, and the correlation of the scores of ``t`` and ``t+2`` for + the same person, by sex and age band at ``t``. + """ + + persons = np.unique(rows) + rng = np.random.default_rng(seed) + chosen = persons[rng.random(len(persons)) < 0.05] + index = np.flatnonzero(np.isin(rows, chosen)) + sub = {k: v[index] for k, v in parts.items()} + z = ndtri( + self._pit(sub, target[index], rows[index], unit_year[index], seed) + ) + first_year = int(unit_year.min()) + n_years = int(unit_year.max()) - first_year + 1 + position = np.searchsorted(chosen, rows[index]) + grid = np.full((len(chosen), n_years), np.nan) + grid[position, unit_year[index] - first_year] = z + sex_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + sex_grid[position, unit_year[index] - first_year] = sub["sex"] + age_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + age_grid[position, unit_year[index] - first_year] = sub["age"] + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex = sex_grid[:, :-2].ravel() + age = age_grid[:, :-2].ravel() + both = np.isfinite(now) & np.isfinite(later) + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = both & (sex == s) & (age == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + overall = float(np.corrcoef(now[both], later[both])[0, 1]) + four_now = grid[:, :-4].ravel() + four_later = grid[:, 4:].ravel() + four = np.isfinite(four_now) & np.isfinite(four_later) + lag4 = float(np.corrcoef(four_now[four], four_later[four])[0, 1]) + return rho, { + "rho_persons": int(len(chosen)), + "rho_pairs": int(both.sum()), + "rho_overall": overall, + "lag4_normal_score_correlation": lag4, + "lag4_ar1_prediction": overall**2, + } + + # -- filling -------------------------------------------------------------- + def fill( + self, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + person_key: np.ndarray, + fill_mask: np.ndarray, + seed: int, + ) -> np.ndarray: + """Fill the masked cells; masked cells with no known neighbour stay NaN.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + parts = self._parts(context, self.share_edges, self.context_median) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + rho = self.rho[np.clip(parts["sex"], 0, 3), parts["age"]] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + valid = parts["valid"] + drawn = np.full(len(rows), np.nan) + if valid.any(): + sub = {k: v[valid] for k, v in parts.items()} + drawn[valid] = self._quantile(sub, ndtr(z[valid])) + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + # -- persistence ---------------------------------------------------------- + def to_bytes(self) -> bytes: + arrays = { + "kind": np.array(self.name), + "share_edges": self.share_edges, + "context_median": np.array(self.context_median), + "rho": self.rho, + } + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> OddQuantileFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + share_edges=arrays["share_edges"], + context_median=float(arrays["context_median"]), + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + rho=arrays["rho"], + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: a quantile regression forest (QRF) draw +# -------------------------------------------------------------------------- + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +#: The reference level of a unit with no positive share around it. +_DEFAULT_LEVEL = 0.3 + + +def reference_level(context: OddContext) -> np.ndarray: + """The level a unit's share is drawn relative to. + + The geometric mean of the positive shares at ``t-1`` and ``t+1``; else + of those at ``t-3`` and ``t+3``; else the mean positive share at + offsets 5-9; else 0.3. + """ + + def geometric(a, b): + a = np.nan_to_num(a, nan=0.0) + b = np.nan_to_num(b, nan=0.0) + both = (a > 0) & (b > 0) + one = np.where(a > 0, a, b) + value = np.where(both, np.sqrt(np.where(both, a * b, 1.0)), one) + return np.where((a > 0) | (b > 0), value, np.nan) + + level = geometric(context.left, context.right) + level = np.where( + np.isnan(level), geometric(context.left3, context.right3), level + ) + level = np.where( + np.isnan(level) & (context.wide_mean > 0), context.wide_mean, level + ) + return np.where(np.isnan(level), _DEFAULT_LEVEL, level) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.61, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + A random forest (scikit-learn; split target ``log(share + 0.01)``) + partitions TRAIN units by :func:`odd_features`. Every TRAIN unit used + in the fit is passed down every tree, and each leaf keeps the sorted + true shares of the units that reach it (zeros and the cap included, + stored as shares times 65,535). The forest's conditional law of a unit + is the average over trees of its leaves' empirical laws, so the draw is + two-part by construction: zero, the cap and every share between keep + their own mass. A draw picks a tree by one seeded uniform and takes the + leaf's value at the quantile of a second, the copula uniform. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + stream: str = "epuf_fill.odd_forest.v3" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + known = np.isfinite(shares) + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x = odd_features(context) + y = target[chosen] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y + 0.01)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + drawn[valid] = self._value(unit["leaves"][valid], ndtr(z[valid])) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + mask &= years[None, :] >= start[:, None] + given = np.where(mask, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +_FOREST_ARRAYS = ( + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + known = np.isfinite(shares) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum): + members = np.flatnonzero(stratum == value) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=context.left[keep].astype(np.float32), + bank_right=context.right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:_DONOR_BANK] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + -1 for rows with no masked cell or no bank donor of their sex and + birth year. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + out = np.full(len(shares), -1, dtype=np.int64) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + recipients = targets[ + (sex[targets] == s) & (birth_year[targets] == b) + ] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + column = np.sort( + donor_match[:, j][np.isfinite(donor_match[:, j])] + ) + ranks_donor[:, j] = np.where( + np.isfinite(donor_match[:, j]), + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + nearest = np.argpartition(distance, k - 1, axis=1)[:, :k] + nearest_distance = np.take_along_axis(distance, nearest, 1) + order = np.lexsort((nearest, nearest_distance), axis=1) + nearest = np.take_along_axis(nearest, order, 1) + pick = np.minimum( + (u[recipients[block]] * k).astype(np.int64), k - 1 + ) + no_match = count.max(axis=1) == 0 + random_donor = np.minimum( + (u[recipients[block]] * len(donors)).astype(np.int64), + len(donors) - 1, + ) + out[recipients[block]] = donors[ + np.where( + no_match, + random_donor, + nearest[np.arange(len(nearest)), pick], + ) + ] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_quantile": OddQuantileFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_565e3ba2717b.py b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_565e3ba2717b.py new file mode 100644 index 00000000..fde0bfa1 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_565e3ba2717b.py @@ -0,0 +1,2066 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddQuantileFill` (odd years, primary): a two-part conditional + draw. The probability of a zero year and the conditional quantiles of a + positive share, relative to the neighbours' level, by sex, age, the + shares at ``t-1`` and ``t+1`` and the context at ``t-3``, ``t+3`` and + further; a Gaussian AR(1) copula correlates a person's draws across + masked years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "OddQuantileFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _invert(table: np.ndarray, rows: np.ndarray, value: np.ndarray): + """The level at which each row's quantile function reaches ``value``. + + Where the function is flat at ``value`` (a run of equal quantiles), the + middle of the run's levels. + """ + + points = table.shape[1] + levels = np.linspace(0.0, 1.0, points) + out = np.empty(len(rows)) + for start in range(0, len(rows), 200_000): + block = slice(start, start + 200_000) + curve = table[rows[block]].astype(np.float64) + target = np.asarray(value[block], dtype=np.float64)[:, None] + below = (curve < target).sum(axis=1) + above = (curve <= target).sum(axis=1) + flat = below < above + result = np.empty(len(curve)) + result[above == 0] = 0.0 + result[below >= points] = 1.0 + middle = flat & (above > 0) & (below < points) + result[middle] = 0.5 * ( + levels[below[middle]] + levels[above[middle] - 1] + ) + between = ~flat & (below > 0) & (below < points) + index = np.flatnonzero(between) + left = curve[index, below[index] - 1] + right = curve[index, below[index]] + share = np.where( + right > left, + (target[index, 0] - left) + / np.where(right > left, right - left, 1), + 0.5, + ) + result[index] = levels[below[index] - 1] + share * ( + levels[below[index]] - levels[below[index] - 1] + ) + out[block] = result + return out + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + buffer = io.BytesIO() + np.savez_compressed(buffer, **arrays) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: the two-part conditional draw +# -------------------------------------------------------------------------- +_N_SHARE_BINS = 20 + + +def _share_bin(value: np.ndarray, edges: np.ndarray) -> np.ndarray: + """0 zero, 1..20 quantile bins of a positive share below the cap, 21 cap.""" + + bins = np.searchsorted(edges, value, side="right") + 1 + bins = np.where(value <= 0, 0, bins) + return np.where(value >= 1.0, _N_SHARE_BINS + 1, bins) + + +def _coarse_age(band: np.ndarray) -> np.ndarray: + """Age bands grouped: under 30, 30-44, 45-59, 60 and over.""" + + return np.digitize(band, [4, 7, 10]) + + +@dataclass(frozen=True) +class OddQuantileFill: + """Two-part conditional draw for masked odd years, with an AR(1) copula. + + A unit's reference level ``m`` is the geometric mean of its positive + neighbours' shares (the one positive neighbour's share if only one is, + 1 if neither is). Its cell is the finest of seven nested keys with at + least ``MIN_CELL`` TRAIN units, built from sex, five-year age band, the + bins of the shares at ``t-1`` and ``t+1`` (zero, 20 quantile bins of a + positive share below the cap, at the cap), the context at ``t-3`` and + ``t+3`` (missing, zero, below or above the median positive share), and + whether any share at offsets 5, 7 or 9 is positive. In the cell: ``p0`` + the share of zero years, and 65 quantiles of ``log(x_t / m)`` among + positive years. A uniform ``u`` maps to zero if ``u < p0``, else to + ``min(m * exp(Q((u - p0) / (1 - p0))), 1)``. The uniforms of a person's + consecutive masked years (two years apart) are joined by a Gaussian + AR(1) copula with correlation ``rho`` by sex and age band, learned on + TRAIN from the probability integral transforms of consecutive units. + """ + + share_edges: np.ndarray + context_median: float + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + rho: np.ndarray + stream: str = "epuf_fill.odd_quantile.v1" + name: str = "odd_quantile" + + # -- keys --------------------------------------------------------------- + @staticmethod + def _parts(context: OddContext, share_edges, median): + left = np.nan_to_num(context.left, nan=-1.0) + right = np.nan_to_num(context.right, nan=-1.0) + # A missing neighbour takes the other's value (the PSID fallback). + left = np.where(left < 0, right, left) + right = np.where(right < 0, left, right) + positive_left = np.where(left > 0, left, 1.0) + positive_right = np.where(right > 0, right, 1.0) + level = np.where( + (left > 0) & (right > 0), + np.sqrt(positive_left * positive_right), + np.where(left > 0, positive_left, positive_right), + ) + + def context_code(value): + return np.where( + np.isnan(value), + 0, + np.where(value <= 0, 1, np.where(value < median, 2, 3)), + ) + + return { + "sex": context.sex, + "age": _age_band(context.age), + "left_bin": _share_bin(left, share_edges), + "right_bin": _share_bin(right, share_edges), + "context3": 4 * context_code(context.left3) + + context_code(context.right3), + "wide": context.wide, + "level": level, + "valid": ~(np.isnan(context.left) & np.isnan(context.right)), + } + + @staticmethod + def _keys(parts) -> list[np.ndarray]: + sex = parts["sex"] + age = parts["age"] + coarse = _coarse_age(age) + left = parts["left_bin"] + right = parts["right_bin"] + context3 = parts["context3"] + wide = parts["wide"] + + # Nested keys from finest to coarsest; a dropped component is held + # at a sentinel (age 16-20 marks the coarse bands, 21 none). + def key(s, a, lb, rb, c3, w): + return ((((s * 22 + a) * 23 + lb) * 23 + rb) * 17 + c3) * 3 + w + + return [ + key(sex, age, left, right, context3, wide), + key(sex, age, left, right, context3, 2), + key(sex, age, left, right, 16, 2), + key(sex, 16 + coarse, left, right, 16, 2), + key(sex, 21, left, right, 16, 2), + key(0 * sex, 21, left, right, 16, 2), + key(0 * sex, 21, np.minimum(left, 1), np.minimum(right, 1), 16, 2), + ] + + def _cells(self, parts) -> tuple[np.ndarray, np.ndarray]: + """(level, row) of each unit's finest populated cell.""" + + keys = self._keys(parts) + level = np.full(len(keys[0]), -1, dtype=np.int64) + row = np.full(len(keys[0]), -1, dtype=np.int64) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + if (level < 0).any(): + raise ValueError("a unit has no populated cell at any level") + return level, row + + def _quantile(self, parts, u: np.ndarray) -> np.ndarray: + level, row = self._cells(parts) + out = np.zeros(len(u)) + for index in np.unique(level): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate(self.level_quantiles[index], row[take], v) + share = np.minimum(parts["level"][take] * np.exp(residual), 1.0) + out[take] = np.where(positive, share, 0.0) + return out + + # -- fitting -------------------------------------------------------------- + @classmethod + def fit( + cls, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + unit_years: tuple[int, ...], + rho_seed: int = 0, + ) -> tuple[OddQuantileFill, dict[str, object]]: + """Fit on complete TRAIN shares; every year of ``unit_years`` a unit. + + Every person-year of ``unit_years`` whose two neighbours are inside + the matrix is a training unit (all years are recorded on TRAIN). + """ + + shares = np.asarray(shares, dtype=np.float64) + known = np.isfinite(shares) + n = len(shares) + rows_list, years_list = [], [] + for year in unit_years: + rows_list.append(np.arange(n)) + years_list.append(np.full(n, year)) + rows = np.concatenate(rows_list) + unit_year = np.concatenate(years_list) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + target = _take(shares, rows, _column_of(years, unit_year)) + neighbours = np.concatenate([context.left, context.right]) + inside = neighbours[(neighbours > 0) & (neighbours < 1.0)] + share_edges = np.quantile( + inside, np.linspace(0, 1, _N_SHARE_BINS + 1)[1:-1] + ) + median = float(np.median(inside)) + parts = cls._parts(context, share_edges, median) + keys = cls._keys(parts) + positive = target > 0 + residual = np.log(np.where(positive, target, 1.0)) - np.log( + parts["level"] + ) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zeros_unique, zeros = np.unique( + key[in_cells & ~positive], return_counts=True + ) + totals = count[count >= MIN_CELL] + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zeros_unique)] = zeros + p0 = p0 / totals + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + # A populated cell with no positive unit draws only zeros. + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0.astype(np.float64)) + level_quantiles.append(full) + provisional = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=np.zeros((4, 16)), + ) + rho, rho_diagnostics = provisional._fit_rho( + parts, target, rows, unit_year, rho_seed + ) + fill = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=rho, + ) + cells = [len(k) for k in level_keys] + return fill, { + "n_units": int(len(target)), + "cells_per_level": cells, + **rho_diagnostics, + } + + def _pit(self, parts, target, rows, unit_year, seed) -> np.ndarray: + """Randomised probability integral transforms of true shares.""" + + level, row = self._cells(parts) + jitter = hash_uniform( + "epuf_fill.odd_quantile.pit", seed, rows, unit_year + ) + out = np.empty(len(target)) + for index in np.unique(level): + take = np.flatnonzero(level == index) + p0 = self.level_p0[index][row[take]] + zero = target[take] <= 0 + out[take[zero]] = jitter[take[zero]] * p0[zero] + positive = take[~zero] + residual = np.log(target[positive]) - np.log( + parts["level"][positive] + ) + # At the cap the residual is censored: spread it over the mass + # the quantile function puts at or above the cap. + at_cap = target[positive] >= 1.0 + v = _invert(self.level_quantiles[index], row[positive], residual) + cap_v = v.copy() + cap_v[at_cap] = v[at_cap] + jitter[positive][at_cap] * ( + 1.0 - v[at_cap] + ) + p0_positive = p0[~zero] + out[positive] = p0_positive + (1.0 - p0_positive) * cap_v + return np.clip(out, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, parts, target, rows, unit_year, seed): + """AR(1) correlation of consecutive units' normal scores (t, t+2). + + On a 5 percent sample of persons (by seed): every unit's + probability integral transform under the fitted cells, its normal + score, and the correlation of the scores of ``t`` and ``t+2`` for + the same person, by sex and age band at ``t``. + """ + + persons = np.unique(rows) + rng = np.random.default_rng(seed) + chosen = persons[rng.random(len(persons)) < 0.05] + index = np.flatnonzero(np.isin(rows, chosen)) + sub = {k: v[index] for k, v in parts.items()} + z = ndtri( + self._pit(sub, target[index], rows[index], unit_year[index], seed) + ) + first_year = int(unit_year.min()) + n_years = int(unit_year.max()) - first_year + 1 + position = np.searchsorted(chosen, rows[index]) + grid = np.full((len(chosen), n_years), np.nan) + grid[position, unit_year[index] - first_year] = z + sex_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + sex_grid[position, unit_year[index] - first_year] = sub["sex"] + age_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + age_grid[position, unit_year[index] - first_year] = sub["age"] + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex = sex_grid[:, :-2].ravel() + age = age_grid[:, :-2].ravel() + both = np.isfinite(now) & np.isfinite(later) + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = both & (sex == s) & (age == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + overall = float(np.corrcoef(now[both], later[both])[0, 1]) + four_now = grid[:, :-4].ravel() + four_later = grid[:, 4:].ravel() + four = np.isfinite(four_now) & np.isfinite(four_later) + lag4 = float(np.corrcoef(four_now[four], four_later[four])[0, 1]) + return rho, { + "rho_persons": int(len(chosen)), + "rho_pairs": int(both.sum()), + "rho_overall": overall, + "lag4_normal_score_correlation": lag4, + "lag4_ar1_prediction": overall**2, + } + + # -- filling -------------------------------------------------------------- + def fill( + self, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + person_key: np.ndarray, + fill_mask: np.ndarray, + seed: int, + ) -> np.ndarray: + """Fill the masked cells; masked cells with no known neighbour stay NaN.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + parts = self._parts(context, self.share_edges, self.context_median) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + rho = self.rho[np.clip(parts["sex"], 0, 3), parts["age"]] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + valid = parts["valid"] + drawn = np.full(len(rows), np.nan) + if valid.any(): + sub = {k: v[valid] for k, v in parts.items()} + drawn[valid] = self._quantile(sub, ndtr(z[valid])) + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + # -- persistence ---------------------------------------------------------- + def to_bytes(self) -> bytes: + arrays = { + "kind": np.array(self.name), + "share_edges": self.share_edges, + "context_median": np.array(self.context_median), + "rho": self.rho, + } + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> OddQuantileFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + share_edges=arrays["share_edges"], + context_median=float(arrays["context_median"]), + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + rho=arrays["rho"], + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: a quantile regression forest (QRF) draw +# -------------------------------------------------------------------------- + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +#: The reference level of a unit with no positive share around it. +_DEFAULT_LEVEL = 0.3 + + +def reference_level(context: OddContext) -> np.ndarray: + """The level a unit's share is drawn relative to. + + The geometric mean of the positive shares at ``t-1`` and ``t+1``; else + of those at ``t-3`` and ``t+3``; else the mean positive share at + offsets 5-9; else 0.3. + """ + + def geometric(a, b): + a = np.nan_to_num(a, nan=0.0) + b = np.nan_to_num(b, nan=0.0) + both = (a > 0) & (b > 0) + one = np.where(a > 0, a, b) + value = np.where(both, np.sqrt(np.where(both, a * b, 1.0)), one) + return np.where((a > 0) | (b > 0), value, np.nan) + + level = geometric(context.left, context.right) + level = np.where( + np.isnan(level), geometric(context.left3, context.right3), level + ) + level = np.where( + np.isnan(level) & (context.wide_mean > 0), context.wide_mean, level + ) + return np.where(np.isnan(level), _DEFAULT_LEVEL, level) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.91, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + Two parts, both random forests (scikit-learn) on :func:`odd_features` + of TRAIN units inside the career, whose contexts see the career only: + + 1. a probability forest for a zero year: ``p0`` is the mean over trees + of the zero share of the unit's leaves; + 2. a quantile regression forest on positive shares (split target + ``log share``): every positive TRAIN unit used in the fit is passed + down every tree, and each leaf keeps the sorted true shares that + reach it (the cap included, stored as shares times 65,535). + + A draw maps the copula uniform ``u`` to zero below ``p0``; otherwise a + second seeded uniform picks a tree, and the share is that tree's leaf + value at the quantile ``(u - p0) / (1 - p0)``. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + zero_tree_offsets: np.ndarray + zero_node_left: np.ndarray + zero_node_right: np.ndarray + zero_node_feature: np.ndarray + zero_node_threshold: np.ndarray + zero_node_leaf: np.ndarray + zero_leaf_p: np.ndarray + stream: str = "epuf_fill.odd_forest.v4" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + # Units lie inside the career, and their contexts see the career + # only, as a fill's do (pre-career years are unknown to it). + inside = unit_year >= career_start(birth_year[rows]) + rows, unit_year = rows[inside], unit_year[inside] + pre_career = years[None, :] < career_start(birth_year)[:, None] + known = np.isfinite(shares) & ~pre_career + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x_all = odd_features(context) + y_all = target[chosen] + # Part one: the probability of a zero year, a probability forest. + from sklearn.ensemble import RandomForestClassifier + + zero_forest = RandomForestClassifier( + n_estimators=n_trees, + min_samples_leaf=4 * min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + zero_forest.fit(x_all, (y_all <= 0).astype(np.int8)) + zero_arrays = _forest_arrays( + zero_forest, x_all, (y_all <= 0).astype(np.float64) + ) + # Part two: the positive share, a quantile regression forest. + positive = y_all > 0 + x = x_all[positive] + y = y_all[positive] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + **zero_arrays, + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y_all)), + "n_positive_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _p_zero(self, x) -> np.ndarray: + """The probability forest's zero-year probability (mean over trees).""" + + x = np.asarray(x, dtype=np.float32) + n_trees = len(self.zero_tree_offsets) - 1 + total = np.zeros(len(x)) + for tree in range(n_trees): + start = self.zero_tree_offsets[tree] + stop = self.zero_tree_offsets[tree + 1] + node = _tree_leaves( + self.zero_node_left[start:stop].astype(np.int64), + self.zero_node_right[start:stop].astype(np.int64), + self.zero_node_feature[start:stop].astype(np.int64), + self.zero_node_threshold[start:stop], + x, + ) + total += self.zero_leaf_p[ + self.zero_node_leaf[start:stop][node].astype(np.int64) + ] + return total / n_trees + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + p_zero = np.ones(len(rows)) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + p_zero[valid] = self._p_zero(features) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "p_zero": p_zero, + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + u = ndtr(z) + p0 = unit["p_zero"] + valid = valid & (u >= p0) + v = (u[valid] - p0[valid]) / np.maximum(1.0 - p0[valid], 1e-12) + drawn[valid] = self._value(unit["leaves"][valid], v) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + pre_career = years[None, :] < start[:, None] + mask &= ~pre_career + given = np.where(mask | pre_career, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +def _forest_arrays(forest, x, y) -> dict[str, np.ndarray]: + """A fitted probability forest as arrays: nodes, and each leaf's mean y.""" + + offsets = [0] + lefts, rights, features, thresholds, leaf_index, means = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + n_leaves = int(is_leaf.sum()) + index = np.full(tree.node_count, -1, dtype=np.int64) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + total = np.bincount(local, minlength=n_leaves) + hits = np.bincount(local, weights=y, minlength=n_leaves) + means.append(hits / np.maximum(total, 1)) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + offsets.append(offsets[-1] + tree.node_count) + return { + "zero_tree_offsets": np.asarray(offsets, dtype=np.int64), + "zero_node_left": np.concatenate(lefts).astype(np.int32), + "zero_node_right": np.concatenate(rights).astype(np.int32), + "zero_node_feature": np.concatenate(features), + "zero_node_threshold": np.concatenate(thresholds), + "zero_node_leaf": np.concatenate(leaf_index).astype(np.int32), + "zero_leaf_p": np.concatenate(means).astype(np.float32), + } + + +_FOREST_ARRAYS = ( + "zero_tree_offsets", + "zero_node_left", + "zero_node_right", + "zero_node_feature", + "zero_node_threshold", + "zero_node_leaf", + "zero_leaf_p", + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + known = np.isfinite(shares) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum): + members = np.flatnonzero(stratum == value) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=context.left[keep].astype(np.float32), + bank_right=context.right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:_DONOR_BANK] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + -1 for rows with no masked cell or no bank donor of their sex and + birth year. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + out = np.full(len(shares), -1, dtype=np.int64) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + recipients = targets[ + (sex[targets] == s) & (birth_year[targets] == b) + ] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + column = np.sort( + donor_match[:, j][np.isfinite(donor_match[:, j])] + ) + ranks_donor[:, j] = np.where( + np.isfinite(donor_match[:, j]), + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + nearest = np.argpartition(distance, k - 1, axis=1)[:, :k] + nearest_distance = np.take_along_axis(distance, nearest, 1) + order = np.lexsort((nearest, nearest_distance), axis=1) + nearest = np.take_along_axis(nearest, order, 1) + pick = np.minimum( + (u[recipients[block]] * k).astype(np.int64), k - 1 + ) + no_match = count.max(axis=1) == 0 + random_donor = np.minimum( + (u[recipients[block]] * len(donors)).astype(np.int64), + len(donors) - 1, + ) + out[recipients[block]] = donors[ + np.where( + no_match, + random_donor, + nearest[np.arange(len(nearest)), pick], + ) + ] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_quantile": OddQuantileFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_5af343166b51.py b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_5af343166b51.py new file mode 100644 index 00000000..787f5aaa --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_5af343166b51.py @@ -0,0 +1,2127 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddQuantileFill` (odd years, primary): a two-part conditional + draw. The probability of a zero year and the conditional quantiles of a + positive share, relative to the neighbours' level, by sex, age, the + shares at ``t-1`` and ``t+1`` and the context at ``t-3``, ``t+3`` and + further; a Gaussian AR(1) copula correlates a person's draws across + masked years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "OddQuantileFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _invert(table: np.ndarray, rows: np.ndarray, value: np.ndarray): + """The level at which each row's quantile function reaches ``value``. + + Where the function is flat at ``value`` (a run of equal quantiles), the + middle of the run's levels. + """ + + points = table.shape[1] + levels = np.linspace(0.0, 1.0, points) + out = np.empty(len(rows)) + for start in range(0, len(rows), 200_000): + block = slice(start, start + 200_000) + curve = table[rows[block]].astype(np.float64) + target = np.asarray(value[block], dtype=np.float64)[:, None] + below = (curve < target).sum(axis=1) + above = (curve <= target).sum(axis=1) + flat = below < above + result = np.empty(len(curve)) + result[above == 0] = 0.0 + result[below >= points] = 1.0 + middle = flat & (above > 0) & (below < points) + result[middle] = 0.5 * ( + levels[below[middle]] + levels[above[middle] - 1] + ) + between = ~flat & (below > 0) & (below < points) + index = np.flatnonzero(between) + left = curve[index, below[index] - 1] + right = curve[index, below[index]] + share = np.where( + right > left, + (target[index, 0] - left) + / np.where(right > left, right - left, 1), + 0.5, + ) + result[index] = levels[below[index] - 1] + share * ( + levels[below[index]] - levels[below[index] - 1] + ) + out[block] = result + return out + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + """A compressed ``.npz`` whose bytes depend only on the arrays. + + ``numpy.savez_compressed`` stamps each member with the time of writing, + so two writes of the same fill differ. This writer fixes every member's + timestamp and order, so a fill's SHA-256 can be registered and refit. + """ + + import zipfile + + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive: + for name in sorted(arrays): + member = io.BytesIO() + np.lib.format.write_array( + member, np.asanyarray(arrays[name]), allow_pickle=False + ) + info = zipfile.ZipInfo( + f"{name}.npy", date_time=(1980, 1, 1, 0, 0, 0) + ) + info.compress_type = zipfile.ZIP_DEFLATED + archive.writestr(info, member.getvalue()) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: the two-part conditional draw +# -------------------------------------------------------------------------- +_N_SHARE_BINS = 20 + + +def _share_bin(value: np.ndarray, edges: np.ndarray) -> np.ndarray: + """0 zero, 1..20 quantile bins of a positive share below the cap, 21 cap.""" + + bins = np.searchsorted(edges, value, side="right") + 1 + bins = np.where(value <= 0, 0, bins) + return np.where(value >= 1.0, _N_SHARE_BINS + 1, bins) + + +def _coarse_age(band: np.ndarray) -> np.ndarray: + """Age bands grouped: under 30, 30-44, 45-59, 60 and over.""" + + return np.digitize(band, [4, 7, 10]) + + +@dataclass(frozen=True) +class OddQuantileFill: + """Two-part conditional draw for masked odd years, with an AR(1) copula. + + A unit's reference level ``m`` is the geometric mean of its positive + neighbours' shares (the one positive neighbour's share if only one is, + 1 if neither is). Its cell is the finest of seven nested keys with at + least ``MIN_CELL`` TRAIN units, built from sex, five-year age band, the + bins of the shares at ``t-1`` and ``t+1`` (zero, 20 quantile bins of a + positive share below the cap, at the cap), the context at ``t-3`` and + ``t+3`` (missing, zero, below or above the median positive share), and + whether any share at offsets 5, 7 or 9 is positive. In the cell: ``p0`` + the share of zero years, and 65 quantiles of ``log(x_t / m)`` among + positive years. A uniform ``u`` maps to zero if ``u < p0``, else to + ``min(m * exp(Q((u - p0) / (1 - p0))), 1)``. The uniforms of a person's + consecutive masked years (two years apart) are joined by a Gaussian + AR(1) copula with correlation ``rho`` by sex and age band, learned on + TRAIN from the probability integral transforms of consecutive units. + """ + + share_edges: np.ndarray + context_median: float + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + rho: np.ndarray + stream: str = "epuf_fill.odd_quantile.v1" + name: str = "odd_quantile" + + # -- keys --------------------------------------------------------------- + @staticmethod + def _parts(context: OddContext, share_edges, median): + left = np.nan_to_num(context.left, nan=-1.0) + right = np.nan_to_num(context.right, nan=-1.0) + # A missing neighbour takes the other's value (the PSID fallback). + left = np.where(left < 0, right, left) + right = np.where(right < 0, left, right) + positive_left = np.where(left > 0, left, 1.0) + positive_right = np.where(right > 0, right, 1.0) + level = np.where( + (left > 0) & (right > 0), + np.sqrt(positive_left * positive_right), + np.where(left > 0, positive_left, positive_right), + ) + + def context_code(value): + return np.where( + np.isnan(value), + 0, + np.where(value <= 0, 1, np.where(value < median, 2, 3)), + ) + + return { + "sex": context.sex, + "age": _age_band(context.age), + "left_bin": _share_bin(left, share_edges), + "right_bin": _share_bin(right, share_edges), + "context3": 4 * context_code(context.left3) + + context_code(context.right3), + "wide": context.wide, + "level": level, + "valid": ~(np.isnan(context.left) & np.isnan(context.right)), + } + + @staticmethod + def _keys(parts) -> list[np.ndarray]: + sex = parts["sex"] + age = parts["age"] + coarse = _coarse_age(age) + left = parts["left_bin"] + right = parts["right_bin"] + context3 = parts["context3"] + wide = parts["wide"] + + # Nested keys from finest to coarsest; a dropped component is held + # at a sentinel (age 16-20 marks the coarse bands, 21 none). + def key(s, a, lb, rb, c3, w): + return ((((s * 22 + a) * 23 + lb) * 23 + rb) * 17 + c3) * 3 + w + + return [ + key(sex, age, left, right, context3, wide), + key(sex, age, left, right, context3, 2), + key(sex, age, left, right, 16, 2), + key(sex, 16 + coarse, left, right, 16, 2), + key(sex, 21, left, right, 16, 2), + key(0 * sex, 21, left, right, 16, 2), + key(0 * sex, 21, np.minimum(left, 1), np.minimum(right, 1), 16, 2), + ] + + def _cells(self, parts) -> tuple[np.ndarray, np.ndarray]: + """(level, row) of each unit's finest populated cell.""" + + keys = self._keys(parts) + level = np.full(len(keys[0]), -1, dtype=np.int64) + row = np.full(len(keys[0]), -1, dtype=np.int64) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + if (level < 0).any(): + raise ValueError("a unit has no populated cell at any level") + return level, row + + def _quantile(self, parts, u: np.ndarray) -> np.ndarray: + level, row = self._cells(parts) + out = np.zeros(len(u)) + for index in np.unique(level): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate(self.level_quantiles[index], row[take], v) + share = np.minimum(parts["level"][take] * np.exp(residual), 1.0) + out[take] = np.where(positive, share, 0.0) + return out + + # -- fitting -------------------------------------------------------------- + @classmethod + def fit( + cls, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + unit_years: tuple[int, ...], + rho_seed: int = 0, + ) -> tuple[OddQuantileFill, dict[str, object]]: + """Fit on complete TRAIN shares; every year of ``unit_years`` a unit. + + Every person-year of ``unit_years`` whose two neighbours are inside + the matrix is a training unit (all years are recorded on TRAIN). + """ + + shares = np.asarray(shares, dtype=np.float64) + known = np.isfinite(shares) + n = len(shares) + rows_list, years_list = [], [] + for year in unit_years: + rows_list.append(np.arange(n)) + years_list.append(np.full(n, year)) + rows = np.concatenate(rows_list) + unit_year = np.concatenate(years_list) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + target = _take(shares, rows, _column_of(years, unit_year)) + neighbours = np.concatenate([context.left, context.right]) + inside = neighbours[(neighbours > 0) & (neighbours < 1.0)] + share_edges = np.quantile( + inside, np.linspace(0, 1, _N_SHARE_BINS + 1)[1:-1] + ) + median = float(np.median(inside)) + parts = cls._parts(context, share_edges, median) + keys = cls._keys(parts) + positive = target > 0 + residual = np.log(np.where(positive, target, 1.0)) - np.log( + parts["level"] + ) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zeros_unique, zeros = np.unique( + key[in_cells & ~positive], return_counts=True + ) + totals = count[count >= MIN_CELL] + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zeros_unique)] = zeros + p0 = p0 / totals + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + # A populated cell with no positive unit draws only zeros. + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0.astype(np.float64)) + level_quantiles.append(full) + provisional = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=np.zeros((4, 16)), + ) + rho, rho_diagnostics = provisional._fit_rho( + parts, target, rows, unit_year, rho_seed + ) + fill = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=rho, + ) + cells = [len(k) for k in level_keys] + return fill, { + "n_units": int(len(target)), + "cells_per_level": cells, + **rho_diagnostics, + } + + def _pit(self, parts, target, rows, unit_year, seed) -> np.ndarray: + """Randomised probability integral transforms of true shares.""" + + level, row = self._cells(parts) + jitter = hash_uniform( + "epuf_fill.odd_quantile.pit", seed, rows, unit_year + ) + out = np.empty(len(target)) + for index in np.unique(level): + take = np.flatnonzero(level == index) + p0 = self.level_p0[index][row[take]] + zero = target[take] <= 0 + out[take[zero]] = jitter[take[zero]] * p0[zero] + positive = take[~zero] + residual = np.log(target[positive]) - np.log( + parts["level"][positive] + ) + # At the cap the residual is censored: spread it over the mass + # the quantile function puts at or above the cap. + at_cap = target[positive] >= 1.0 + v = _invert(self.level_quantiles[index], row[positive], residual) + cap_v = v.copy() + cap_v[at_cap] = v[at_cap] + jitter[positive][at_cap] * ( + 1.0 - v[at_cap] + ) + p0_positive = p0[~zero] + out[positive] = p0_positive + (1.0 - p0_positive) * cap_v + return np.clip(out, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, parts, target, rows, unit_year, seed): + """AR(1) correlation of consecutive units' normal scores (t, t+2). + + On a 5 percent sample of persons (by seed): every unit's + probability integral transform under the fitted cells, its normal + score, and the correlation of the scores of ``t`` and ``t+2`` for + the same person, by sex and age band at ``t``. + """ + + persons = np.unique(rows) + rng = np.random.default_rng(seed) + chosen = persons[rng.random(len(persons)) < 0.05] + index = np.flatnonzero(np.isin(rows, chosen)) + sub = {k: v[index] for k, v in parts.items()} + z = ndtri( + self._pit(sub, target[index], rows[index], unit_year[index], seed) + ) + first_year = int(unit_year.min()) + n_years = int(unit_year.max()) - first_year + 1 + position = np.searchsorted(chosen, rows[index]) + grid = np.full((len(chosen), n_years), np.nan) + grid[position, unit_year[index] - first_year] = z + sex_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + sex_grid[position, unit_year[index] - first_year] = sub["sex"] + age_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + age_grid[position, unit_year[index] - first_year] = sub["age"] + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex = sex_grid[:, :-2].ravel() + age = age_grid[:, :-2].ravel() + both = np.isfinite(now) & np.isfinite(later) + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = both & (sex == s) & (age == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + overall = float(np.corrcoef(now[both], later[both])[0, 1]) + four_now = grid[:, :-4].ravel() + four_later = grid[:, 4:].ravel() + four = np.isfinite(four_now) & np.isfinite(four_later) + lag4 = float(np.corrcoef(four_now[four], four_later[four])[0, 1]) + return rho, { + "rho_persons": int(len(chosen)), + "rho_pairs": int(both.sum()), + "rho_overall": overall, + "lag4_normal_score_correlation": lag4, + "lag4_ar1_prediction": overall**2, + } + + # -- filling -------------------------------------------------------------- + def fill( + self, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + person_key: np.ndarray, + fill_mask: np.ndarray, + seed: int, + ) -> np.ndarray: + """Fill the masked cells; masked cells with no known neighbour stay NaN.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + parts = self._parts(context, self.share_edges, self.context_median) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + rho = self.rho[np.clip(parts["sex"], 0, 3), parts["age"]] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + valid = parts["valid"] + drawn = np.full(len(rows), np.nan) + if valid.any(): + sub = {k: v[valid] for k, v in parts.items()} + drawn[valid] = self._quantile(sub, ndtr(z[valid])) + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + # -- persistence ---------------------------------------------------------- + def to_bytes(self) -> bytes: + arrays = { + "kind": np.array(self.name), + "share_edges": self.share_edges, + "context_median": np.array(self.context_median), + "rho": self.rho, + } + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> OddQuantileFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + share_edges=arrays["share_edges"], + context_median=float(arrays["context_median"]), + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + rho=arrays["rho"], + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: a quantile regression forest (QRF) draw +# -------------------------------------------------------------------------- + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +#: The reference level of a unit with no positive share around it. +_DEFAULT_LEVEL = 0.3 + + +def reference_level(context: OddContext) -> np.ndarray: + """The level a unit's share is drawn relative to. + + The geometric mean of the positive shares at ``t-1`` and ``t+1``; else + of those at ``t-3`` and ``t+3``; else the mean positive share at + offsets 5-9; else 0.3. + """ + + def geometric(a, b): + a = np.nan_to_num(a, nan=0.0) + b = np.nan_to_num(b, nan=0.0) + both = (a > 0) & (b > 0) + one = np.where(a > 0, a, b) + value = np.where(both, np.sqrt(np.where(both, a * b, 1.0)), one) + return np.where((a > 0) | (b > 0), value, np.nan) + + level = geometric(context.left, context.right) + level = np.where( + np.isnan(level), geometric(context.left3, context.right3), level + ) + level = np.where( + np.isnan(level) & (context.wide_mean > 0), context.wide_mean, level + ) + return np.where(np.isnan(level), _DEFAULT_LEVEL, level) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.91, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + Two parts, both random forests (scikit-learn) on :func:`odd_features` + of TRAIN units inside the career, whose contexts see the career only: + + 1. a probability forest for a zero year: ``p0`` is the mean over trees + of the zero share of the unit's leaves; + 2. a quantile regression forest on positive shares (split target + ``log share``): every positive TRAIN unit used in the fit is passed + down every tree, and each leaf keeps the sorted true shares that + reach it (the cap included, stored as shares times 65,535). + + A draw maps the copula uniform ``u`` to zero below ``p0``; otherwise a + second seeded uniform picks a tree, and the share is that tree's leaf + value at the quantile ``(u - p0) / (1 - p0)``. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + zero_tree_offsets: np.ndarray + zero_node_left: np.ndarray + zero_node_right: np.ndarray + zero_node_feature: np.ndarray + zero_node_threshold: np.ndarray + zero_node_leaf: np.ndarray + zero_leaf_p: np.ndarray + stream: str = "epuf_fill.odd_forest.v4" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + # Units lie inside the career, and their contexts see the career + # only, as a fill's do (pre-career years are unknown to it). + inside = unit_year >= career_start(birth_year[rows]) + rows, unit_year = rows[inside], unit_year[inside] + pre_career = years[None, :] < career_start(birth_year)[:, None] + known = np.isfinite(shares) & ~pre_career + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x_all = odd_features(context) + y_all = target[chosen] + # Part one: the probability of a zero year, a probability forest. + from sklearn.ensemble import RandomForestClassifier + + zero_forest = RandomForestClassifier( + n_estimators=n_trees, + min_samples_leaf=4 * min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + zero_forest.fit(x_all, (y_all <= 0).astype(np.int8)) + zero_arrays = _forest_arrays( + zero_forest, x_all, (y_all <= 0).astype(np.float64) + ) + # Part two: the positive share, a quantile regression forest. + positive = y_all > 0 + x = x_all[positive] + y = y_all[positive] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + **zero_arrays, + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y_all)), + "n_positive_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _p_zero(self, x) -> np.ndarray: + """The probability forest's zero-year probability (mean over trees).""" + + x = np.asarray(x, dtype=np.float32) + n_trees = len(self.zero_tree_offsets) - 1 + total = np.zeros(len(x)) + for tree in range(n_trees): + start = self.zero_tree_offsets[tree] + stop = self.zero_tree_offsets[tree + 1] + node = _tree_leaves( + self.zero_node_left[start:stop].astype(np.int64), + self.zero_node_right[start:stop].astype(np.int64), + self.zero_node_feature[start:stop].astype(np.int64), + self.zero_node_threshold[start:stop], + x, + ) + total += self.zero_leaf_p[ + self.zero_node_leaf[start:stop][node].astype(np.int64) + ] + return total / n_trees + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + p_zero = np.ones(len(rows)) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + p_zero[valid] = self._p_zero(features) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "p_zero": p_zero, + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + u = ndtr(z) + p0 = unit["p_zero"] + valid = valid & (u >= p0) + v = (u[valid] - p0[valid]) / np.maximum(1.0 - p0[valid], 1e-12) + drawn[valid] = self._value(unit["leaves"][valid], v) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + pre_career = years[None, :] < start[:, None] + mask &= ~pre_career + given = np.where(mask | pre_career, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +def _forest_arrays(forest, x, y) -> dict[str, np.ndarray]: + """A fitted probability forest as arrays: nodes, and each leaf's mean y.""" + + offsets = [0] + lefts, rights, features, thresholds, leaf_index, means = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + n_leaves = int(is_leaf.sum()) + index = np.full(tree.node_count, -1, dtype=np.int64) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + total = np.bincount(local, minlength=n_leaves) + hits = np.bincount(local, weights=y, minlength=n_leaves) + means.append(hits / np.maximum(total, 1)) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + offsets.append(offsets[-1] + tree.node_count) + return { + "zero_tree_offsets": np.asarray(offsets, dtype=np.int64), + "zero_node_left": np.concatenate(lefts).astype(np.int32), + "zero_node_right": np.concatenate(rights).astype(np.int32), + "zero_node_feature": np.concatenate(features), + "zero_node_threshold": np.concatenate(thresholds), + "zero_node_leaf": np.concatenate(leaf_index).astype(np.int32), + "zero_leaf_p": np.concatenate(means).astype(np.float32), + } + + +_FOREST_ARRAYS = ( + "zero_tree_offsets", + "zero_node_left", + "zero_node_right", + "zero_node_feature", + "zero_node_threshold", + "zero_node_leaf", + "zero_leaf_p", + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + # Units inside the career; contexts see the career only. + start = career_start(np.asarray(birth_year)) + inside = unit_year >= start[rows] + rows, unit_year = rows[inside], unit_year[inside] + known = np.isfinite(shares) & ~( + np.asarray(years)[None, :] < start[:, None] + ) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum): + members = np.flatnonzero(stratum == value) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=context.left[keep].astype(np.float32), + bank_right=context.right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: Nearest-donor lists by input content, reused across draw seeds. +_NEAREST_CACHE: dict = {} +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:_DONOR_BANK] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def _nearest(self, match, birth_year, sex, targets): + """Each target's ``k`` nearest bank rows, and its group's bank rows. + + Seed-free, so it is computed once for a matrix and reused across + draw seeds (cached by the content of its inputs). + """ + + digest = hashlib.sha256( + np.ascontiguousarray(match[targets]).tobytes() + + np.ascontiguousarray(birth_year[targets]).tobytes() + + np.ascontiguousarray(sex[targets]).tobytes() + + np.ascontiguousarray(targets).tobytes() + + str((id(self), self.k)).encode() + ).hexdigest() + if digest in _NEAREST_CACHE: + return _NEAREST_CACHE[digest] + nearest = np.full((len(targets), self.k), -1, dtype=np.int64) + group_first = np.full(len(targets), -1, dtype=np.int64) + group_size = np.zeros(len(targets), dtype=np.int64) + no_match = np.zeros(len(targets), dtype=bool) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + local = np.flatnonzero( + (sex[targets] == s) & (birth_year[targets] == b) + ) + recipients = targets[local] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + group_first[local] = donors[0] + group_size[local] = len(donors) + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + finite = np.isfinite(donor_match[:, j]) + column = np.sort(donor_match[finite, j]) + ranks_donor[:, j] = np.where( + finite, + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + order = np.argpartition(distance, k - 1, axis=1)[:, :k] + near_distance = np.take_along_axis(distance, order, 1) + ranked = np.lexsort((order, near_distance), axis=1) + order = np.take_along_axis(order, ranked, 1) + rows = local[block] + nearest[rows, :k] = donors[order] + no_match[rows] = count.max(axis=1) == 0 + result = (nearest, group_first, group_size, no_match) + if len(_NEAREST_CACHE) >= 4: + _NEAREST_CACHE.pop(next(iter(_NEAREST_CACHE))) + _NEAREST_CACHE[digest] = result + return result + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + One of the ``k`` nearest bank donors of the recipient's sex and + birth year, chosen by the seeded uniform; a recipient with no + recorded match feature takes a random donor of its group. -1 for + rows with no masked cell or no bank donor of their group. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + nearest, group_first, group_size, no_match = self._nearest( + match, birth_year, sex, targets + ) + out = np.full(len(shares), -1, dtype=np.int64) + has_group = group_size > 0 + k_available = (nearest >= 0).sum(axis=1) + pick = np.minimum( + (u[targets] * np.maximum(k_available, 1)).astype(np.int64), + np.maximum(k_available - 1, 0), + ) + chosen = nearest[np.arange(len(targets)), pick] + random_donor = group_first + np.minimum( + (u[targets] * np.maximum(group_size, 1)).astype(np.int64), + np.maximum(group_size - 1, 0), + ) + chosen = np.where(no_match, random_donor, chosen) + out[targets[has_group]] = chosen[has_group] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_quantile": OddQuantileFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_c5e4a37417ec.py b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_c5e4a37417ec.py new file mode 100644 index 00000000..49420071 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidate_code/epuf_fill_c5e4a37417ec.py @@ -0,0 +1,2131 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddQuantileFill` (odd years, primary): a two-part conditional + draw. The probability of a zero year and the conditional quantiles of a + positive share, relative to the neighbours' level, by sex, age, the + shares at ``t-1`` and ``t+1`` and the context at ``t-3``, ``t+3`` and + further; a Gaussian AR(1) copula correlates a person's draws across + masked years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "OddQuantileFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _invert(table: np.ndarray, rows: np.ndarray, value: np.ndarray): + """The level at which each row's quantile function reaches ``value``. + + Where the function is flat at ``value`` (a run of equal quantiles), the + middle of the run's levels. + """ + + points = table.shape[1] + levels = np.linspace(0.0, 1.0, points) + out = np.empty(len(rows)) + for start in range(0, len(rows), 200_000): + block = slice(start, start + 200_000) + curve = table[rows[block]].astype(np.float64) + target = np.asarray(value[block], dtype=np.float64)[:, None] + below = (curve < target).sum(axis=1) + above = (curve <= target).sum(axis=1) + flat = below < above + result = np.empty(len(curve)) + result[above == 0] = 0.0 + result[below >= points] = 1.0 + middle = flat & (above > 0) & (below < points) + result[middle] = 0.5 * ( + levels[below[middle]] + levels[above[middle] - 1] + ) + between = ~flat & (below > 0) & (below < points) + index = np.flatnonzero(between) + left = curve[index, below[index] - 1] + right = curve[index, below[index]] + share = np.where( + right > left, + (target[index, 0] - left) + / np.where(right > left, right - left, 1), + 0.5, + ) + result[index] = levels[below[index] - 1] + share * ( + levels[below[index]] - levels[below[index] - 1] + ) + out[block] = result + return out + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + """A compressed ``.npz`` whose bytes depend only on the arrays. + + ``numpy.savez_compressed`` stamps each member with the time of writing, + so two writes of the same fill differ. This writer fixes every member's + timestamp and order, so a fill's SHA-256 can be registered and refit. + """ + + import zipfile + + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive: + for name in sorted(arrays): + member = io.BytesIO() + np.lib.format.write_array( + member, np.asanyarray(arrays[name]), allow_pickle=False + ) + info = zipfile.ZipInfo( + f"{name}.npy", date_time=(1980, 1, 1, 0, 0, 0) + ) + info.compress_type = zipfile.ZIP_DEFLATED + archive.writestr(info, member.getvalue()) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: the two-part conditional draw +# -------------------------------------------------------------------------- +_N_SHARE_BINS = 20 + + +def _share_bin(value: np.ndarray, edges: np.ndarray) -> np.ndarray: + """0 zero, 1..20 quantile bins of a positive share below the cap, 21 cap.""" + + bins = np.searchsorted(edges, value, side="right") + 1 + bins = np.where(value <= 0, 0, bins) + return np.where(value >= 1.0, _N_SHARE_BINS + 1, bins) + + +def _coarse_age(band: np.ndarray) -> np.ndarray: + """Age bands grouped: under 30, 30-44, 45-59, 60 and over.""" + + return np.digitize(band, [4, 7, 10]) + + +@dataclass(frozen=True) +class OddQuantileFill: + """Two-part conditional draw for masked odd years, with an AR(1) copula. + + A unit's reference level ``m`` is the geometric mean of its positive + neighbours' shares (the one positive neighbour's share if only one is, + 1 if neither is). Its cell is the finest of seven nested keys with at + least ``MIN_CELL`` TRAIN units, built from sex, five-year age band, the + bins of the shares at ``t-1`` and ``t+1`` (zero, 20 quantile bins of a + positive share below the cap, at the cap), the context at ``t-3`` and + ``t+3`` (missing, zero, below or above the median positive share), and + whether any share at offsets 5, 7 or 9 is positive. In the cell: ``p0`` + the share of zero years, and 65 quantiles of ``log(x_t / m)`` among + positive years. A uniform ``u`` maps to zero if ``u < p0``, else to + ``min(m * exp(Q((u - p0) / (1 - p0))), 1)``. The uniforms of a person's + consecutive masked years (two years apart) are joined by a Gaussian + AR(1) copula with correlation ``rho`` by sex and age band, learned on + TRAIN from the probability integral transforms of consecutive units. + """ + + share_edges: np.ndarray + context_median: float + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + rho: np.ndarray + stream: str = "epuf_fill.odd_quantile.v1" + name: str = "odd_quantile" + + # -- keys --------------------------------------------------------------- + @staticmethod + def _parts(context: OddContext, share_edges, median): + left = np.nan_to_num(context.left, nan=-1.0) + right = np.nan_to_num(context.right, nan=-1.0) + # A missing neighbour takes the other's value (the PSID fallback). + left = np.where(left < 0, right, left) + right = np.where(right < 0, left, right) + positive_left = np.where(left > 0, left, 1.0) + positive_right = np.where(right > 0, right, 1.0) + level = np.where( + (left > 0) & (right > 0), + np.sqrt(positive_left * positive_right), + np.where(left > 0, positive_left, positive_right), + ) + + def context_code(value): + return np.where( + np.isnan(value), + 0, + np.where(value <= 0, 1, np.where(value < median, 2, 3)), + ) + + return { + "sex": context.sex, + "age": _age_band(context.age), + "left_bin": _share_bin(left, share_edges), + "right_bin": _share_bin(right, share_edges), + "context3": 4 * context_code(context.left3) + + context_code(context.right3), + "wide": context.wide, + "level": level, + "valid": ~(np.isnan(context.left) & np.isnan(context.right)), + } + + @staticmethod + def _keys(parts) -> list[np.ndarray]: + sex = parts["sex"] + age = parts["age"] + coarse = _coarse_age(age) + left = parts["left_bin"] + right = parts["right_bin"] + context3 = parts["context3"] + wide = parts["wide"] + + # Nested keys from finest to coarsest; a dropped component is held + # at a sentinel (age 16-20 marks the coarse bands, 21 none). + def key(s, a, lb, rb, c3, w): + return ((((s * 22 + a) * 23 + lb) * 23 + rb) * 17 + c3) * 3 + w + + return [ + key(sex, age, left, right, context3, wide), + key(sex, age, left, right, context3, 2), + key(sex, age, left, right, 16, 2), + key(sex, 16 + coarse, left, right, 16, 2), + key(sex, 21, left, right, 16, 2), + key(0 * sex, 21, left, right, 16, 2), + key(0 * sex, 21, np.minimum(left, 1), np.minimum(right, 1), 16, 2), + ] + + def _cells(self, parts) -> tuple[np.ndarray, np.ndarray]: + """(level, row) of each unit's finest populated cell.""" + + keys = self._keys(parts) + level = np.full(len(keys[0]), -1, dtype=np.int64) + row = np.full(len(keys[0]), -1, dtype=np.int64) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + if (level < 0).any(): + raise ValueError("a unit has no populated cell at any level") + return level, row + + def _quantile(self, parts, u: np.ndarray) -> np.ndarray: + level, row = self._cells(parts) + out = np.zeros(len(u)) + for index in np.unique(level): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate(self.level_quantiles[index], row[take], v) + share = np.minimum(parts["level"][take] * np.exp(residual), 1.0) + out[take] = np.where(positive, share, 0.0) + return out + + # -- fitting -------------------------------------------------------------- + @classmethod + def fit( + cls, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + unit_years: tuple[int, ...], + rho_seed: int = 0, + ) -> tuple[OddQuantileFill, dict[str, object]]: + """Fit on complete TRAIN shares; every year of ``unit_years`` a unit. + + Every person-year of ``unit_years`` whose two neighbours are inside + the matrix is a training unit (all years are recorded on TRAIN). + """ + + shares = np.asarray(shares, dtype=np.float64) + known = np.isfinite(shares) + n = len(shares) + rows_list, years_list = [], [] + for year in unit_years: + rows_list.append(np.arange(n)) + years_list.append(np.full(n, year)) + rows = np.concatenate(rows_list) + unit_year = np.concatenate(years_list) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + target = _take(shares, rows, _column_of(years, unit_year)) + neighbours = np.concatenate([context.left, context.right]) + inside = neighbours[(neighbours > 0) & (neighbours < 1.0)] + share_edges = np.quantile( + inside, np.linspace(0, 1, _N_SHARE_BINS + 1)[1:-1] + ) + median = float(np.median(inside)) + parts = cls._parts(context, share_edges, median) + keys = cls._keys(parts) + positive = target > 0 + residual = np.log(np.where(positive, target, 1.0)) - np.log( + parts["level"] + ) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zeros_unique, zeros = np.unique( + key[in_cells & ~positive], return_counts=True + ) + totals = count[count >= MIN_CELL] + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zeros_unique)] = zeros + p0 = p0 / totals + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + # A populated cell with no positive unit draws only zeros. + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0.astype(np.float64)) + level_quantiles.append(full) + provisional = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=np.zeros((4, 16)), + ) + rho, rho_diagnostics = provisional._fit_rho( + parts, target, rows, unit_year, rho_seed + ) + fill = cls( + share_edges=share_edges, + context_median=median, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + rho=rho, + ) + cells = [len(k) for k in level_keys] + return fill, { + "n_units": int(len(target)), + "cells_per_level": cells, + **rho_diagnostics, + } + + def _pit(self, parts, target, rows, unit_year, seed) -> np.ndarray: + """Randomised probability integral transforms of true shares.""" + + level, row = self._cells(parts) + jitter = hash_uniform( + "epuf_fill.odd_quantile.pit", seed, rows, unit_year + ) + out = np.empty(len(target)) + for index in np.unique(level): + take = np.flatnonzero(level == index) + p0 = self.level_p0[index][row[take]] + zero = target[take] <= 0 + out[take[zero]] = jitter[take[zero]] * p0[zero] + positive = take[~zero] + residual = np.log(target[positive]) - np.log( + parts["level"][positive] + ) + # At the cap the residual is censored: spread it over the mass + # the quantile function puts at or above the cap. + at_cap = target[positive] >= 1.0 + v = _invert(self.level_quantiles[index], row[positive], residual) + cap_v = v.copy() + cap_v[at_cap] = v[at_cap] + jitter[positive][at_cap] * ( + 1.0 - v[at_cap] + ) + p0_positive = p0[~zero] + out[positive] = p0_positive + (1.0 - p0_positive) * cap_v + return np.clip(out, 1e-9, 1.0 - 1e-9) + + def _fit_rho(self, parts, target, rows, unit_year, seed): + """AR(1) correlation of consecutive units' normal scores (t, t+2). + + On a 5 percent sample of persons (by seed): every unit's + probability integral transform under the fitted cells, its normal + score, and the correlation of the scores of ``t`` and ``t+2`` for + the same person, by sex and age band at ``t``. + """ + + persons = np.unique(rows) + rng = np.random.default_rng(seed) + chosen = persons[rng.random(len(persons)) < 0.05] + index = np.flatnonzero(np.isin(rows, chosen)) + sub = {k: v[index] for k, v in parts.items()} + z = ndtri( + self._pit(sub, target[index], rows[index], unit_year[index], seed) + ) + first_year = int(unit_year.min()) + n_years = int(unit_year.max()) - first_year + 1 + position = np.searchsorted(chosen, rows[index]) + grid = np.full((len(chosen), n_years), np.nan) + grid[position, unit_year[index] - first_year] = z + sex_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + sex_grid[position, unit_year[index] - first_year] = sub["sex"] + age_grid = np.zeros((len(chosen), n_years), dtype=np.int64) + age_grid[position, unit_year[index] - first_year] = sub["age"] + now = grid[:, :-2].ravel() + later = grid[:, 2:].ravel() + sex = sex_grid[:, :-2].ravel() + age = age_grid[:, :-2].ravel() + both = np.isfinite(now) & np.isfinite(later) + rho = np.zeros((4, 16)) + for s in (1, 2): + for a in range(16): + take = both & (sex == s) & (age == a) + if take.sum() >= MIN_CELL: + rho[s, a] = np.corrcoef(now[take], later[take])[0, 1] + overall = float(np.corrcoef(now[both], later[both])[0, 1]) + four_now = grid[:, :-4].ravel() + four_later = grid[:, 4:].ravel() + four = np.isfinite(four_now) & np.isfinite(four_later) + lag4 = float(np.corrcoef(four_now[four], four_later[four])[0, 1]) + return rho, { + "rho_persons": int(len(chosen)), + "rho_pairs": int(both.sum()), + "rho_overall": overall, + "lag4_normal_score_correlation": lag4, + "lag4_ar1_prediction": overall**2, + } + + # -- filling -------------------------------------------------------------- + def fill( + self, + shares: np.ndarray, + years: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + person_key: np.ndarray, + fill_mask: np.ndarray, + seed: int, + ) -> np.ndarray: + """Fill the masked cells; masked cells with no known neighbour stay NaN.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + latent = np.full(len(shares), np.nan) + last_year = np.full(len(shares), -10, dtype=np.int64) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + parts = self._parts(context, self.share_edges, self.context_median) + epsilon = ndtri( + hash_uniform(self.stream, seed, person_key[rows], unit_year) + ) + rho = self.rho[np.clip(parts["sex"], 0, 3), parts["age"]] + follows = last_year[rows] == year - 2 + z = np.where( + follows, + rho * np.nan_to_num(latent[rows]) + + np.sqrt(1.0 - rho**2) * epsilon, + epsilon, + ) + valid = parts["valid"] + drawn = np.full(len(rows), np.nan) + if valid.any(): + sub = {k: v[valid] for k, v in parts.items()} + drawn[valid] = self._quantile(sub, ndtr(z[valid])) + out[rows, column] = drawn + latent[rows] = np.where(valid, z, np.nan) + last_year[rows] = np.where(valid, year, -10) + return out + + # -- persistence ---------------------------------------------------------- + def to_bytes(self) -> bytes: + arrays = { + "kind": np.array(self.name), + "share_edges": self.share_edges, + "context_median": np.array(self.context_median), + "rho": self.rho, + } + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> OddQuantileFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + share_edges=arrays["share_edges"], + context_median=float(arrays["context_median"]), + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + rho=arrays["rho"], + ) + + +# -------------------------------------------------------------------------- +# Odd years, primary: a quantile regression forest (QRF) draw +# -------------------------------------------------------------------------- + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +#: The reference level of a unit with no positive share around it. +_DEFAULT_LEVEL = 0.3 + + +def reference_level(context: OddContext) -> np.ndarray: + """The level a unit's share is drawn relative to. + + The geometric mean of the positive shares at ``t-1`` and ``t+1``; else + of those at ``t-3`` and ``t+3``; else the mean positive share at + offsets 5-9; else 0.3. + """ + + def geometric(a, b): + a = np.nan_to_num(a, nan=0.0) + b = np.nan_to_num(b, nan=0.0) + both = (a > 0) & (b > 0) + one = np.where(a > 0, a, b) + value = np.where(both, np.sqrt(np.where(both, a * b, 1.0)), one) + return np.where((a > 0) | (b > 0), value, np.nan) + + level = geometric(context.left, context.right) + level = np.where( + np.isnan(level), geometric(context.left3, context.right3), level + ) + level = np.where( + np.isnan(level) & (context.wide_mean > 0), context.wide_mean, level + ) + return np.where(np.isnan(level), _DEFAULT_LEVEL, level) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.91, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + Two parts, both random forests (scikit-learn) on :func:`odd_features` + of TRAIN units inside the career, whose contexts see the career only: + + 1. a probability forest for a zero year: ``p0`` is the mean over trees + of the zero share of the unit's leaves; + 2. a quantile regression forest on positive shares (split target + ``log share``): every positive TRAIN unit used in the fit is passed + down every tree, and each leaf keeps the sorted true shares that + reach it (the cap included, stored as shares times 65,535). + + A draw maps the copula uniform ``u`` to zero below ``p0``; otherwise a + second seeded uniform picks a tree, and the share is that tree's leaf + value at the quantile ``(u - p0) / (1 - p0)``. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + zero_tree_offsets: np.ndarray + zero_node_left: np.ndarray + zero_node_right: np.ndarray + zero_node_feature: np.ndarray + zero_node_threshold: np.ndarray + zero_node_leaf: np.ndarray + zero_leaf_p: np.ndarray + stream: str = "epuf_fill.odd_forest.v4" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + # Units lie inside the career, and their contexts see the career + # only, as a fill's do (pre-career years are unknown to it). + inside = unit_year >= career_start(birth_year[rows]) + rows, unit_year = rows[inside], unit_year[inside] + pre_career = years[None, :] < career_start(birth_year)[:, None] + known = np.isfinite(shares) & ~pre_career + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x_all = odd_features(context) + y_all = target[chosen] + # Part one: the probability of a zero year, a probability forest. + from sklearn.ensemble import RandomForestClassifier + + zero_forest = RandomForestClassifier( + n_estimators=n_trees, + min_samples_leaf=4 * min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + zero_forest.fit(x_all, (y_all <= 0).astype(np.int8)) + zero_arrays = _forest_arrays( + zero_forest, x_all, (y_all <= 0).astype(np.float64) + ) + # Part two: the positive share, a quantile regression forest. + positive = y_all > 0 + x = x_all[positive] + y = y_all[positive] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + **zero_arrays, + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y_all)), + "n_positive_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _p_zero(self, x) -> np.ndarray: + """The probability forest's zero-year probability (mean over trees).""" + + x = np.asarray(x, dtype=np.float32) + n_trees = len(self.zero_tree_offsets) - 1 + total = np.zeros(len(x)) + for tree in range(n_trees): + start = self.zero_tree_offsets[tree] + stop = self.zero_tree_offsets[tree + 1] + node = _tree_leaves( + self.zero_node_left[start:stop].astype(np.int64), + self.zero_node_right[start:stop].astype(np.int64), + self.zero_node_feature[start:stop].astype(np.int64), + self.zero_node_threshold[start:stop], + x, + ) + total += self.zero_leaf_p[ + self.zero_node_leaf[start:stop][node].astype(np.int64) + ] + return total / n_trees + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + p_zero = np.ones(len(rows)) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + p_zero[valid] = self._p_zero(features) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "p_zero": p_zero, + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + u = ndtr(z) + p0 = unit["p_zero"] + valid = valid & (u >= p0) + v = (u[valid] - p0[valid]) / np.maximum(1.0 - p0[valid], 1e-12) + drawn[valid] = self._value(unit["leaves"][valid], v) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + pre_career = years[None, :] < start[:, None] + mask &= ~pre_career + given = np.where(mask | pre_career, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +def _forest_arrays(forest, x, y) -> dict[str, np.ndarray]: + """A fitted probability forest as arrays: nodes, and each leaf's mean y.""" + + offsets = [0] + lefts, rights, features, thresholds, leaf_index, means = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + n_leaves = int(is_leaf.sum()) + index = np.full(tree.node_count, -1, dtype=np.int64) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + total = np.bincount(local, minlength=n_leaves) + hits = np.bincount(local, weights=y, minlength=n_leaves) + means.append(hits / np.maximum(total, 1)) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + offsets.append(offsets[-1] + tree.node_count) + return { + "zero_tree_offsets": np.asarray(offsets, dtype=np.int64), + "zero_node_left": np.concatenate(lefts).astype(np.int32), + "zero_node_right": np.concatenate(rights).astype(np.int32), + "zero_node_feature": np.concatenate(features), + "zero_node_threshold": np.concatenate(thresholds), + "zero_node_leaf": np.concatenate(leaf_index).astype(np.int32), + "zero_leaf_p": np.concatenate(means).astype(np.float32), + } + + +_FOREST_ARRAYS = ( + "zero_tree_offsets", + "zero_node_left", + "zero_node_right", + "zero_node_feature", + "zero_node_threshold", + "zero_node_leaf", + "zero_leaf_p", + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + # Units inside the career; contexts see the career only. + start = career_start(np.asarray(birth_year)) + inside = unit_year >= start[rows] + rows, unit_year = rows[inside], unit_year[inside] + known = np.isfinite(shares) & ~( + np.asarray(years)[None, :] < start[:, None] + ) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + # A missing neighbour takes the other's value, as in the draw. + left = np.where(np.isnan(context.left), context.right, context.left) + right = np.where(np.isnan(context.right), context.left, context.right) + usable = np.isfinite(left) & np.isfinite(right) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum[usable]): + members = np.flatnonzero((stratum == value) & usable) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=left[keep].astype(np.float32), + bank_right=right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: Nearest-donor lists by input content, reused across draw seeds. +_NEAREST_CACHE: dict = {} +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + career = years[None, :] >= career_start(birth_year)[:, None] + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 +_SHARE_SCALE = 65_535 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to 2,000 TRAIN donors (those with + a positive share from their career start through 2006, chosen by the + lowest hash of their person id) holds each donor's shares in the years + from :func:`block_first_year` to the year before the career start (at + most the 17 years 1951-1967; stored as shares times 65,535, rounded), + and their shares in the first five years from the career start. A + recipient's match vector is its percentile mid-rank, within the bank, + in each of those five years it has recorded; distance is Euclidean over + the recorded years, scaled by five over their number. One of the ``k`` + nearest donors is chosen by the seeded uniform and its block copied; + masked years before :func:`block_first_year` are zero. A recipient with + no recorded match year takes a donor chosen at random from the bank. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:_DONOR_BANK] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round(values * _SHARE_SCALE).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + def _nearest(self, match, birth_year, sex, targets): + """Each target's ``k`` nearest bank rows, and its group's bank rows. + + Seed-free, so it is computed once for a matrix and reused across + draw seeds (cached by the content of its inputs). + """ + + digest = hashlib.sha256( + np.ascontiguousarray(match[targets]).tobytes() + + np.ascontiguousarray(birth_year[targets]).tobytes() + + np.ascontiguousarray(sex[targets]).tobytes() + + np.ascontiguousarray(targets).tobytes() + + str((id(self), self.k)).encode() + ).hexdigest() + if digest in _NEAREST_CACHE: + return _NEAREST_CACHE[digest] + nearest = np.full((len(targets), self.k), -1, dtype=np.int64) + group_first = np.full(len(targets), -1, dtype=np.int64) + group_size = np.zeros(len(targets), dtype=np.int64) + no_match = np.zeros(len(targets), dtype=bool) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + local = np.flatnonzero( + (sex[targets] == s) & (birth_year[targets] == b) + ) + recipients = targets[local] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + group_first[local] = donors[0] + group_size[local] = len(donors) + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + finite = np.isfinite(donor_match[:, j]) + column = np.sort(donor_match[finite, j]) + ranks_donor[:, j] = np.where( + finite, + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + order = np.argpartition(distance, k - 1, axis=1)[:, :k] + near_distance = np.take_along_axis(distance, order, 1) + ranked = np.lexsort((order, near_distance), axis=1) + order = np.take_along_axis(order, ranked, 1) + rows = local[block] + nearest[rows, :k] = donors[order] + no_match[rows] = count.max(axis=1) == 0 + result = (nearest, group_first, group_size, no_match) + if len(_NEAREST_CACHE) >= 4: + _NEAREST_CACHE.pop(next(iter(_NEAREST_CACHE))) + _NEAREST_CACHE[digest] = result + return result + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + One of the ``k`` nearest bank donors of the recipient's sex and + birth year, chosen by the seeded uniform; a recipient with no + recorded match feature takes a random donor of its group. -1 for + rows with no masked cell or no bank donor of their group. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + nearest, group_first, group_size, no_match = self._nearest( + match, birth_year, sex, targets + ) + out = np.full(len(shares), -1, dtype=np.int64) + has_group = group_size > 0 + k_available = (nearest >= 0).sum(axis=1) + pick = np.minimum( + (u[targets] * np.maximum(k_available, 1)).astype(np.int64), + np.maximum(k_available - 1, 0), + ) + chosen = nearest[np.arange(len(targets)), pick] + random_donor = group_first + np.minimum( + (u[targets] * np.maximum(group_size, 1)).astype(np.int64), + np.maximum(group_size - 1, 0), + ) + chosen = np.where(no_match, random_donor, chosen) + out[targets[has_group]] = chosen[has_group] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + ok = np.ones(len(rows), dtype=bool) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = (level == index) & ok + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + sex[rows], + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_quantile": OddQuantileFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/docs/amendments/gate_epuf_fill_candidates_registration.md b/docs/amendments/gate_epuf_fill_candidates_registration.md new file mode 100644 index 00000000..03ff4652 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_candidates_registration.md @@ -0,0 +1,240 @@ +# gate_epuf_fill: registered candidates + +- **Registration id**: `2026-10-03-epuf-career-fill` +- **Gate**: `gate_epuf_fill` + (`docs/amendments/gate_epuf_fill_registration_proposal.md`). A gate here is + a pass-or-fail test whose rules and thresholds are fixed and published + before anything is scored against it. +- **Stage**: candidates registered before any TEST read. TEST is scored once, + by `scripts/score_epuf_fill_test.py`, after `gates.yaml` locks the gate on + Max's ratification (decision d927). +- **Code**: `src/populace_dynamics/estimates/epuf_fill.py` at the manifest's + `code_commit`. +- **Fitted artifacts**: `runs/epuf_fill_candidates_v1.json` records each + file's SHA-256, size, parameters and diagnostics, and the library versions. + - The files are fitted on EPUF TRAIN only, by `scripts/fit_epuf_fills.py`. + - They are byte-reproducible `.npz` files, staged outside the repository + as EPUF is (`~/PolicyEngine/epuf-data/fills`). + - `epuf_fill_scoring.score_registered` loads each one through `load_fill` + and refuses other bytes. + +## The registered artifacts + +- **Manifest**: `runs/epuf_fill_candidates_v1.json`, SHA-256 `83d17a14f960033c7c0ed0d602ae395ea4b8d66ff5facab85115f493f3b96e2c`. +- **Fitted at**: `b722382e` on TRAIN, with the code files clean. +- **Environment**: numpy 2.5.1, scipy 1.18.0, scikit-learn 1.9.0, Python 3.14.4, zlib 1.2.12, on macOS-26.6.2-arm64-arm-64bit-Mach-O. +- **Reproducibility**: + - A second fit at the same commit reproduced all four files byte for + byte. + - The refit after code review (below) did too. + - Reproducing the bytes needs the same library, zlib and platform + versions. The deflate output can depend on the zlib build. + +| Name | Role | File | SHA-256 | Bytes | +|---|---|---|---|---:| +| `odd_forest` | odd primary | `odd_forest_v1.npz` | `37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa` | 44,836,853 | +| `odd_knn` | odd alternative | `odd_knn_v1.npz` | `8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291` | 4,076,580 | +| `pre_donor` | pre primary | `pre_donor_v1.npz` | `3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7` | 31,768,107 | +| `pre_chain` | pre alternative | `pre_chain_v1.npz` | `8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724` | 236,450 | + +## The four candidates + +| Family | Role | Name | What it is | +|---|---|---|---| +| odd | primary | `odd_forest` | Two-part random-forest draw, fitted per sex | +| odd | alternative | `odd_knn` | kNN triples | +| pre | primary | `pre_donor` | Rank-kNN donor careers | +| pre | alternative | `pre_chain` | Chained one-sided draw | + +**`odd_forest`** (odd primary) is fitted per sex. +- **Part one**: a probability forest gives the chance of a zero year. +- **Part two**: a quantile regression forest gives the positive share. Its + leaves keep the true TRAIN shares, so its draws are true values. +- **Features**: the recorded shares at `t-1`, `t+1`, `t-3` and `t+3`; their + mean, geometric mean and count positive; the mean positive share and the + share of positive years at offsets 5-9; sex; age; and the year. +- **Training units**: TRAIN units inside the career, with contexts that see + the career only. `n_units`, 3,000,000, applies to each sex's forest. +- **Copula**: a person-level Gaussian copula with correlation by sex and + age band. It is calibrated on one TRAIN person in ten, held out of the + forests, to match two- and four-year persistence between masked years. + +**`odd_knn`** (odd alternative) draws the share at `t` from one of the 10 +nearest TRAIN career units in the shares at `t-1` and `t+1`, by sex and +five-year age band. + +**`pre_donor`** (pre primary) copies a whole masked block from one of the 3 +nearest TRAIN donors of the same sex and birth year. +- **Bank**: every TRAIN donor in the `pre` universe, up to 100,000 per sex + and birth year. +- **Distance**: percentile ranks over the first five recorded career years, + plus the career's mean share and share of positive years. + +**`pre_chain`** (pre alternative) draws year `y` from the next known later +year's share, sex and age, backward from the career start. Ages 15-24 are +single years. +- **What it is**: a one-step binned conditional-quantile chain, not a + quantile regression forest. Each year conditions on one earnings value + only: the nearest later share that is known or already drawn (`y+1` in + the TRAIN fit; `y+2` when `y+1` is the career start and a masked odd + year; zero when none is known). That share picks one of 22 bins (zero, 20 TRAIN quantile bins, + and the cap) and scales the draw. A sex-age-bin cell with fewer than 200 + TRAIN person-years falls back to sex and bin, then to bin alone. Each + cell stores `P(zero)` and 65 quantiles of the log ratio to the next share + (of the log share when that is zero). Each person-year's draw uses its + own uniform, is capped at the wage base, and is zero below age 15. +- **What its result can show**: how a fill conditioned on one later year + scores on the gate's cells. It says nothing about a QRF that conditions + on many of a person's real years. Describe its DEV and TEST results in + these terms. + +Each candidate's exact fit parameters are in the manifest and in +`scripts/fit_epuf_fills.py` (`REGISTERED`). + +## DEV development, disclosed + +Candidates were developed against DEV, as the registration allows. Every DEV +score is disclosed: +- `docs/amendments/gate_epuf_fill_dev_scores_before_amendment_1.json`; +- `docs/amendments/gate_epuf_fill_dev_scores_after_amendment_1.json`; +- `docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl`, which + logs every later score through the registered scoring path. + +The candidate code behind each later score is kept in +`docs/amendments/gate_epuf_fill_candidate_code/`. + +The last DEV scores of the four registered designs, against the v3 +tolerances, are below. They are scored on the registered artifacts with 20 +draw seeds; see the dry run below. + +| Candidate | Gating cells failed on DEV | Tier on DEV (fallback current rule) | +|---|---:|---| +| `odd_forest` | 7 of 183 (worst: `odd.men.a22_29.r1`, 2.69 tolerances) | improves | +| `odd_knn` | 46 of 183 (worst: `wint`, up to 10 tolerances) | not adopted | +| `pre_donor` | 0 of 136 (worst: 0.77 tolerances) | certified | +| `pre_chain` | 50 of 136 (worst: youth `ylevel`, 20 tolerances) | not adopted | + +`pre_chain`'s 50 failures are those of the one-step binned chain described +above, not of a QRF. Its five worst DEV cells are youth earnings levels and +zero shares over ages 15-21 (`ylevel`, `yzero`) and one pre-career level +(`plevel`); the log records only the five worst cells. + +**A sweep of K, shown to the ratifier.** Before Max ruled on d927 (ratify +`K = 1`), the dry run below was re-scored on a 0.005 grid of `K` from 0.25 +to 4 with the repository's own scoring and adoption functions, and every +breakpoint was checked exactly. `K = 1` reproduces the record exactly. The +same primaries are adopted for every `K` from 0.8968 to 4. The odd primary +is certified from `K = 2.6903` and not adopted below 0.8968. The pre +primary is certified from `K = 0.7693`. `K = 1` was +registered at `14045be4`, before any DEV score; the sweep is disclosed so +the ratification is read with it in view. Script, input and output: +`gate_epuf_fill_dev_k_sweep.py`, `gate_epuf_fill_dev_registered_dryrun.json` +(the dry run with every cell) and `gate_epuf_fill_dev_k_sweep.txt`, logged +as the last line of the DEV log. + +**The registered procedure, dry-run on DEV.** On 2026-10-04, +`epuf_fill_scoring.score_registered` ran with the DEV matrix in place of +TEST, the registered artifacts loaded by SHA-256, and all 20 draw seeds. It +is logged in `docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl`. + +| Family | Current rule failing | Primary | Alternative | Adopted | +|---|---|---|---|---| +| odd | 100 (fallback reading), 99 (two-sided) | improves, 7 of 183 failing | not adopted, 46 failing | primary, uncertified | +| pre | 131 | certified, 0 of 136 failing | not adopted, 50 failing | primary | + +The odd primary's seven failing cells on DEV: + +| Cell | Tolerances | +|---|---:| +| `odd.men.a22_29.r1` | 2.69 | +| `odd.women.a22_29.r1` | 2.02 | +| `odd.women.a22_29.zint` | 1.58 | +| `odd.men.a22_29.zint` | 1.37 | +| `odd.women.a22_74.zint` | 1.23 | +| `odd.women.a60_74.r1` | 1.17 | +| `odd.women.a22_74.r1` | 1.09 | + +**What DEV predicts, stated before TEST.** DEV and TEST are disjoint random +fifths of EPUF of nearly equal size, so their scores should agree closely. +- The pre-career primary should certify. +- The odd primary should be adopted as an uncertified improvement, unless + its young-band cells move. +- Most of the odd primary's residual misses are at ages 22-29, where the year + before the career is hidden from every fill. The oracle O1 misses there + too. + +## Code review and refit + +An independent code review of PR #516 returned REQUEST CHANGES +(`reviews/gate_epuf_fill_pr516_code_review_20261004.md`). The fixes: + +**`fill_careers` (PSID-2010).** +- It fills only the gap years the gate scored (1997-2005) unless the + caller opts into every gap year, which is an uncertified extrapolation. +- A gap year with no neighbour the fill can see keeps the assembler's + value. + +**`epuf_fill.py`.** +- The donor cache is keyed by the bank's content. +- Persons of uncoded sex take the routed part's copula. +- A forest leaf can never be empty: none is, and the smallest holds 16 + values. +- The career summaries in the donor match stop at 2006. + +**The scripts.** +- The fit script refuses an existing manifest before fitting and never + replaces a staged file with other bytes. +- The score script pins this manifest's SHA-256, records whether its code + was clean, and publishes no local paths. + +**The manifest.** The first manifest (SHA-256 `8d42153...`) was withdrawn +and refitted at the reviewed code (`598e4436`, SHA-256 `a3043113...`). +Rebasing onto #515's fixes then left that commit off the branch, so the +manifest was refitted once more at `b722382e`. The four +artifacts' SHA-256 values are the same in all three manifests. + +**The DEV dry run still stands.** It scored the same artifacts, and none of +the fixes changes a draw for a person the gate scores. They touch the PSID +application, which EPUF scoring does not use; the cache's key; the draws +for persons of uncoded sex, who are never scored; and the bank summaries' +years, which on EPUF end in 2006 anyway. + +**The confirmation review** of both PRs +(`reviews/gate_epuf_fill_pr515_pr516_confirmation_review_20261004.md`) +returned APPROVE WITH NITS. Its fixes in unpinned files are applied: +- The score script checks the staged files against the manifest before it + leaves any marker. +- Tests tie the score script's pin to the manifest and check its refusals. +- A run with any injected input is marked as not the registered scoring. + +Two nits in the pinned `epuf_fill.py` are left as they are, because fixing +them would need a refit. Neither changes a registered draw: +- `_BANK_DIGESTS` keeps each digested donor fill alive for the process's + life. +- The zero-year forest has no empty-leaf check. An empty leaf would give a + zero probability. + +## Procedure on TEST (after lock) + +1. Confirm that `gates.yaml` locks `gate_epuf_fill` and that the staged files + match the manifest's SHA-256. +2. Run `scripts/score_epuf_fill_test.py --manifest + runs/epuf_fill_candidates_v1.json --output runs/epuf_fill_gate_test_v1.json` + once. +3. Publish the result whether it passes or fails. Record each family's tier + and adoption under the registered rule. +4. Record the adoption in this document. A family whose primary and + alternative both fail to be adopted keeps the current rule. +5. Describe `pre_chain`'s result as that of a one-step binned + conditional-quantile chain (see "The four candidates"), never as a QRF + result. + +## Exploratory follow-up (after TEST, not a candidate) + +A full-career QRF fill, using microcosm-fit's QRF, in which each +pre-career year conditions on all of the person's recorded years. It would +show whether a QRF given many real predictors scores better on the cells +the one-step chain fails. If it is run, it is run only after the +registered TEST scoring and is reported beside the registered results, +labelled exploratory. It is not a candidate: it cannot change either +family's tier or adoption, and adopting it would need a new registration. diff --git a/docs/amendments/gate_epuf_fill_dev_k_sweep.py b/docs/amendments/gate_epuf_fill_dev_k_sweep.py new file mode 100644 index 00000000..d8c7bbfe --- /dev/null +++ b/docs/amendments/gate_epuf_fill_dev_k_sweep.py @@ -0,0 +1,225 @@ +"""DEV-only, post hoc: sensitivity of gate_epuf_fill's verdicts to K. + +Shown to Max before he ruled on d927 (ratify K = 1), and disclosed in the +DEV log for that reason. K = 1 was registered at 14045be4, before any DEV +score; this sweep does not change it. + +Reads only gate_epuf_fill_dev_registered_dryrun.json (the registered TEST +procedure dry-run on DEV, with every cell) and the registered floor build +(hash-checked via epuf_fill_scoring.load_registered_floors). No EPUF +microdata, no TEST. Run from the repository root with PYTHONPATH=src; +its output is gate_epuf_fill_dev_k_sweep.txt. + +For each K, every gating cell's tolerance is K times the registered +(K=1) tolerance, and each recorded score is re-scored with the +repository's own epuf_fill_gate.score / adoption_tier / adopt and +epuf_fill_scoring.combined_current. + +Reconstruction: score() takes truth CellValues and per-draw filled +CellValues and uses only .value (epuf_fill_gate.py:1015-1053). The dry run +recorded each cell's truth and the mean filled value over draws, so one +"draw" equal to the recorded mean reproduces the recorded gap exactly +(the mean of one value is that value; gap() is recomputed from it). +""" + +import json +import math +import sys +from pathlib import Path + +import numpy as np + +from populace_dynamics.harness import epuf_fill_gate as g +from populace_dynamics.harness import epuf_fill_scoring as scoring +from populace_dynamics.harness.epuf_cells import CellValue + +DRY = Path("docs/amendments/gate_epuf_fill_dev_registered_dryrun.json") +KS = (0.5, 0.75, 1.0, 1.25, 1.5, 2.0, 2.5, 3.0) +GRID = np.round(np.arange(0.25, 4.0001, 0.005), 3) + + +def num(x): + return float(x) if not isinstance(x, str) else float(x) # "inf"/"nan" + + +rec = json.loads(DRY.read_text()) +floors = scoring.load_registered_floors() # SHA-256 checked +assert rec["floors_sha256"] == floors["sha256"] +assert g.K_TOLERANCE == 1.0 + +fams = {} +for fam, e in rec["families"].items(): + truth = { + c: CellValue(num(v["value"]), int(v["events"]), int(v["events"])) + for c, v in e["truth"].items() + } + gating = [ + c + for c in floors["gating"] + if c.startswith(f"{fam}.") and c not in e["dropped_undefined_truth"] + ] + rec_gating = sorted( + c for c, r in e["primary"]["cells"].items() if "tolerance" in r + ) + assert rec_gating == sorted(gating), fam + for c in gating: # recorded tolerances are the floor build's (K=1) + assert e["primary"]["cells"][c]["tolerance"] == floors["tolerance"][c] + + def draws(block): + return [ + {c: CellValue(num(r["filled"]), 0, 0) for c, r in block.items()} + ] + + roles = {r: draws(e[r]["cells"]) for r in ("primary", "alternative")} + readings = {k: draws(v["cells"]) for k, v in e["current_rule"].items()} + fams[fam] = dict( + e=e, truth=truth, gating=gating, roles=roles, readings=readings + ) + + +def run(fam, k): + f = fams[fam] + tol = {c: k * floors["tolerance"][c] for c in f["gating"]} + sc = lambda d: g.score(f["truth"], d, tol, f["gating"]) # noqa: E731 + cur = {n: sc(d) for n, d in f["readings"].items()} + ref = scoring.combined_current(*cur.values()) + out = { + "current": {n: s["n_failing"] for n, s in cur.items()}, + "ref_failing": ref["n_failing"], + "scores": {}, + } + tiers = {} + for role, d in f["roles"].items(): + s = sc(d) + tiers[role] = g.adoption_tier(s, ref) + out["scores"][role] = s + out[role] = {"n_failing": s["n_failing"], "tier": tiers[role]} + out["adopted"] = g.adopt(tiers["primary"], tiers["alternative"]) + return out + + +# ---- K = 1 must reproduce the record exactly -------------------------------- +mismatch = [] +for fam, f in fams.items(): + e, r = f["e"], run(fam, 1.0) + if r["adopted"] != e["adopted"]: + mismatch.append((fam, "adopted", r["adopted"], e["adopted"])) + for role in ("primary", "alternative"): + for key in ("n_failing", "tier"): + if r[role][key] != e[role][key]: + mismatch.append((fam, role, key, r[role][key], e[role][key])) + for c, row in r["scores"][role]["cells"].items(): + want = e[role]["cells"][c] + same_gap = (row["gap"] == num(want["gap"])) or ( + math.isnan(row["gap"]) and math.isnan(num(want["gap"])) + ) + if not same_gap or row.get("passes") != want.get("passes"): + mismatch.append((fam, role, c)) + for n, v in r["current"].items(): + if v != e["current_rule"][n]["n_failing"]: + mismatch.append( + (fam, "current", n, v, e["current_rule"][n]["n_failing"]) + ) +print("K=1 reproduction mismatches:", mismatch or "none") +if mismatch: + sys.exit(1) + +# ---- table ------------------------------------------------------------------ +print( + "\n| K | odd primary | odd alt | odd cur fb/2s | odd adopted " + "| pre primary | pre alt | pre cur | pre adopted |" +) +print("|---|---|---|---|---|---|---|---|---|") +for k in KS: + o, p = run("odd", k), run("pre", k) + cell = lambda r, role: f"{r[role]['n_failing']} {r[role]['tier']}" # noqa + print( + f"| {k} | {cell(o,'primary')} | {cell(o,'alternative')} | " + f"{o['current']['fallback']}/{o['current']['two_sided']} | " + f"{o['adopted']} | {cell(p,'primary')} | {cell(p,'alternative')} " + f"| {p['current']['fallback']} | {p['adopted']} |" + ) + +# ---- odd primary worst cells, |gap| / sigma (sigma = K=1 tolerance) --------- +e = fams["odd"]["e"] +ratios = sorted( + ( + (abs(num(r["gap"])) / r["tolerance"], c) + for c, r in e["primary"]["cells"].items() + if "tolerance" in r + ), + reverse=True, +) +print("\nodd primary worst 10 |gap|/sigma:") +for x, c in ratios[:10]: + print(f" {x:.3f} {c}") +kmin = ratios[0][0] +print(f"smallest K certifying odd primary = max |gap|/sigma = {kmin:.4f}") +r = run("odd", kmin) +print( + " check at that K:", + r["primary"], + "; at K-1e-9:", + run("odd", kmin - 1e-9)["primary"], +) + +# ---- fine grid: tier transitions ------------------------------------------- +print("\ntransitions on K grid 0.25..4.0 step 0.005:") +for fam in ("odd", "pre"): + prev = None + for k in GRID: + r = run(fam, float(k)) + state = (r["primary"]["tier"], r["alternative"]["tier"], r["adopted"]) + if state != prev: + print( + f" {fam} K>={k}: primary={state[0]} " + f"(fail {r['primary']['n_failing']}), " + f"alt={state[1]} (fail {r['alternative']['n_failing']}), " + f"cur={r['current']}, adopted={state[2]}" + ) + prev = state + +# odd primary: which cells block "improves" below K=1 +for k in (0.5, 0.75): + f = fams["odd"] + tol = {c: k * floors["tolerance"][c] for c in f["gating"]} + cur = [ + g.score(f["truth"], d, tol, f["gating"]) + for d in f["readings"].values() + ] + ref = scoring.combined_current(*cur) + s = g.score(f["truth"], f["roles"]["primary"], tol, f["gating"]) + blockers = [] + for c, row in s["cells"].items(): + if "passes" not in row: + continue + b = float(ref["cells"][c]["gap"]) + b = abs(b) if np.isfinite(b) else np.inf + allowed = max( + row["tolerance"], min(b, g.IMPROVES_CAP * row["tolerance"]) + ) + if abs(row["gap"]) > allowed: + blockers.append( + ( + c, + round(abs(row["gap"]) / floors["tolerance"][c], 3), + round(b / floors["tolerance"][c], 3), + ) + ) + print( + f"\nodd primary improves-blockers at K={k} " + f"(cell, |gap|/sigma, |cur gap|/sigma): {blockers}" + ) + +# pre primary: max |gap|/sigma (headroom of the certification) +pr = sorted( + ( + (abs(num(r["gap"])) / r["tolerance"], c) + for c, r in fams["pre"]["e"]["primary"]["cells"].items() + if "tolerance" in r + ), + reverse=True, +) +print( + "\npre primary worst 5 |gap|/sigma:", [(round(x, 3), c) for x, c in pr[:5]] +) diff --git a/docs/amendments/gate_epuf_fill_dev_k_sweep.txt b/docs/amendments/gate_epuf_fill_dev_k_sweep.txt new file mode 100644 index 00000000..d75f9299 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_dev_k_sweep.txt @@ -0,0 +1,41 @@ +K=1 reproduction mismatches: none + +| K | odd primary | odd alt | odd cur fb/2s | odd adopted | pre primary | pre alt | pre cur | pre adopted | +|---|---|---|---|---|---|---|---|---| +| 0.5 | 29 not_adopted | 65 not_adopted | 103/101 | None | 6 improves | 74 not_adopted | 136 | primary | +| 0.75 | 14 not_adopted | 53 not_adopted | 101/100 | None | 1 improves | 63 not_adopted | 132 | primary | +| 1.0 | 7 improves | 46 not_adopted | 100/99 | primary | 0 certified | 50 not_adopted | 131 | primary | +| 1.25 | 4 improves | 43 not_adopted | 98/99 | primary | 0 certified | 42 not_adopted | 126 | primary | +| 1.5 | 3 improves | 39 not_adopted | 97/98 | primary | 0 certified | 39 not_adopted | 123 | primary | +| 2.0 | 2 improves | 29 not_adopted | 96/97 | primary | 0 certified | 29 not_adopted | 112 | primary | +| 2.5 | 1 improves | 22 not_adopted | 91/93 | primary | 0 certified | 23 not_adopted | 98 | primary | +| 3.0 | 0 certified | 19 not_adopted | 85/88 | primary | 0 certified | 18 not_adopted | 88 | primary | + +odd primary worst 10 |gap|/sigma: + 2.690 odd.men.a22_29.r1 + 2.019 odd.women.a22_29.r1 + 1.585 odd.women.a22_29.zint + 1.365 odd.men.a22_29.zint + 1.232 odd.women.a22_74.zint + 1.170 odd.women.a60_74.r1 + 1.093 odd.women.a22_74.r1 + 0.969 odd.men.a22_74.r1 + 0.928 odd.men.a22_29.r3 + 0.858 odd.women.a22_29.wint +smallest K certifying odd primary = max |gap|/sigma = 2.6903 + check at that K: {'n_failing': 0, 'tier': 'certified'} ; at K-1e-9: {'n_failing': 1, 'tier': 'improves'} + +transitions on K grid 0.25..4.0 step 0.005: + odd K>=0.25: primary=not_adopted (fail 49), alt=not_adopted (fail 81), cur={'fallback': 108, 'two_sided': 104}, adopted=None + odd K>=0.9: primary=improves (fail 9), alt=not_adopted (fail 48), cur={'fallback': 100, 'two_sided': 99}, adopted=primary + odd K>=2.695: primary=certified (fail 0), alt=not_adopted (fail 20), cur={'fallback': 87, 'two_sided': 90}, adopted=primary + odd K>=3.415: primary=certified (fail 0), alt=improves (fail 14), cur={'fallback': 81, 'two_sided': 86}, adopted=primary + pre K>=0.25: primary=not_adopted (fail 34), alt=not_adopted (fail 97), cur={'fallback': 136}, adopted=None + pre K>=0.26: primary=improves (fail 32), alt=not_adopted (fail 96), cur={'fallback': 136}, adopted=primary + pre K>=0.77: primary=certified (fail 0), alt=not_adopted (fail 60), cur={'fallback': 132}, adopted=primary + +odd primary improves-blockers at K=0.5 (cell, |gap|/sigma, |cur gap|/sigma): [('odd.men.a22_29.r1', 2.69, 11.583), ('odd.women.a22_29.r1', 2.019, 13.643), ('odd.women.a22_29.zint', 1.585, inf), ('odd.women.a60_74.r3', 0.651, 0.617)] + +odd primary improves-blockers at K=0.75 (cell, |gap|/sigma, |cur gap|/sigma): [('odd.men.a22_29.r1', 2.69, 11.583)] + +pre primary worst 5 |gap|/sigma: [(0.769, 'pre.men.b1946_1980.yzero'), (0.723, 'pre.women.b1956_1965.yr_cross'), (0.718, 'pre.women.b1946_1980.yzero'), (0.657, 'pre.men.b1956_1965.yzero'), (0.607, 'pre.men.b1946_1980.ylevel')] diff --git a/docs/amendments/gate_epuf_fill_dev_registered_dryrun.json b/docs/amendments/gate_epuf_fill_dev_registered_dryrun.json new file mode 100644 index 00000000..94ca4427 --- /dev/null +++ b/docs/amendments/gate_epuf_fill_dev_registered_dryrun.json @@ -0,0 +1 @@ +{"registration_id": "2026-10-03-epuf-career-fill", "floors_sha256": "d403a824416f00524fadceefb897f5bdcaa197c12ebee0b1f1fd52304d98e25e", "constants_sha256": "cabb7611fd088bffc94fb8e1ee5d454df2cb8afbfbf909a6ed837c471fb4e04a", "candidates": {"odd": {"primary": {"path": "/Users/maxghenis/PolicyEngine/epuf-data/fills/odd_forest_v1.npz", "sha256": "37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa"}, "alternative": {"path": "/Users/maxghenis/PolicyEngine/epuf-data/fills/odd_knn_v1.npz", "sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291"}}, "pre": {"alternative": {"path": "/Users/maxghenis/PolicyEngine/epuf-data/fills/pre_chain_v1.npz", "sha256": "8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724"}, "primary": {"path": "/Users/maxghenis/PolicyEngine/epuf-data/fills/pre_donor_v1.npz", "sha256": "3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7"}}}, "n_persons": 875829, "seeds": [7100, 7101, 7102, 7103, 7104, 7105, 7106, 7107, 7108, 7109, 7110, 7111, 7112, 7113, 7114, 7115, 7116, 7117, 7118, 7119], "families": {"odd": {"dropped_undefined_truth": [], "truth": {"odd.men.a22_29.r1": {"value": 0.8114342592359591, "events": 24883}, "odd.men.a22_29.r3": {"value": 0.6020834292042946, "events": 23658}, "odd.men.a22_29.r2": {"value": 0.7090594261869381, "events": 24265}, "odd.men.a22_29.r4": {"value": 0.5990582702554191, "events": 23718}, "odd.men.a22_29.zint": {"value": 0.027589420573962787, "events": 3434}, "odd.men.a22_29.zexit": {"value": 0.4099387400566883, "events": 8967}, "odd.men.a22_29.wint": {"value": 0.08288543140028289, "events": 1172}, "odd.men.a22_29.atcap": {"value": 0.014676604027739744, "events": 1983}, "odd.men.a22_29.level": {"value": 0.23657185599421368, "events": 135113}, "odd.men.a22_29.q10": {"value": 0.040229885057471264, "events": 135113}, "odd.men.a22_29.q50": {"value": 0.24222222222222223, "events": 135113}, "odd.men.a22_29.q90": {"value": 0.5661157024793388, "events": 135113}, "odd.men.a30_44.r1": {"value": 0.9000861304840623, "events": 51112}, "odd.men.a30_44.r3": {"value": 0.8030903893396322, "events": 50209}, "odd.men.a30_44.r2": {"value": 0.8479233224100629, "events": 52027}, "odd.men.a30_44.r4": {"value": 0.7869108820616492, "events": 52730}, "odd.men.a30_44.zint": {"value": 0.017933882760031997, "events": 4820}, "odd.men.a30_44.zexit": {"value": 0.4342024282437196, "events": 15521}, "odd.men.a30_44.wint": {"value": 0.07239145006330527, "events": 1944}, "odd.men.a30_44.atcap": {"value": 0.10574805846620577, "events": 30256}, "odd.men.a30_44.level": {"value": 0.3982899917120792, "events": 286114}, "odd.men.a30_44.q10": {"value": 0.08706467661691543, "events": 286114}, "odd.men.a30_44.q50": {"value": 0.41333333333333333, "events": 286114}, "odd.men.a30_44.q90": {"value": 1.0, "events": 286114}, "odd.men.a45_59.r1": {"value": 0.9077661818197654, "events": 35553}, "odd.men.a45_59.r3": {"value": 0.8146289743583418, "events": 33477}, "odd.men.a45_59.r2": {"value": 0.850421444941529, "events": 34460}, "odd.men.a45_59.r4": {"value": 0.7598303885400162, "events": 32366}, "odd.men.a45_59.zint": {"value": 0.015835938476928848, "events": 3166}, "odd.men.a45_59.zexit": {"value": 0.45185706741770815, "events": 12835}, "odd.men.a45_59.wint": {"value": 0.05825168693530555, "events": 1528}, "odd.men.a45_59.atcap": {"value": 0.15826463477931516, "events": 33846}, "odd.men.a45_59.level": {"value": 0.4279852808128974, "events": 213857}, "odd.men.a45_59.q10": {"value": 0.08868501529051988, "events": 213857}, "odd.men.a45_59.q50": {"value": 0.4701492537313433, "events": 213857}, "odd.men.a45_59.q90": {"value": 1.0, "events": 213857}, "odd.men.a60_74.r1": {"value": 0.8712782553777247, "events": 9471}, "odd.men.a60_74.r3": {"value": 0.7283746799072385, "events": 7450}, "odd.men.a60_74.r2": {"value": 0.7888015745113253, "events": 8329}, "odd.men.a60_74.r4": {"value": 0.6781549627202678, "events": 6609}, "odd.men.a60_74.zint": {"value": 0.027132152458672565, "events": 1423}, "odd.men.a60_74.zexit": {"value": 0.5047452182800409, "events": 10176}, "odd.men.a60_74.wint": {"value": 0.03084255319148936, "events": 906}, "odd.men.a60_74.atcap": {"value": 0.10024796315975912, "events": 6226}, "odd.men.a60_74.level": {"value": 0.20453937198442498, "events": 62106}, "odd.men.a60_74.q10": {"value": 0.01529051987767584, "events": 62106}, "odd.men.a60_74.q50": {"value": 0.21666666666666667, "events": 62106}, "odd.men.a60_74.q90": {"value": 1.0, "events": 62106}, "odd.men.a22_74.r1": {"value": 0.8935990156167083, "events": 127726}, "odd.men.a22_74.r3": {"value": 0.7806492619651291, "events": 121284}, "odd.men.a22_74.r2": {"value": 0.8291027164330249, "events": 124053}, "odd.men.a22_74.r4": {"value": 0.7414979369803346, "events": 118165}, "odd.men.a22_74.zint": {"value": 0.019892968610837895, "events": 12843}, "odd.men.a22_74.zexit": {"value": 0.44752843148294114, "events": 47694}, "odd.men.a22_74.wint": {"value": 0.05745341614906832, "events": 5550}, "odd.men.a22_74.atcap": {"value": 0.10371778137953785, "events": 72311}, "odd.men.a22_74.level": {"value": 0.35325148977531445, "events": 697190}, "odd.men.a22_74.q10": {"value": 0.06111111111111111, "events": 697190}, "odd.men.a22_74.q50": {"value": 0.37052341597796146, "events": 697190}, "odd.men.a22_74.q90": {"value": 1.0, "events": 697190}, "odd.men.b1936_1940.aime_p10": {"value": 346.0, "events": 13032}, "odd.men.b1936_1940.aime_p25": {"value": 1131.0, "events": 13032}, "odd.men.b1936_1940.aime_p50": {"value": 2430.0, "events": 13032}, "odd.men.b1936_1940.aime_p75": {"value": 3582.0, "events": 13032}, "odd.men.b1936_1940.aime_p90": {"value": 4382.0, "events": 13032}, "odd.men.b1941_1945.aime_p10": {"value": 338.0, "events": 15907}, "odd.men.b1941_1945.aime_p25": {"value": 1174.25, "events": 15907}, "odd.men.b1941_1945.aime_p50": {"value": 2821.5, "events": 15907}, "odd.men.b1941_1945.aime_p75": {"value": 4421.0, "events": 15907}, "odd.men.b1941_1945.aime_p90": {"value": 5528.300000000001, "events": 15907}, "odd.men.b1936_1945.aime_p10": {"value": 342.8000000000002, "events": 28939}, "odd.men.b1936_1945.aime_p25": {"value": 1152.0, "events": 28939}, "odd.men.b1936_1945.aime_p50": {"value": 2625.0, "events": 28939}, "odd.men.b1936_1945.aime_p75": {"value": 3991.5, "events": 28939}, "odd.men.b1936_1945.aime_p90": {"value": 5075.0, "events": 28939}, "odd.men.b1946_1955.paime_p10": {"value": 308.0, "events": 44074}, "odd.men.b1946_1955.paime_p25": {"value": 1026.0, "events": 44074}, "odd.men.b1946_1955.paime_p50": {"value": 2614.0, "events": 44074}, "odd.men.b1946_1955.paime_p75": {"value": 4390.0, "events": 44074}, "odd.men.b1946_1955.paime_p90": {"value": 5810.0, "events": 44074}, "odd.men.b1956_1965.paime_p10": {"value": 253.0, "events": 49852}, "odd.men.b1956_1965.paime_p25": {"value": 765.0, "events": 49852}, "odd.men.b1956_1965.paime_p50": {"value": 1764.0, "events": 49852}, "odd.men.b1956_1965.paime_p75": {"value": 2958.0, "events": 49852}, "odd.men.b1956_1965.paime_p90": {"value": 4111.0, "events": 49852}, "odd.men.b1966_1980.paime_p10": {"value": 112.0, "events": 61607}, "odd.men.b1966_1980.paime_p25": {"value": 303.25, "events": 61607}, "odd.men.b1966_1980.paime_p50": {"value": 655.0, "events": 61607}, "odd.men.b1966_1980.paime_p75": {"value": 1225.0, "events": 61607}, "odd.men.b1966_1980.paime_p90": {"value": 1888.9000000000015, "events": 61607}, "odd.men.b1946_1980.paime_p10": {"value": 175.0, "events": 155533}, "odd.men.b1946_1980.paime_p25": {"value": 487.0, "events": 155533}, "odd.men.b1946_1980.paime_p50": {"value": 1245.0, "events": 155533}, "odd.men.b1946_1980.paime_p75": {"value": 2644.0, "events": 155533}, "odd.men.b1946_1980.paime_p90": {"value": 4310.0, "events": 155533}, "odd.women.a22_29.r1": {"value": 0.7942016511246733, "events": 22841}, "odd.women.a22_29.r3": {"value": 0.5535614164842262, "events": 21369}, "odd.women.a22_29.r2": {"value": 0.6731585499668502, "events": 22004}, "odd.women.a22_29.r4": {"value": 0.552312691565311, "events": 21259}, "odd.women.a22_29.zint": {"value": 0.0315547703180212, "events": 3572}, "odd.women.a22_29.zexit": {"value": 0.41757828810020875, "events": 10001}, "odd.women.a22_29.wint": {"value": 0.08501885498800137, "events": 1240}, "odd.women.a22_29.atcap": {"value": 0.005728386357627567, "events": 715}, "odd.women.a22_29.level": {"value": 0.1854665923300802, "events": 124817}, "odd.women.a22_29.q10": {"value": 0.028735632183908046, "events": 124817}, "odd.women.a22_29.q50": {"value": 0.191131498470948, "events": 124817}, "odd.women.a22_29.q90": {"value": 0.46005509641873277, "events": 124817}, "odd.women.a30_44.r1": {"value": 0.888241419643356, "events": 44822}, "odd.women.a30_44.r3": {"value": 0.7685608654079126, "events": 42936}, "odd.women.a30_44.r2": {"value": 0.8242277470219269, "events": 44797}, "odd.women.a30_44.r4": {"value": 0.7490435327097834, "events": 44806}, "odd.women.a30_44.zint": {"value": 0.022222222222222223, "events": 5098}, "odd.women.a30_44.zexit": {"value": 0.45066799061202384, "events": 19970}, "odd.women.a30_44.wint": {"value": 0.06073485056210584, "events": 2215}, "odd.women.a30_44.atcap": {"value": 0.034619662054697874, "events": 8685}, "odd.women.a30_44.level": {"value": 0.2579924798566703, "events": 250869}, "odd.women.a30_44.q10": {"value": 0.04228855721393035, "events": 250869}, "odd.women.a30_44.q50": {"value": 0.26859504132231404, "events": 250869}, "odd.women.a30_44.q90": {"value": 0.6716417910447762, "events": 250869}, "odd.women.a45_59.r1": {"value": 0.9101006329634369, "events": 31760}, "odd.women.a45_59.r3": {"value": 0.8096130207660378, "events": 29595}, "odd.women.a45_59.r2": {"value": 0.8498173267452248, "events": 30540}, "odd.women.a45_59.r4": {"value": 0.7620167777938601, "events": 28452}, "odd.women.a45_59.zint": {"value": 0.01632776600720909, "events": 2967}, "odd.women.a45_59.zexit": {"value": 0.4693136110029843, "events": 14468}, "odd.women.a45_59.wint": {"value": 0.050115932427956277, "events": 1513}, "odd.women.a45_59.atcap": {"value": 0.04053483605515179, "events": 7970}, "odd.women.a45_59.level": {"value": 0.28104586423928846, "events": 196621}, "odd.women.a45_59.q10": {"value": 0.053516819571865444, "events": 196621}, "odd.women.a45_59.q50": {"value": 0.29338842975206614, "events": 196621}, "odd.women.a45_59.q90": {"value": 0.7300275482093664, "events": 196621}, "odd.women.a60_74.r1": {"value": 0.8610627385333514, "events": 6985}, "odd.women.a60_74.r3": {"value": 0.7141078828415282, "events": 5282}, "odd.women.a60_74.r2": {"value": 0.7722738686836694, "events": 6041}, "odd.women.a60_74.r4": {"value": 0.6501204938024134, "events": 4624}, "odd.women.a60_74.zint": {"value": 0.022290698541288734, "events": 897}, "odd.women.a60_74.zexit": {"value": 0.5155483759303063, "events": 8397}, "odd.women.a60_74.wint": {"value": 0.023107482596493787, "events": 634}, "odd.women.a60_74.atcap": {"value": 0.01812919896640827, "events": 877}, "odd.women.a60_74.level": {"value": 0.12926943691615103, "events": 48375}, "odd.women.a60_74.q10": {"value": 0.017412935323383085, "events": 48375}, "odd.women.a60_74.q50": {"value": 0.15517241379310345, "events": 48375}, "odd.women.a60_74.q90": {"value": 0.5360696517412935, "events": 48375}, "odd.women.a22_74.r1": {"value": 0.8812684347612898, "events": 110453}, "odd.women.a22_74.r3": {"value": 0.7468316241328095, "events": 103479}, "odd.women.a22_74.r2": {"value": 0.8056395091372208, "events": 106291}, "odd.women.a22_74.r4": {"value": 0.7116744894807431, "events": 100276}, "odd.women.a22_74.zint": {"value": 0.022201124403524123, "events": 12534}, "odd.women.a22_74.zexit": {"value": 0.45845752128015943, "events": 53375}, "odd.women.a22_74.wint": {"value": 0.051544874036178946, "events": 5602}, "odd.women.a22_74.atcap": {"value": 0.029398307023564402, "events": 18247}, "odd.women.a22_74.level": {"value": 0.2372854094489719, "events": 620682}, "odd.women.a22_74.q10": {"value": 0.03777777777777778, "events": 620682}, "odd.women.a22_74.q50": {"value": 0.24875621890547264, "events": 620682}, "odd.women.a22_74.q90": {"value": 0.6444444444444445, "events": 620682}, "odd.women.b1936_1940.aime_p10": {"value": 82.29999999999995, "events": 11967}, "odd.women.b1936_1940.aime_p25": {"value": 287.75, "events": 11967}, "odd.women.b1936_1940.aime_p50": {"value": 786.0, "events": 11967}, "odd.women.b1936_1940.aime_p75": {"value": 1566.0, "events": 11967}, "odd.women.b1936_1940.aime_p90": {"value": 2438.7000000000007, "events": 11967}, "odd.women.b1941_1945.aime_p10": {"value": 115.0, "events": 15070}, "odd.women.b1941_1945.aime_p25": {"value": 399.0, "events": 15070}, "odd.women.b1941_1945.aime_p50": {"value": 1077.0, "events": 15070}, "odd.women.b1941_1945.aime_p75": {"value": 2116.0, "events": 15070}, "odd.women.b1941_1945.aime_p90": {"value": 3268.6000000000004, "events": 15070}, "odd.women.b1936_1945.aime_p10": {"value": 98.0, "events": 27037}, "odd.women.b1936_1945.aime_p25": {"value": 345.0, "events": 27037}, "odd.women.b1936_1945.aime_p50": {"value": 932.0, "events": 27037}, "odd.women.b1936_1945.aime_p75": {"value": 1865.0, "events": 27037}, "odd.women.b1936_1945.aime_p90": {"value": 2920.0, "events": 27037}, "odd.women.b1946_1955.paime_p10": {"value": 158.0, "events": 41669}, "odd.women.b1946_1955.paime_p25": {"value": 519.0, "events": 41669}, "odd.women.b1946_1955.paime_p50": {"value": 1313.0, "events": 41669}, "odd.women.b1946_1955.paime_p75": {"value": 2486.0, "events": 41669}, "odd.women.b1946_1955.paime_p90": {"value": 3780.0, "events": 41669}, "odd.women.b1956_1965.paime_p10": {"value": 140.0, "events": 46740}, "odd.women.b1956_1965.paime_p25": {"value": 426.0, "events": 46740}, "odd.women.b1956_1965.paime_p50": {"value": 1000.0, "events": 46740}, "odd.women.b1956_1965.paime_p75": {"value": 1842.0, "events": 46740}, "odd.women.b1956_1965.paime_p90": {"value": 2793.0, "events": 46740}, "odd.women.b1966_1980.paime_p10": {"value": 80.0, "events": 58000}, "odd.women.b1966_1980.paime_p25": {"value": 211.0, "events": 58000}, "odd.women.b1966_1980.paime_p50": {"value": 463.0, "events": 58000}, "odd.women.b1966_1980.paime_p75": {"value": 866.75, "events": 58000}, "odd.women.b1966_1980.paime_p90": {"value": 1387.0, "events": 58000}, "odd.women.b1946_1980.paime_p10": {"value": 109.0, "events": 146409}, "odd.women.b1946_1980.paime_p25": {"value": 309.0, "events": 146409}, "odd.women.b1946_1980.paime_p50": {"value": 758.0, "events": 146409}, "odd.women.b1946_1980.paime_p75": {"value": 1590.0, "events": 146409}, "odd.women.b1946_1980.paime_p90": {"value": 2696.0, "events": 146409}}, "current_rule": {"fallback": {"passes": false, "n_gating": 183, "n_failing": 100, "cells": {"odd.men.a22_29.atcap": {"truth": 0.014676604027739744, "filled": 0.004275744117403658, "gap": -1.2332965133723093, "seed_sd": 0.0, "tolerance": 0.18023370223284765, "passes": false}, "odd.men.a22_29.level": {"truth": 0.23657185599421368, "filled": 0.2422219209391976, "gap": 0.02360234199005551, "seed_sd": 0.0, "tolerance": 0.02171146931539837, "passes": false}, "odd.men.a22_29.q10": {"truth": 0.040229885057471264, "filled": 0.038505747126436785, "gap": -0.04380262265839274, "seed_sd": 0.0, "tolerance": 0.07329266358707544, "passes": true}, "odd.men.a22_29.q50": {"truth": 0.24222222222222223, "filled": 0.231651376146789, "gap": -0.04462202597255316, "seed_sd": 0.0, "tolerance": 0.022980444394274636, "passes": false}, "odd.men.a22_29.q90": {"truth": 0.5661157024793388, "filled": 0.54, "gap": -0.04722933909525551, "seed_sd": 0.0, "tolerance": 0.02353059621893112, "passes": false}, "odd.men.a22_29.r1": {"truth": 0.8114342592359591, "filled": 0.8956863944371684, "gap": 0.08425213520120922, "seed_sd": 0.0, "tolerance": 0.007273490161268873, "passes": false}, "odd.men.a22_29.r2": {"truth": 0.7090594261869381, "filled": 0.863164005737995, "gap": 0.15410457955105694, "seed_sd": 0.0, "tolerance": 0.014356564488812985, "passes": false}, "odd.men.a22_29.r3": {"truth": 0.6020834292042946, "filled": 0.6369726646985666, "gap": 0.03488923549427203, "seed_sd": 0.0, "tolerance": 0.013280307041936976, "passes": false}, "odd.men.a22_29.r4": {"truth": 0.5990582702554191, "filled": 0.6928618295102605, "gap": 0.0938035592548414, "seed_sd": 0.0, "tolerance": 0.02119963449400435, "passes": false}, "odd.men.a22_29.wint": {"truth": 0.08288543140028289, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.men.a22_29.zexit": {"truth": 0.4099387400566883, "filled": 0.06116851056048277, "gap": -1.9023752102638736, "seed_sd": 0.0, "tolerance": 0.052311207438163365, "passes": false}, "odd.men.a22_29.zint": {"truth": 0.027589420573962787, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.12140160693630424, "passes": false}, "odd.men.a22_74.atcap": {"truth": 0.10371778137953785, "filled": 0.0461710166893302, "gap": -0.8093213131248014, "seed_sd": 0.0, "tolerance": 0.03690628859850606, "passes": false}, "odd.men.a22_74.level": {"truth": 0.35325148977531445, "filled": 0.3545980146560176, "gap": 0.003804555902007456, "seed_sd": 0.0, "tolerance": 0.01106822547283481, "passes": true}, "odd.men.a22_74.q10": {"truth": 0.06111111111111111, "filled": 0.04700413223140496, "gap": -0.26245818322783254, "seed_sd": 0.0, "tolerance": 0.03462864626687699, "passes": false}, "odd.men.a22_74.q50": {"truth": 0.37052341597796146, "filled": 0.3390804597701149, "gap": -0.08867922008585272, "seed_sd": 0.0, "tolerance": 0.012463773427145929, "passes": false}, "odd.men.a22_74.q90": {"truth": 1.0, "filled": 0.9277777777777778, "gap": -0.07496303847345523, "seed_sd": 0.0, "tolerance": 0.003451823067723808, "passes": false}, "odd.men.a22_74.r1": {"truth": 0.8935990156167083, "filled": 0.9477352806293625, "gap": 0.05413626501265423, "seed_sd": 0.0, "tolerance": 0.00258049589094515, "passes": false}, "odd.men.a22_74.r2": {"truth": 0.8291027164330249, "filled": 0.9096502660547399, "gap": 0.08054754962171495, "seed_sd": 0.0, "tolerance": 0.004576681103770273, "passes": false}, "odd.men.a22_74.r3": {"truth": 0.7806492619651291, "filled": 0.7999851781213285, "gap": 0.01933591615619945, "seed_sd": 0.0, "tolerance": 0.004722018684826323, "passes": false}, "odd.men.a22_74.r4": {"truth": 0.7414979369803346, "filled": 0.7846725429897913, "gap": 0.04317460600945666, "seed_sd": 0.0, "tolerance": 0.007000982741779954, "passes": false}, "odd.men.a22_74.wint": {"truth": 0.05745341614906832, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07337302886151653, "passes": false}, "odd.men.a22_74.zexit": {"truth": 0.44752843148294114, "filled": 0.01255489246706452, "gap": -3.573629642112994, "seed_sd": 0.0, "tolerance": 0.01916127382099423, "passes": false}, "odd.men.a22_74.zint": {"truth": 0.019892968610837895, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.05481211657295162, "passes": false}, "odd.men.a30_44.atcap": {"truth": 0.10574805846620577, "filled": 0.04650078322293776, "gap": -0.82159030214472, "seed_sd": 0.0, "tolerance": 0.049095580366705964, "passes": false}, "odd.men.a30_44.level": {"truth": 0.3982899917120792, "filled": 0.39901752453379963, "gap": 0.001824974702530513, "seed_sd": 0.0, "tolerance": 0.015220021463125325, "passes": true}, "odd.men.a30_44.q10": {"truth": 0.08706467661691543, "filled": 0.06778606965174129, "gap": -0.25029454038016086, "seed_sd": 0.0, "tolerance": 0.047669064141889934, "passes": false}, "odd.men.a30_44.q50": {"truth": 0.41333333333333333, "filled": 0.38685015290519875, "gap": -0.06621695367851432, "seed_sd": 0.0, "tolerance": 0.016660415997544607, "passes": false}, "odd.men.a30_44.q90": {"truth": 1.0, "filled": 0.9390547263681592, "gap": -0.06288151992994163, "seed_sd": 0.0, "tolerance": 0.0037405831290436837, "passes": false}, "odd.men.a30_44.r1": {"truth": 0.9000861304840623, "filled": 0.9541063174829294, "gap": 0.054020186998867126, "seed_sd": 0.0, "tolerance": 0.003556208950302802, "passes": false}, "odd.men.a30_44.r2": {"truth": 0.8479233224100629, "filled": 0.9269327325683842, "gap": 0.0790094101583213, "seed_sd": 0.0, "tolerance": 0.006500650570758907, "passes": false}, "odd.men.a30_44.r3": {"truth": 0.8030903893396322, "filled": 0.8276167881663179, "gap": 0.024526398826685725, "seed_sd": 0.0, "tolerance": 0.006087370619188444, "passes": false}, "odd.men.a30_44.r4": {"truth": 0.7869108820616492, "filled": 0.8327212421045639, "gap": 0.04581036004291472, "seed_sd": 0.0, "tolerance": 0.009218001294372115, "passes": false}, "odd.men.a30_44.wint": {"truth": 0.07239145006330527, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.12001713872675526, "passes": false}, "odd.men.a30_44.zexit": {"truth": 0.4342024282437196, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.031953074094816486, "passes": false}, "odd.men.a30_44.zint": {"truth": 0.017933882760031997, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.08989768311292742, "passes": false}, "odd.men.a45_59.atcap": {"truth": 0.15826463477931516, "filled": 0.07469452108789909, "gap": -0.750861791701275, "seed_sd": 0.0, "tolerance": 0.04711609868499565, "passes": false}, "odd.men.a45_59.level": {"truth": 0.4279852808128974, "filled": 0.4276293059548939, "gap": -0.0008320916544616308, "seed_sd": 0.0, "tolerance": 0.015834008317210692, "passes": true}, "odd.men.a45_59.q10": {"truth": 0.08868501529051988, "filled": 0.06592039800995025, "gap": -0.29664301471606525, "seed_sd": 0.0, "tolerance": 0.055701610600039544, "passes": false}, "odd.men.a45_59.q50": {"truth": 0.4701492537313433, "filled": 0.4345730027548209, "gap": -0.07868625928414963, "seed_sd": 0.0, "tolerance": 0.020385938992396834, "passes": false}, "odd.men.a45_59.q90": {"truth": 1.0, "filled": 0.9958677685950413, "gap": -0.004140792666031388, "seed_sd": 0.0}, "odd.men.a45_59.r1": {"truth": 0.9077661818197654, "filled": 0.9568133762839999, "gap": 0.049047194464234445, "seed_sd": 0.0, "tolerance": 0.004032009924506552, "passes": false}, "odd.men.a45_59.r2": {"truth": 0.850421444941529, "filled": 0.9210348902278979, "gap": 0.07061344528636881, "seed_sd": 0.0, "tolerance": 0.0074262368998509335, "passes": false}, "odd.men.a45_59.r3": {"truth": 0.8146289743583418, "filled": 0.8339594060387736, "gap": 0.019330431680431803, "seed_sd": 0.0, "tolerance": 0.007198396861260139, "passes": false}, "odd.men.a45_59.r4": {"truth": 0.7598303885400162, "filled": 0.7967982689055731, "gap": 0.03696788036555698, "seed_sd": 0.0, "tolerance": 0.012107705260459355, "passes": false}, "odd.men.a45_59.wint": {"truth": 0.05825168693530555, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.13344110142481172, "passes": false}, "odd.men.a45_59.zexit": {"truth": 0.45185706741770815, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.036314919811016276, "passes": false}, "odd.men.a45_59.zint": {"truth": 0.015835938476928848, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09287476169914739, "passes": false}, "odd.men.a60_74.atcap": {"truth": 0.10024796315975912, "filled": 0.038797709400772665, "gap": -0.949285539547915, "seed_sd": 0.0, "tolerance": 0.12472307183935011, "passes": false}, "odd.men.a60_74.level": {"truth": 0.20453937198442498, "filled": 0.20537657883929814, "gap": 0.004084778927416988, "seed_sd": 0.0, "tolerance": 0.0522939828282486, "passes": true}, "odd.men.a60_74.q10": {"truth": 0.01529051987767584, "filled": 0.009950248756218905, "gap": -0.4296354690359774, "seed_sd": 0.0, "tolerance": 0.20186846491463123, "passes": false}, "odd.men.a60_74.q50": {"truth": 0.21666666666666667, "filled": 0.17507645259938837, "gap": -0.21313732370234062, "seed_sd": 0.0, "tolerance": 0.07446457229432804, "passes": false}, "odd.men.a60_74.q90": {"truth": 1.0, "filled": 0.8222222222222222, "gap": -0.19574457712609536, "seed_sd": 0.0, "tolerance": 0.04069424979546094, "passes": false}, "odd.men.a60_74.r1": {"truth": 0.8712782553777247, "filled": 0.9364895298673185, "gap": 0.06521127448959374, "seed_sd": 0.0, "tolerance": 0.009750532797445479, "passes": false}, "odd.men.a60_74.r2": {"truth": 0.7888015745113253, "filled": 0.8691602172532571, "gap": 0.08035864274193183, "seed_sd": 0.0, "tolerance": 0.020757725353477027, "passes": false}, "odd.men.a60_74.r3": {"truth": 0.7283746799072385, "filled": 0.7453814856416601, "gap": 0.017006805734421593, "seed_sd": 0.0, "tolerance": 0.02064690204069302, "passes": true}, "odd.men.a60_74.r4": {"truth": 0.6781549627202678, "filled": 0.6972387280502411, "gap": 0.019083765329973357, "seed_sd": 0.0, "tolerance": 0.03820786124425089, "passes": true}, "odd.men.a60_74.wint": {"truth": 0.03084255319148936, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.men.a60_74.zexit": {"truth": 0.5047452182800409, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04434932710092839, "passes": false}, "odd.men.a60_74.zint": {"truth": 0.027132152458672565, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.17758159695027936, "passes": false}, "odd.men.b1936_1940.aime_p10": {"truth": 346.0, "filled": 345.0, "gap": -0.0028943580263645075, "seed_sd": 0.0, "tolerance": 0.30817736367768905, "passes": true}, "odd.men.b1936_1940.aime_p25": {"truth": 1131.0, "filled": 1131.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.15436566295506535, "passes": true}, "odd.men.b1936_1940.aime_p50": {"truth": 2430.0, "filled": 2426.0, "gap": -0.0016474468305984757, "seed_sd": 0.0, "tolerance": 0.0822816707411384, "passes": true}, "odd.men.b1936_1940.aime_p75": {"truth": 3582.0, "filled": 3576.0, "gap": -0.001676446327252279, "seed_sd": 0.0, "tolerance": 0.04915519601813279, "passes": true}, "odd.men.b1936_1940.aime_p90": {"truth": 4382.0, "filled": 4372.0, "gap": -0.0022846708589838727, "seed_sd": 0.0, "tolerance": 0.03801502836217192, "passes": true}, "odd.men.b1936_1945.aime_p10": {"truth": 342.8000000000002, "filled": 341.0, "gap": -0.005264709440107929, "seed_sd": 0.0, "tolerance": 0.19387427864829543, "passes": true}, "odd.men.b1936_1945.aime_p25": {"truth": 1152.0, "filled": 1152.5, "gap": 0.00043393361496679717, "seed_sd": 0.0, "tolerance": 0.09928513681983075, "passes": true}, "odd.men.b1936_1945.aime_p50": {"truth": 2625.0, "filled": 2621.0, "gap": -0.001524971702317579, "seed_sd": 0.0, "tolerance": 0.05013200755321697, "passes": true}, "odd.men.b1936_1945.aime_p75": {"truth": 3991.5, "filled": 3985.5, "gap": -0.0015043252178763566, "seed_sd": 0.0, "tolerance": 0.030282584009309735, "passes": true}, "odd.men.b1936_1945.aime_p90": {"truth": 5075.0, "filled": 5060.0, "gap": -0.0029600416284765174, "seed_sd": 0.0, "tolerance": 0.029246187918705275, "passes": true}, "odd.men.b1941_1945.aime_p10": {"truth": 338.0, "filled": 336.0, "gap": -0.005934735519814716, "seed_sd": 0.0, "tolerance": 0.2434370939708618, "passes": true}, "odd.men.b1941_1945.aime_p25": {"truth": 1174.25, "filled": 1176.25, "gap": 0.0017017659924842832, "seed_sd": 0.0, "tolerance": 0.15717605214643726, "passes": true}, "odd.men.b1941_1945.aime_p50": {"truth": 2821.5, "filled": 2818.0, "gap": -0.0012412449505694312, "seed_sd": 0.0, "tolerance": 0.06794319334096698, "passes": true}, "odd.men.b1941_1945.aime_p75": {"truth": 4421.0, "filled": 4414.0, "gap": -0.0015846070095602016, "seed_sd": 0.0, "tolerance": 0.039611439643930886, "passes": true}, "odd.men.b1941_1945.aime_p90": {"truth": 5528.300000000001, "filled": 5520.300000000001, "gap": -0.0014481475296577173, "seed_sd": 0.0, "tolerance": 0.024092287379584104, "passes": true}, "odd.men.b1946_1955.paime_p10": {"truth": 308.0, "filled": 309.0, "gap": 0.0032414939241718344, "seed_sd": 0.0, "tolerance": 0.11641286886749184, "passes": true}, "odd.men.b1946_1955.paime_p25": {"truth": 1026.0, "filled": 1028.0, "gap": 0.001947420284395207, "seed_sd": 0.0, "tolerance": 0.07923354406873329, "passes": true}, "odd.men.b1946_1955.paime_p50": {"truth": 2614.0, "filled": 2609.0, "gap": -0.0019146090474384536, "seed_sd": 0.0, "tolerance": 0.03884145918889322, "passes": true}, "odd.men.b1946_1955.paime_p75": {"truth": 4390.0, "filled": 4389.0, "gap": -0.0002278163809830147, "seed_sd": 0.0, "tolerance": 0.02464882648778213, "passes": true}, "odd.men.b1946_1955.paime_p90": {"truth": 5810.0, "filled": 5808.0, "gap": -0.00034429334132468625, "seed_sd": 0.0, "tolerance": 0.01935089604023114, "passes": true}, "odd.men.b1946_1980.paime_p10": {"truth": 175.0, "filled": 176.0, "gap": 0.005698021114636909, "seed_sd": 0.0, "tolerance": 0.05432313062129484, "passes": true}, "odd.men.b1946_1980.paime_p25": {"truth": 487.0, "filled": 489.0, "gap": 0.004098366392282671, "seed_sd": 0.0, "tolerance": 0.03064646902486962, "passes": true}, "odd.men.b1946_1980.paime_p50": {"truth": 1245.0, "filled": 1248.0, "gap": 0.002406740030565402, "seed_sd": 0.0, "tolerance": 0.025757396517852603, "passes": true}, "odd.men.b1946_1980.paime_p75": {"truth": 2644.0, "filled": 2645.0, "gap": 0.0003781433208231988, "seed_sd": 0.0, "tolerance": 0.0206417529534006, "passes": true}, "odd.men.b1946_1980.paime_p90": {"truth": 4310.0, "filled": 4310.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.01797004074133569, "passes": true}, "odd.men.b1956_1965.paime_p10": {"truth": 253.0, "filled": 253.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.10325340800222761, "passes": true}, "odd.men.b1956_1965.paime_p25": {"truth": 765.0, "filled": 767.0, "gap": 0.0026109675407202104, "seed_sd": 0.0, "tolerance": 0.06714959239176221, "passes": true}, "odd.men.b1956_1965.paime_p50": {"truth": 1764.0, "filled": 1765.0, "gap": 0.000566732800660219, "seed_sd": 0.0, "tolerance": 0.03783036836459788, "passes": true}, "odd.men.b1956_1965.paime_p75": {"truth": 2958.0, "filled": 2961.0, "gap": 0.0010136848308457402, "seed_sd": 0.0, "tolerance": 0.02582244416868233, "passes": true}, "odd.men.b1956_1965.paime_p90": {"truth": 4111.0, "filled": 4110.0, "gap": -0.00024327940759860667, "seed_sd": 0.0, "tolerance": 0.022664845011086624, "passes": true}, "odd.men.b1966_1980.paime_p10": {"truth": 112.0, "filled": 112.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.07763263566160054, "passes": true}, "odd.men.b1966_1980.paime_p25": {"truth": 303.25, "filled": 305.0, "gap": 0.00575422878325238, "seed_sd": 0.0, "tolerance": 0.04145124387911718, "passes": true}, "odd.men.b1966_1980.paime_p50": {"truth": 655.0, "filled": 661.0, "gap": 0.009118604216434179, "seed_sd": 0.0, "tolerance": 0.02859495381248372, "passes": true}, "odd.men.b1966_1980.paime_p75": {"truth": 1225.0, "filled": 1230.0, "gap": 0.00407332538763594, "seed_sd": 0.0, "tolerance": 0.024541637055534683, "passes": true}, "odd.men.b1966_1980.paime_p90": {"truth": 1888.9000000000015, "filled": 1891.0, "gap": 0.00111114062068296, "seed_sd": 0.0, "tolerance": 0.025554311796701996, "passes": true}, "odd.women.a22_29.atcap": {"truth": 0.005728386357627567, "filled": 0.0014958367106329674, "gap": -1.3427481551382767, "seed_sd": 0.0}, "odd.women.a22_29.level": {"truth": 0.1854665923300802, "filled": 0.1897248573494353, "gap": 0.022700132836240172, "seed_sd": 0.0, "tolerance": 0.018517427407013797, "passes": false}, "odd.women.a22_29.q10": {"truth": 0.028735632183908046, "filled": 0.026111111111111113, "gap": -0.09577695539376885, "seed_sd": 0.0, "tolerance": 0.0684466769976442, "passes": false}, "odd.women.a22_29.q50": {"truth": 0.191131498470948, "filled": 0.17813455657492355, "gap": -0.07042246429654586, "seed_sd": 0.0, "tolerance": 0.021849929026002995, "passes": false}, "odd.women.a22_29.q90": {"truth": 0.46005509641873277, "filled": 0.43654434250764523, "gap": -0.05245630051303818, "seed_sd": 0.0, "tolerance": 0.020751504651450387, "passes": false}, "odd.women.a22_29.r1": {"truth": 0.7942016511246733, "filled": 0.8821403084868372, "gap": 0.08793865736216389, "seed_sd": 0.0, "tolerance": 0.006445547562698701, "passes": false}, "odd.women.a22_29.r2": {"truth": 0.6731585499668502, "filled": 0.8489459186770097, "gap": 0.17578736871015954, "seed_sd": 0.0, "tolerance": 0.013111971249341917, "passes": false}, "odd.women.a22_29.r3": {"truth": 0.5535614164842262, "filled": 0.5908983271961422, "gap": 0.03733691071191603, "seed_sd": 0.0, "tolerance": 0.012054115899361516, "passes": false}, "odd.women.a22_29.r4": {"truth": 0.552312691565311, "filled": 0.656690179347997, "gap": 0.10437748778268596, "seed_sd": 0.0, "tolerance": 0.01982762523559322, "passes": false}, "odd.women.a22_29.wint": {"truth": 0.08501885498800137, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.15037337545812676, "passes": false}, "odd.women.a22_29.zexit": {"truth": 0.41757828810020875, "filled": 0.06012526096033403, "gap": -1.9380419744064699, "seed_sd": 0.0, "tolerance": 0.041287246240518535, "passes": false}, "odd.women.a22_29.zint": {"truth": 0.0315547703180212, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09342121518593281, "passes": false}, "odd.women.a22_74.atcap": {"truth": 0.029398307023564402, "filled": 0.011993248463319055, "gap": -0.8965932250573987, "seed_sd": 0.0, "tolerance": 0.06709882226274566, "passes": false}, "odd.women.a22_74.level": {"truth": 0.2372854094489719, "filled": 0.23838128106693332, "gap": 0.0046077372189554655, "seed_sd": 0.0, "tolerance": 0.010902725750872375, "passes": true}, "odd.women.a22_74.q10": {"truth": 0.03777777777777778, "filled": 0.027186544342507644, "gap": -0.3289988826632908, "seed_sd": 0.0, "tolerance": 0.031129845660342752, "passes": false}, "odd.women.a22_74.q50": {"truth": 0.24875621890547264, "filled": 0.22111111111111112, "gap": -0.11780803596888845, "seed_sd": 0.0, "tolerance": 0.012092425838785583, "passes": false}, "odd.women.a22_74.q90": {"truth": 0.6444444444444445, "filled": 0.6086206896551725, "gap": -0.057193386823322534, "seed_sd": 0.0, "tolerance": 0.013338685039058783, "passes": false}, "odd.women.a22_74.r1": {"truth": 0.8812684347612898, "filled": 0.9394347478682354, "gap": 0.058166313106945644, "seed_sd": 0.0, "tolerance": 0.002475922564645967, "passes": false}, "odd.women.a22_74.r2": {"truth": 0.8056395091372208, "filled": 0.8970038287402264, "gap": 0.09136431960300562, "seed_sd": 0.0, "tolerance": 0.004772836686607259, "passes": false}, "odd.women.a22_74.r3": {"truth": 0.7468316241328095, "filled": 0.7676196288675307, "gap": 0.0207880047347212, "seed_sd": 0.0, "tolerance": 0.005171487109756781, "passes": false}, "odd.women.a22_74.r4": {"truth": 0.7116744894807431, "filled": 0.7579270771987335, "gap": 0.04625258771799046, "seed_sd": 0.0, "tolerance": 0.007666262733486471, "passes": false}, "odd.women.a22_74.wint": {"truth": 0.051544874036178946, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07133033012317215, "passes": false}, "odd.women.a22_74.zexit": {"truth": 0.45845752128015943, "filled": 0.01236869003547409, "gap": -3.6126993579608797, "seed_sd": 0.0, "tolerance": 0.016039678883419197, "passes": false}, "odd.women.a22_74.zint": {"truth": 0.022201124403524123, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04476234327525859, "passes": false}, "odd.women.a30_44.atcap": {"truth": 0.034619662054697874, "filled": 0.013831551720358612, "gap": -0.917469449162823, "seed_sd": 0.0, "tolerance": 0.08510646139536093, "passes": false}, "odd.women.a30_44.level": {"truth": 0.2579924798566703, "filled": 0.2586995070802617, "gap": 0.0027367471637131935, "seed_sd": 0.0, "tolerance": 0.01578879525774269, "passes": true}, "odd.women.a30_44.q10": {"truth": 0.04228855721393035, "filled": 0.030303030303030304, "gap": -0.33326881690367527, "seed_sd": 0.0, "tolerance": 0.04573918824793483, "passes": false}, "odd.women.a30_44.q50": {"truth": 0.26859504132231404, "filled": 0.23850574712643677, "gap": -0.11881141571682718, "seed_sd": 0.0, "tolerance": 0.018488627942446087, "passes": false}, "odd.women.a30_44.q90": {"truth": 0.6716417910447762, "filled": 0.636085626911315, "gap": -0.054391961575289194, "seed_sd": 0.0, "tolerance": 0.018806950089013175, "passes": false}, "odd.women.a30_44.r1": {"truth": 0.888241419643356, "filled": 0.9468343895814233, "gap": 0.058592969938067285, "seed_sd": 0.0, "tolerance": 0.0034876380830913202, "passes": false}, "odd.women.a30_44.r2": {"truth": 0.8242277470219269, "filled": 0.9096684494259086, "gap": 0.08544070240398172, "seed_sd": 0.0, "tolerance": 0.007078708407960211, "passes": false}, "odd.women.a30_44.r3": {"truth": 0.7685608654079126, "filled": 0.7927366027222539, "gap": 0.024175737314341306, "seed_sd": 0.0, "tolerance": 0.0073196290204859075, "passes": false}, "odd.women.a30_44.r4": {"truth": 0.7490435327097834, "filled": 0.7920119748169124, "gap": 0.04296844210712902, "seed_sd": 0.0, "tolerance": 0.010187042430903364, "passes": false}, "odd.women.a30_44.wint": {"truth": 0.06073485056210584, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.10093703169083498, "passes": false}, "odd.women.a30_44.zexit": {"truth": 0.45066799061202384, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.024719366356404402, "passes": false}, "odd.women.a30_44.zint": {"truth": 0.022222222222222223, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07029190354044747, "passes": false}, "odd.women.a45_59.atcap": {"truth": 0.04053483605515179, "filled": 0.01771406256616308, "gap": -0.827802934506876, "seed_sd": 0.0, "tolerance": 0.09107631963722376, "passes": false}, "odd.women.a45_59.level": {"truth": 0.28104586423928846, "filled": 0.28103084273355944, "gap": -5.345002041967639e-05, "seed_sd": 0.0, "tolerance": 0.01605667435242628, "passes": true}, "odd.women.a45_59.q10": {"truth": 0.053516819571865444, "filled": 0.03581267217630854, "gap": -0.40169418683552927, "seed_sd": 0.0, "tolerance": 0.051646333169991114, "passes": false}, "odd.women.a45_59.q50": {"truth": 0.29338842975206614, "filled": 0.26666666666666666, "gap": -0.09549799086694866, "seed_sd": 0.0, "tolerance": 0.0182941133196311, "passes": false}, "odd.women.a45_59.q90": {"truth": 0.7300275482093664, "filled": 0.6954022988505747, "gap": -0.0485917453391595, "seed_sd": 0.0, "tolerance": 0.0208569023153089, "passes": false}, "odd.women.a45_59.r1": {"truth": 0.9101006329634369, "filled": 0.95644972873215, "gap": 0.046349095768713044, "seed_sd": 0.0, "tolerance": 0.0037361334549905496, "passes": false}, "odd.women.a45_59.r2": {"truth": 0.8498173267452248, "filled": 0.9175157606106729, "gap": 0.06769843386544805, "seed_sd": 0.0, "tolerance": 0.0074184353060767995, "passes": false}, "odd.women.a45_59.r3": {"truth": 0.8096130207660378, "filled": 0.8275574170028757, "gap": 0.017944396236837856, "seed_sd": 0.0, "tolerance": 0.007740061426920219, "passes": false}, "odd.women.a45_59.r4": {"truth": 0.7620167777938601, "filled": 0.7929407426303564, "gap": 0.030923964836496287, "seed_sd": 0.0, "tolerance": 0.012512985825002505, "passes": false}, "odd.women.a45_59.wint": {"truth": 0.050115932427956277, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.1271041647107106, "passes": false}, "odd.women.a45_59.zexit": {"truth": 0.4693136110029843, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.030024476078798885, "passes": false}, "odd.women.a45_59.zint": {"truth": 0.01632776600720909, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09017841563442361, "passes": false}, "odd.women.a60_74.atcap": {"truth": 0.01812919896640827, "filled": 0.006878104700038212, "gap": -0.9691807066740816, "seed_sd": 0.0}, "odd.women.a60_74.level": {"truth": 0.12926943691615103, "filled": 0.12931157523148915, "gap": 0.00032591964399220075, "seed_sd": 0.0, "tolerance": 0.04408494999775865, "passes": true}, "odd.women.a60_74.q10": {"truth": 0.017412935323383085, "filled": 0.009938837920489297, "gap": -0.5607632349918994, "seed_sd": 0.0, "tolerance": 0.13013445963696416, "passes": false}, "odd.women.a60_74.q50": {"truth": 0.15517241379310345, "filled": 0.12241379310344827, "gap": -0.23712979328894956, "seed_sd": 0.0, "tolerance": 0.05365060769995596, "passes": false}, "odd.women.a60_74.q90": {"truth": 0.5360696517412935, "filled": 0.47055555555555556, "gap": -0.13035007015698297, "seed_sd": 0.0, "tolerance": 0.04960336282899433, "passes": false}, "odd.women.a60_74.r1": {"truth": 0.8610627385333514, "filled": 0.9308441096943134, "gap": 0.06978137116096206, "seed_sd": 0.0, "tolerance": 0.009751654703991025, "passes": false}, "odd.women.a60_74.r2": {"truth": 0.7722738686836694, "filled": 0.8536224830395716, "gap": 0.08134861435590213, "seed_sd": 0.0, "tolerance": 0.022174026993056747, "passes": false}, "odd.women.a60_74.r3": {"truth": 0.7141078828415282, "filled": 0.7255034910997732, "gap": 0.011395608258245038, "seed_sd": 0.0, "tolerance": 0.018479646240433655, "passes": true}, "odd.women.a60_74.r4": {"truth": 0.6501204938024134, "filled": 0.6629121666010215, "gap": 0.012791672798608045, "seed_sd": 0.0, "tolerance": 0.036314307016843086, "passes": true}, "odd.women.a60_74.wint": {"truth": 0.023107482596493787, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.women.a60_74.zexit": {"truth": 0.5155483759303063, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.03896610501214408, "passes": false}, "odd.women.a60_74.zint": {"truth": 0.022290698541288734, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.women.b1936_1940.aime_p10": {"truth": 82.29999999999995, "filled": 83.0, "gap": 0.00846950011357439, "seed_sd": 0.0, "tolerance": 0.32371200850504067, "passes": true}, "odd.women.b1936_1940.aime_p25": {"truth": 287.75, "filled": 287.75, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.20139115911902655, "passes": true}, "odd.women.b1936_1940.aime_p50": {"truth": 786.0, "filled": 786.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.12524071173214943, "passes": true}, "odd.women.b1936_1940.aime_p75": {"truth": 1566.0, "filled": 1565.0, "gap": -0.0006387735764947777, "seed_sd": 0.0, "tolerance": 0.10110924143579125, "passes": true}, "odd.women.b1936_1940.aime_p90": {"truth": 2438.7000000000007, "filled": 2435.7000000000007, "gap": -0.0012309208841250197, "seed_sd": 0.0, "tolerance": 0.08967006892401257, "passes": true}, "odd.women.b1936_1945.aime_p10": {"truth": 98.0, "filled": 98.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.1995625804481066, "passes": true}, "odd.women.b1936_1945.aime_p25": {"truth": 345.0, "filled": 344.0, "gap": -0.0029027596579620507, "seed_sd": 0.0, "tolerance": 0.11108088073804002, "passes": true}, "odd.women.b1936_1945.aime_p50": {"truth": 932.0, "filled": 930.0, "gap": -0.002148228538289665, "seed_sd": 0.0, "tolerance": 0.07343389253317807, "passes": true}, "odd.women.b1936_1945.aime_p75": {"truth": 1865.0, "filled": 1863.0, "gap": -0.0010729614763267392, "seed_sd": 0.0, "tolerance": 0.06283602704614394, "passes": true}, "odd.women.b1936_1945.aime_p90": {"truth": 2920.0, "filled": 2924.2000000000007, "gap": 0.001437322721010048, "seed_sd": 0.0, "tolerance": 0.0536209031123137, "passes": true}, "odd.women.b1941_1945.aime_p10": {"truth": 115.0, "filled": 116.0, "gap": 0.008658062743114314, "seed_sd": 0.0, "tolerance": 0.2647761737592696, "passes": true}, "odd.women.b1941_1945.aime_p25": {"truth": 399.0, "filled": 398.0, "gap": -0.0025094116054260596, "seed_sd": 0.0, "tolerance": 0.15691889792263147, "passes": true}, "odd.women.b1941_1945.aime_p50": {"truth": 1077.0, "filled": 1075.0, "gap": -0.001858736594625654, "seed_sd": 0.0, "tolerance": 0.08797933476607916, "passes": true}, "odd.women.b1941_1945.aime_p75": {"truth": 2116.0, "filled": 2116.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.07205053130731136, "passes": true}, "odd.women.b1941_1945.aime_p90": {"truth": 3268.6000000000004, "filled": 3270.6000000000004, "gap": 0.0006116956393338313, "seed_sd": 0.0, "tolerance": 0.06548294022251817, "passes": true}, "odd.women.b1946_1955.paime_p10": {"truth": 158.0, "filled": 159.0, "gap": 0.00630916919326463, "seed_sd": 0.0, "tolerance": 0.1085059910489472, "passes": true}, "odd.women.b1946_1955.paime_p25": {"truth": 519.0, "filled": 519.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.062889445073137, "passes": true}, "odd.women.b1946_1955.paime_p50": {"truth": 1313.0, "filled": 1313.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.04313158587991037, "passes": true}, "odd.women.b1946_1955.paime_p75": {"truth": 2486.0, "filled": 2486.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.03158401312013406, "passes": true}, "odd.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3779.0, "gap": -0.0002645852641443014, "seed_sd": 0.0, "tolerance": 0.02998098721497899, "passes": true}, "odd.women.b1946_1980.paime_p10": {"truth": 109.0, "filled": 108.0, "gap": -0.009216655104923532, "seed_sd": 0.0, "tolerance": 0.0500438197853722, "passes": true}, "odd.women.b1946_1980.paime_p25": {"truth": 309.0, "filled": 311.0, "gap": 0.006451635281488066, "seed_sd": 0.0, "tolerance": 0.030021121307637916, "passes": true}, "odd.women.b1946_1980.paime_p50": {"truth": 758.0, "filled": 759.0, "gap": 0.001318391753258652, "seed_sd": 0.0, "tolerance": 0.02113414008122686, "passes": true}, "odd.women.b1946_1980.paime_p75": {"truth": 1590.0, "filled": 1590.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.018434974631993433, "passes": true}, "odd.women.b1946_1980.paime_p90": {"truth": 2696.0, "filled": 2697.0, "gap": 0.00037085110753221073, "seed_sd": 0.0, "tolerance": 0.018162919937725473, "passes": true}, "odd.women.b1956_1965.paime_p10": {"truth": 140.0, "filled": 140.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0987946317785998, "passes": true}, "odd.women.b1956_1965.paime_p25": {"truth": 426.0, "filled": 425.0, "gap": -0.0023501773449536856, "seed_sd": 0.0, "tolerance": 0.05702004966656188, "passes": true}, "odd.women.b1956_1965.paime_p50": {"truth": 1000.0, "filled": 1000.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0353321966023362, "passes": true}, "odd.women.b1956_1965.paime_p75": {"truth": 1842.0, "filled": 1846.0, "gap": 0.0021691982475449123, "seed_sd": 0.0, "tolerance": 0.026005856561860215, "passes": true}, "odd.women.b1956_1965.paime_p90": {"truth": 2793.0, "filled": 2795.0999999999985, "gap": 0.0007515971793115028, "seed_sd": 0.0, "tolerance": 0.027293077620429543, "passes": true}, "odd.women.b1966_1980.paime_p10": {"truth": 80.0, "filled": 80.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.06283109225080236, "passes": true}, "odd.women.b1966_1980.paime_p25": {"truth": 211.0, "filled": 211.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.033157126538046186, "passes": true}, "odd.women.b1966_1980.paime_p50": {"truth": 463.0, "filled": 466.0, "gap": 0.006458580039411466, "seed_sd": 0.0, "tolerance": 0.02601990880890944, "passes": true}, "odd.women.b1966_1980.paime_p75": {"truth": 866.75, "filled": 870.0, "gap": 0.003742627083497041, "seed_sd": 0.0, "tolerance": 0.02735691355887302, "passes": true}, "odd.women.b1966_1980.paime_p90": {"truth": 1387.0, "filled": 1389.0, "gap": 0.0014409224395128817, "seed_sd": 0.0, "tolerance": 0.027821321010049294, "passes": true}}}, "two_sided": {"passes": false, "n_gating": 183, "n_failing": 99, "cells": {"odd.men.a22_29.atcap": {"truth": 0.014676604027739744, "filled": 0.003963318801164396, "gap": -1.309172907789832, "seed_sd": 0.0, "tolerance": 0.18023370223284765, "passes": false}, "odd.men.a22_29.level": {"truth": 0.23657185599421368, "filled": 0.23815412941641792, "gap": 0.006666074030311719, "seed_sd": 0.0, "tolerance": 0.02171146931539837, "passes": true}, "odd.men.a22_29.q10": {"truth": 0.040229885057471264, "filled": 0.03611111111111111, "gap": -0.10800952382940343, "seed_sd": 0.0, "tolerance": 0.07329266358707544, "passes": false}, "odd.men.a22_29.q50": {"truth": 0.24222222222222223, "filled": 0.22201492537313433, "gap": -0.08711096742405089, "seed_sd": 0.0, "tolerance": 0.022980444394274636, "passes": false}, "odd.men.a22_29.q90": {"truth": 0.5661157024793388, "filled": 0.5323266998341619, "gap": -0.06154108035996475, "seed_sd": 0.0, "tolerance": 0.02353059621893112, "passes": false}, "odd.men.a22_29.r1": {"truth": 0.8114342592359591, "filled": 0.9077680195907103, "gap": 0.09633376035475116, "seed_sd": 0.0, "tolerance": 0.007273490161268873, "passes": false}, "odd.men.a22_29.r2": {"truth": 0.7090594261869381, "filled": 0.8581398280676814, "gap": 0.14908040188074334, "seed_sd": 0.0, "tolerance": 0.014356564488812985, "passes": false}, "odd.men.a22_29.r3": {"truth": 0.6020834292042946, "filled": 0.6504088491993362, "gap": 0.048325419995041585, "seed_sd": 0.0, "tolerance": 0.013280307041936976, "passes": false}, "odd.men.a22_29.r4": {"truth": 0.5990582702554191, "filled": 0.6892222646133446, "gap": 0.09016399435792544, "seed_sd": 0.0, "tolerance": 0.02119963449400435, "passes": false}, "odd.men.a22_29.wint": {"truth": 0.08288543140028289, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.men.a22_29.zexit": {"truth": 0.4099387400566883, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.052311207438163365, "passes": false}, "odd.men.a22_29.zint": {"truth": 0.027589420573962787, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.12140160693630424, "passes": false}, "odd.men.a22_74.atcap": {"truth": 0.10371778137953785, "filled": 0.046035707021086794, "gap": -0.8122562350030691, "seed_sd": 0.0, "tolerance": 0.03690628859850606, "passes": false}, "odd.men.a22_74.level": {"truth": 0.35325148977531445, "filled": 0.3538288994241502, "gap": 0.0016332224341244483, "seed_sd": 0.0, "tolerance": 0.01106822547283481, "passes": true}, "odd.men.a22_74.q10": {"truth": 0.06111111111111111, "filled": 0.04611111111111111, "gap": -0.2816397579958183, "seed_sd": 0.0, "tolerance": 0.03462864626687699, "passes": false}, "odd.men.a22_74.q50": {"truth": 0.37052341597796146, "filled": 0.3367816091954023, "gap": -0.09548196740860526, "seed_sd": 0.0, "tolerance": 0.012463773427145929, "passes": false}, "odd.men.a22_74.q90": {"truth": 1.0, "filled": 0.9266169154228856, "gap": -0.07621505079940682, "seed_sd": 0.0, "tolerance": 0.003451823067723808, "passes": false}, "odd.men.a22_74.r1": {"truth": 0.8935990156167083, "filled": 0.9490345370051931, "gap": 0.05543552138848484, "seed_sd": 0.0, "tolerance": 0.00258049589094515, "passes": false}, "odd.men.a22_74.r2": {"truth": 0.8291027164330249, "filled": 0.9084834928643107, "gap": 0.07938077643128583, "seed_sd": 0.0, "tolerance": 0.004576681103770273, "passes": false}, "odd.men.a22_74.r3": {"truth": 0.7806492619651291, "filled": 0.8016928703964389, "gap": 0.021043608431309813, "seed_sd": 0.0, "tolerance": 0.004722018684826323, "passes": false}, "odd.men.a22_74.r4": {"truth": 0.7414979369803346, "filled": 0.7832471318914956, "gap": 0.041749194911161025, "seed_sd": 0.0, "tolerance": 0.007000982741779954, "passes": false}, "odd.men.a22_74.wint": {"truth": 0.05745341614906832, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07337302886151653, "passes": false}, "odd.men.a22_74.zexit": {"truth": 0.44752843148294114, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.01916127382099423, "passes": false}, "odd.men.a22_74.zint": {"truth": 0.019892968610837895, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.05481211657295162, "passes": false}, "odd.men.a30_44.atcap": {"truth": 0.10574805846620577, "filled": 0.04650078322293776, "gap": -0.82159030214472, "seed_sd": 0.0, "tolerance": 0.049095580366705964, "passes": false}, "odd.men.a30_44.level": {"truth": 0.3982899917120792, "filled": 0.39901752453379963, "gap": 0.001824974702530513, "seed_sd": 0.0, "tolerance": 0.015220021463125325, "passes": true}, "odd.men.a30_44.q10": {"truth": 0.08706467661691543, "filled": 0.06778606965174129, "gap": -0.25029454038016086, "seed_sd": 0.0, "tolerance": 0.047669064141889934, "passes": false}, "odd.men.a30_44.q50": {"truth": 0.41333333333333333, "filled": 0.38685015290519875, "gap": -0.06621695367851432, "seed_sd": 0.0, "tolerance": 0.016660415997544607, "passes": false}, "odd.men.a30_44.q90": {"truth": 1.0, "filled": 0.9390547263681592, "gap": -0.06288151992994163, "seed_sd": 0.0, "tolerance": 0.0037405831290436837, "passes": false}, "odd.men.a30_44.r1": {"truth": 0.9000861304840623, "filled": 0.9541064209919281, "gap": 0.05402029050786583, "seed_sd": 0.0, "tolerance": 0.003556208950302802, "passes": false}, "odd.men.a30_44.r2": {"truth": 0.8479233224100629, "filled": 0.9269323018569282, "gap": 0.07900897944686536, "seed_sd": 0.0, "tolerance": 0.006500650570758907, "passes": false}, "odd.men.a30_44.r3": {"truth": 0.8030903893396322, "filled": 0.8276167677302705, "gap": 0.024526378390638315, "seed_sd": 0.0, "tolerance": 0.006087370619188444, "passes": false}, "odd.men.a30_44.r4": {"truth": 0.7869108820616492, "filled": 0.832720868870299, "gap": 0.04580998680864978, "seed_sd": 0.0, "tolerance": 0.009218001294372115, "passes": false}, "odd.men.a30_44.wint": {"truth": 0.07239145006330527, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.12001713872675526, "passes": false}, "odd.men.a30_44.zexit": {"truth": 0.4342024282437196, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.031953074094816486, "passes": false}, "odd.men.a30_44.zint": {"truth": 0.017933882760031997, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.08989768311292742, "passes": false}, "odd.men.a45_59.atcap": {"truth": 0.15826463477931516, "filled": 0.07469452108789909, "gap": -0.750861791701275, "seed_sd": 0.0, "tolerance": 0.04711609868499565, "passes": false}, "odd.men.a45_59.level": {"truth": 0.4279852808128974, "filled": 0.4276293059548939, "gap": -0.0008320916544616308, "seed_sd": 0.0, "tolerance": 0.015834008317210692, "passes": true}, "odd.men.a45_59.q10": {"truth": 0.08868501529051988, "filled": 0.06592039800995025, "gap": -0.29664301471606525, "seed_sd": 0.0, "tolerance": 0.055701610600039544, "passes": false}, "odd.men.a45_59.q50": {"truth": 0.4701492537313433, "filled": 0.4345730027548209, "gap": -0.07868625928414963, "seed_sd": 0.0, "tolerance": 0.020385938992396834, "passes": false}, "odd.men.a45_59.q90": {"truth": 1.0, "filled": 0.9958677685950413, "gap": -0.004140792666031388, "seed_sd": 0.0}, "odd.men.a45_59.r1": {"truth": 0.9077661818197654, "filled": 0.9568133880651853, "gap": 0.049047206245419916, "seed_sd": 0.0, "tolerance": 0.004032009924506552, "passes": false}, "odd.men.a45_59.r2": {"truth": 0.850421444941529, "filled": 0.9210343284479916, "gap": 0.07061288350646255, "seed_sd": 0.0, "tolerance": 0.0074262368998509335, "passes": false}, "odd.men.a45_59.r3": {"truth": 0.8146289743583418, "filled": 0.8339593243046802, "gap": 0.01933034994633842, "seed_sd": 0.0, "tolerance": 0.007198396861260139, "passes": false}, "odd.men.a45_59.r4": {"truth": 0.7598303885400162, "filled": 0.796798027663896, "gap": 0.03696763912387979, "seed_sd": 0.0, "tolerance": 0.012107705260459355, "passes": false}, "odd.men.a45_59.wint": {"truth": 0.05825168693530555, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.13344110142481172, "passes": false}, "odd.men.a45_59.zexit": {"truth": 0.45185706741770815, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.036314919811016276, "passes": false}, "odd.men.a45_59.zint": {"truth": 0.015835938476928848, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09287476169914739, "passes": false}, "odd.men.a60_74.atcap": {"truth": 0.10024796315975912, "filled": 0.038797709400772665, "gap": -0.949285539547915, "seed_sd": 0.0, "tolerance": 0.12472307183935011, "passes": false}, "odd.men.a60_74.level": {"truth": 0.20453937198442498, "filled": 0.20537657883929814, "gap": 0.004084778927416988, "seed_sd": 0.0, "tolerance": 0.0522939828282486, "passes": true}, "odd.men.a60_74.q10": {"truth": 0.01529051987767584, "filled": 0.009950248756218905, "gap": -0.4296354690359774, "seed_sd": 0.0, "tolerance": 0.20186846491463123, "passes": false}, "odd.men.a60_74.q50": {"truth": 0.21666666666666667, "filled": 0.17507645259938837, "gap": -0.21313732370234062, "seed_sd": 0.0, "tolerance": 0.07446457229432804, "passes": false}, "odd.men.a60_74.q90": {"truth": 1.0, "filled": 0.8222222222222222, "gap": -0.19574457712609536, "seed_sd": 0.0, "tolerance": 0.04069424979546094, "passes": false}, "odd.men.a60_74.r1": {"truth": 0.8712782553777247, "filled": 0.9364895683685154, "gap": 0.06521131299079064, "seed_sd": 0.0, "tolerance": 0.009750532797445479, "passes": false}, "odd.men.a60_74.r2": {"truth": 0.7888015745113253, "filled": 0.8691600366633899, "gap": 0.08035846215206466, "seed_sd": 0.0, "tolerance": 0.020757725353477027, "passes": false}, "odd.men.a60_74.r3": {"truth": 0.7283746799072385, "filled": 0.7453822584843823, "gap": 0.017007578577143856, "seed_sd": 0.0, "tolerance": 0.02064690204069302, "passes": true}, "odd.men.a60_74.r4": {"truth": 0.6781549627202678, "filled": 0.6972394023844436, "gap": 0.019084439664175834, "seed_sd": 0.0, "tolerance": 0.03820786124425089, "passes": true}, "odd.men.a60_74.wint": {"truth": 0.03084255319148936, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.men.a60_74.zexit": {"truth": 0.5047452182800409, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04434932710092839, "passes": false}, "odd.men.a60_74.zint": {"truth": 0.027132152458672565, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.17758159695027936, "passes": false}, "odd.men.b1936_1940.aime_p10": {"truth": 346.0, "filled": 345.0, "gap": -0.0028943580263645075, "seed_sd": 0.0, "tolerance": 0.30817736367768905, "passes": true}, "odd.men.b1936_1940.aime_p25": {"truth": 1131.0, "filled": 1131.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.15436566295506535, "passes": true}, "odd.men.b1936_1940.aime_p50": {"truth": 2430.0, "filled": 2426.0, "gap": -0.0016474468305984757, "seed_sd": 0.0, "tolerance": 0.0822816707411384, "passes": true}, "odd.men.b1936_1940.aime_p75": {"truth": 3582.0, "filled": 3576.0, "gap": -0.001676446327252279, "seed_sd": 0.0, "tolerance": 0.04915519601813279, "passes": true}, "odd.men.b1936_1940.aime_p90": {"truth": 4382.0, "filled": 4372.0, "gap": -0.0022846708589838727, "seed_sd": 0.0, "tolerance": 0.03801502836217192, "passes": true}, "odd.men.b1936_1945.aime_p10": {"truth": 342.8000000000002, "filled": 341.0, "gap": -0.005264709440107929, "seed_sd": 0.0, "tolerance": 0.19387427864829543, "passes": true}, "odd.men.b1936_1945.aime_p25": {"truth": 1152.0, "filled": 1152.5, "gap": 0.00043393361496679717, "seed_sd": 0.0, "tolerance": 0.09928513681983075, "passes": true}, "odd.men.b1936_1945.aime_p50": {"truth": 2625.0, "filled": 2621.0, "gap": -0.001524971702317579, "seed_sd": 0.0, "tolerance": 0.05013200755321697, "passes": true}, "odd.men.b1936_1945.aime_p75": {"truth": 3991.5, "filled": 3985.5, "gap": -0.0015043252178763566, "seed_sd": 0.0, "tolerance": 0.030282584009309735, "passes": true}, "odd.men.b1936_1945.aime_p90": {"truth": 5075.0, "filled": 5060.0, "gap": -0.0029600416284765174, "seed_sd": 0.0, "tolerance": 0.029246187918705275, "passes": true}, "odd.men.b1941_1945.aime_p10": {"truth": 338.0, "filled": 336.0, "gap": -0.005934735519814716, "seed_sd": 0.0, "tolerance": 0.2434370939708618, "passes": true}, "odd.men.b1941_1945.aime_p25": {"truth": 1174.25, "filled": 1176.25, "gap": 0.0017017659924842832, "seed_sd": 0.0, "tolerance": 0.15717605214643726, "passes": true}, "odd.men.b1941_1945.aime_p50": {"truth": 2821.5, "filled": 2818.0, "gap": -0.0012412449505694312, "seed_sd": 0.0, "tolerance": 0.06794319334096698, "passes": true}, "odd.men.b1941_1945.aime_p75": {"truth": 4421.0, "filled": 4414.0, "gap": -0.0015846070095602016, "seed_sd": 0.0, "tolerance": 0.039611439643930886, "passes": true}, "odd.men.b1941_1945.aime_p90": {"truth": 5528.300000000001, "filled": 5520.300000000001, "gap": -0.0014481475296577173, "seed_sd": 0.0, "tolerance": 0.024092287379584104, "passes": true}, "odd.men.b1946_1955.paime_p10": {"truth": 308.0, "filled": 309.0, "gap": 0.0032414939241718344, "seed_sd": 0.0, "tolerance": 0.11641286886749184, "passes": true}, "odd.men.b1946_1955.paime_p25": {"truth": 1026.0, "filled": 1028.0, "gap": 0.001947420284395207, "seed_sd": 0.0, "tolerance": 0.07923354406873329, "passes": true}, "odd.men.b1946_1955.paime_p50": {"truth": 2614.0, "filled": 2609.0, "gap": -0.0019146090474384536, "seed_sd": 0.0, "tolerance": 0.03884145918889322, "passes": true}, "odd.men.b1946_1955.paime_p75": {"truth": 4390.0, "filled": 4389.0, "gap": -0.0002278163809830147, "seed_sd": 0.0, "tolerance": 0.02464882648778213, "passes": true}, "odd.men.b1946_1955.paime_p90": {"truth": 5810.0, "filled": 5808.0, "gap": -0.00034429334132468625, "seed_sd": 0.0, "tolerance": 0.01935089604023114, "passes": true}, "odd.men.b1946_1980.paime_p10": {"truth": 175.0, "filled": 176.0, "gap": 0.005698021114636909, "seed_sd": 0.0, "tolerance": 0.05432313062129484, "passes": true}, "odd.men.b1946_1980.paime_p25": {"truth": 487.0, "filled": 488.0, "gap": 0.002051282770557883, "seed_sd": 0.0, "tolerance": 0.03064646902486962, "passes": true}, "odd.men.b1946_1980.paime_p50": {"truth": 1245.0, "filled": 1247.0, "gap": 0.0016051367812286443, "seed_sd": 0.0, "tolerance": 0.025757396517852603, "passes": true}, "odd.men.b1946_1980.paime_p75": {"truth": 2644.0, "filled": 2645.0, "gap": 0.0003781433208231988, "seed_sd": 0.0, "tolerance": 0.0206417529534006, "passes": true}, "odd.men.b1946_1980.paime_p90": {"truth": 4310.0, "filled": 4310.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.01797004074133569, "passes": true}, "odd.men.b1956_1965.paime_p10": {"truth": 253.0, "filled": 253.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.10325340800222761, "passes": true}, "odd.men.b1956_1965.paime_p25": {"truth": 765.0, "filled": 767.0, "gap": 0.0026109675407202104, "seed_sd": 0.0, "tolerance": 0.06714959239176221, "passes": true}, "odd.men.b1956_1965.paime_p50": {"truth": 1764.0, "filled": 1765.0, "gap": 0.000566732800660219, "seed_sd": 0.0, "tolerance": 0.03783036836459788, "passes": true}, "odd.men.b1956_1965.paime_p75": {"truth": 2958.0, "filled": 2961.0, "gap": 0.0010136848308457402, "seed_sd": 0.0, "tolerance": 0.02582244416868233, "passes": true}, "odd.men.b1956_1965.paime_p90": {"truth": 4111.0, "filled": 4110.0, "gap": -0.00024327940759860667, "seed_sd": 0.0, "tolerance": 0.022664845011086624, "passes": true}, "odd.men.b1966_1980.paime_p10": {"truth": 112.0, "filled": 112.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.07763263566160054, "passes": true}, "odd.men.b1966_1980.paime_p25": {"truth": 303.25, "filled": 304.0, "gap": 0.0024701535820623732, "seed_sd": 0.0, "tolerance": 0.04145124387911718, "passes": true}, "odd.men.b1966_1980.paime_p50": {"truth": 655.0, "filled": 658.0, "gap": 0.0045696956900656005, "seed_sd": 0.0, "tolerance": 0.02859495381248372, "passes": true}, "odd.men.b1966_1980.paime_p75": {"truth": 1225.0, "filled": 1228.0, "gap": 0.0024459857282606023, "seed_sd": 0.0, "tolerance": 0.024541637055534683, "passes": true}, "odd.men.b1966_1980.paime_p90": {"truth": 1888.9000000000015, "filled": 1889.9000000000015, "gap": 0.0005292685632181104, "seed_sd": 0.0, "tolerance": 0.025554311796701996, "passes": true}, "odd.women.a22_29.atcap": {"truth": 0.005728386357627567, "filled": 0.0014145096609551587, "gap": -1.3986509364070692, "seed_sd": 0.0}, "odd.women.a22_29.level": {"truth": 0.1854665923300802, "filled": 0.18583978254257602, "gap": 0.002010147756208447, "seed_sd": 0.0, "tolerance": 0.018517427407013797, "passes": true}, "odd.women.a22_29.q10": {"truth": 0.028735632183908046, "filled": 0.02522935779816514, "gap": -0.13012958377023498, "seed_sd": 0.0, "tolerance": 0.0684466769976442, "passes": false}, "odd.women.a22_29.q50": {"truth": 0.191131498470948, "filled": 0.16944444444444445, "gap": -0.1204365531219469, "seed_sd": 0.0, "tolerance": 0.021849929026002995, "passes": false}, "odd.women.a22_29.q90": {"truth": 0.46005509641873277, "filled": 0.42838544878911294, "gap": -0.07132288554928723, "seed_sd": 0.0, "tolerance": 0.020751504651450387, "passes": false}, "odd.women.a22_29.r1": {"truth": 0.7942016511246733, "filled": 0.8964267346621329, "gap": 0.10222508353745952, "seed_sd": 0.0, "tolerance": 0.006445547562698701, "passes": false}, "odd.women.a22_29.r2": {"truth": 0.6731585499668502, "filled": 0.8403968435266072, "gap": 0.16723829355975695, "seed_sd": 0.0, "tolerance": 0.013111971249341917, "passes": false}, "odd.women.a22_29.r3": {"truth": 0.5535614164842262, "filled": 0.6047968332303263, "gap": 0.051235416746100104, "seed_sd": 0.0, "tolerance": 0.012054115899361516, "passes": false}, "odd.women.a22_29.r4": {"truth": 0.552312691565311, "filled": 0.649332972302572, "gap": 0.09702028073726099, "seed_sd": 0.0, "tolerance": 0.01982762523559322, "passes": false}, "odd.women.a22_29.wint": {"truth": 0.08501885498800137, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.15037337545812676, "passes": false}, "odd.women.a22_29.zexit": {"truth": 0.41757828810020875, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.041287246240518535, "passes": false}, "odd.women.a22_29.zint": {"truth": 0.0315547703180212, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09342121518593281, "passes": false}, "odd.women.a22_74.atcap": {"truth": 0.029398307023564402, "filled": 0.011954671808208356, "gap": -0.8998149401825817, "seed_sd": 0.0, "tolerance": 0.06709882226274566, "passes": false}, "odd.women.a22_74.level": {"truth": 0.2372854094489719, "filled": 0.2376347653333952, "gap": 0.001471219653273792, "seed_sd": 0.0, "tolerance": 0.010902725750872375, "passes": true}, "odd.women.a22_74.q10": {"truth": 0.03777777777777778, "filled": 0.026859504132231406, "gap": -0.3411013105469456, "seed_sd": 0.0, "tolerance": 0.031129845660342752, "passes": false}, "odd.women.a22_74.q50": {"truth": 0.24875621890547264, "filled": 0.21888888888888888, "gap": -0.12790913195539244, "seed_sd": 0.0, "tolerance": 0.012092425838785583, "passes": false}, "odd.women.a22_74.q90": {"truth": 0.6444444444444445, "filled": 0.6072222222222222, "gap": -0.059493795923871884, "seed_sd": 0.0, "tolerance": 0.013338685039058783, "passes": false}, "odd.women.a22_74.r1": {"truth": 0.8812684347612898, "filled": 0.9414914928166868, "gap": 0.06022305805539696, "seed_sd": 0.0, "tolerance": 0.002475922564645967, "passes": false}, "odd.women.a22_74.r2": {"truth": 0.8056395091372208, "filled": 0.894980676078455, "gap": 0.08934116694123417, "seed_sd": 0.0, "tolerance": 0.004772836686607259, "passes": false}, "odd.women.a22_74.r3": {"truth": 0.7468316241328095, "filled": 0.769997332317142, "gap": 0.0231657081843325, "seed_sd": 0.0, "tolerance": 0.005171487109756781, "passes": false}, "odd.women.a22_74.r4": {"truth": 0.7116744894807431, "filled": 0.7556167205497889, "gap": 0.04394223106904582, "seed_sd": 0.0, "tolerance": 0.007666262733486471, "passes": false}, "odd.women.a22_74.wint": {"truth": 0.051544874036178946, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07133033012317215, "passes": false}, "odd.women.a22_74.zexit": {"truth": 0.45845752128015943, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.016039678883419197, "passes": false}, "odd.women.a22_74.zint": {"truth": 0.022201124403524123, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04476234327525859, "passes": false}, "odd.women.a30_44.atcap": {"truth": 0.034619662054697874, "filled": 0.013831551720358612, "gap": -0.917469449162823, "seed_sd": 0.0, "tolerance": 0.08510646139536093, "passes": false}, "odd.women.a30_44.level": {"truth": 0.2579924798566703, "filled": 0.2586995070802617, "gap": 0.0027367471637131935, "seed_sd": 0.0, "tolerance": 0.01578879525774269, "passes": true}, "odd.women.a30_44.q10": {"truth": 0.04228855721393035, "filled": 0.030303030303030304, "gap": -0.33326881690367527, "seed_sd": 0.0, "tolerance": 0.04573918824793483, "passes": false}, "odd.women.a30_44.q50": {"truth": 0.26859504132231404, "filled": 0.23850574712643677, "gap": -0.11881141571682718, "seed_sd": 0.0, "tolerance": 0.018488627942446087, "passes": false}, "odd.women.a30_44.q90": {"truth": 0.6716417910447762, "filled": 0.636085626911315, "gap": -0.054391961575289194, "seed_sd": 0.0, "tolerance": 0.018806950089013175, "passes": false}, "odd.women.a30_44.r1": {"truth": 0.888241419643356, "filled": 0.9468346426950571, "gap": 0.058593223051701115, "seed_sd": 0.0, "tolerance": 0.0034876380830913202, "passes": false}, "odd.women.a30_44.r2": {"truth": 0.8242277470219269, "filled": 0.9096682274710212, "gap": 0.08544048044909425, "seed_sd": 0.0, "tolerance": 0.007078708407960211, "passes": false}, "odd.women.a30_44.r3": {"truth": 0.7685608654079126, "filled": 0.792737244038737, "gap": 0.024176378630824447, "seed_sd": 0.0, "tolerance": 0.0073196290204859075, "passes": false}, "odd.women.a30_44.r4": {"truth": 0.7490435327097834, "filled": 0.792012608430596, "gap": 0.042969075720812544, "seed_sd": 0.0, "tolerance": 0.010187042430903364, "passes": false}, "odd.women.a30_44.wint": {"truth": 0.06073485056210584, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.10093703169083498, "passes": false}, "odd.women.a30_44.zexit": {"truth": 0.45066799061202384, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.024719366356404402, "passes": false}, "odd.women.a30_44.zint": {"truth": 0.022222222222222223, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.07029190354044747, "passes": false}, "odd.women.a45_59.atcap": {"truth": 0.04053483605515179, "filled": 0.01771406256616308, "gap": -0.827802934506876, "seed_sd": 0.0, "tolerance": 0.09107631963722376, "passes": false}, "odd.women.a45_59.level": {"truth": 0.28104586423928846, "filled": 0.28103084273355944, "gap": -5.345002041967639e-05, "seed_sd": 0.0, "tolerance": 0.01605667435242628, "passes": true}, "odd.women.a45_59.q10": {"truth": 0.053516819571865444, "filled": 0.03581267217630854, "gap": -0.40169418683552927, "seed_sd": 0.0, "tolerance": 0.051646333169991114, "passes": false}, "odd.women.a45_59.q50": {"truth": 0.29338842975206614, "filled": 0.26666666666666666, "gap": -0.09549799086694866, "seed_sd": 0.0, "tolerance": 0.0182941133196311, "passes": false}, "odd.women.a45_59.q90": {"truth": 0.7300275482093664, "filled": 0.6954022988505747, "gap": -0.0485917453391595, "seed_sd": 0.0, "tolerance": 0.0208569023153089, "passes": false}, "odd.women.a45_59.r1": {"truth": 0.9101006329634369, "filled": 0.9564498180345706, "gap": 0.046349185071133725, "seed_sd": 0.0, "tolerance": 0.0037361334549905496, "passes": false}, "odd.women.a45_59.r2": {"truth": 0.8498173267452248, "filled": 0.9175149282013461, "gap": 0.06769760145612125, "seed_sd": 0.0, "tolerance": 0.0074184353060767995, "passes": false}, "odd.women.a45_59.r3": {"truth": 0.8096130207660378, "filled": 0.8275570329366764, "gap": 0.017944012170638568, "seed_sd": 0.0, "tolerance": 0.007740061426920219, "passes": false}, "odd.women.a45_59.r4": {"truth": 0.7620167777938601, "filled": 0.7929402009981047, "gap": 0.03092342320424457, "seed_sd": 0.0, "tolerance": 0.012512985825002505, "passes": false}, "odd.women.a45_59.wint": {"truth": 0.050115932427956277, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.1271041647107106, "passes": false}, "odd.women.a45_59.zexit": {"truth": 0.4693136110029843, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.030024476078798885, "passes": false}, "odd.women.a45_59.zint": {"truth": 0.01632776600720909, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09017841563442361, "passes": false}, "odd.women.a60_74.atcap": {"truth": 0.01812919896640827, "filled": 0.006878104700038212, "gap": -0.9691807066740816, "seed_sd": 0.0}, "odd.women.a60_74.level": {"truth": 0.12926943691615103, "filled": 0.12931157523148915, "gap": 0.00032591964399220075, "seed_sd": 0.0, "tolerance": 0.04408494999775865, "passes": true}, "odd.women.a60_74.q10": {"truth": 0.017412935323383085, "filled": 0.009938837920489297, "gap": -0.5607632349918994, "seed_sd": 0.0, "tolerance": 0.13013445963696416, "passes": false}, "odd.women.a60_74.q50": {"truth": 0.15517241379310345, "filled": 0.12241379310344827, "gap": -0.23712979328894956, "seed_sd": 0.0, "tolerance": 0.05365060769995596, "passes": false}, "odd.women.a60_74.q90": {"truth": 0.5360696517412935, "filled": 0.47055555555555556, "gap": -0.13035007015698297, "seed_sd": 0.0, "tolerance": 0.04960336282899433, "passes": false}, "odd.women.a60_74.r1": {"truth": 0.8610627385333514, "filled": 0.9308438442070122, "gap": 0.0697811056736608, "seed_sd": 0.0, "tolerance": 0.009751654703991025, "passes": false}, "odd.women.a60_74.r2": {"truth": 0.7722738686836694, "filled": 0.8536239349818906, "gap": 0.08135006629822117, "seed_sd": 0.0, "tolerance": 0.022174026993056747, "passes": false}, "odd.women.a60_74.r3": {"truth": 0.7141078828415282, "filled": 0.7255042892168905, "gap": 0.011396406375362322, "seed_sd": 0.0, "tolerance": 0.018479646240433655, "passes": true}, "odd.women.a60_74.r4": {"truth": 0.6501204938024134, "filled": 0.6629196573096237, "gap": 0.01279916350721022, "seed_sd": 0.0, "tolerance": 0.036314307016843086, "passes": true}, "odd.women.a60_74.wint": {"truth": 0.023107482596493787, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.women.a60_74.zexit": {"truth": 0.5155483759303063, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.03896610501214408, "passes": false}, "odd.women.a60_74.zint": {"truth": 0.022290698541288734, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0}, "odd.women.b1936_1940.aime_p10": {"truth": 82.29999999999995, "filled": 83.0, "gap": 0.00846950011357439, "seed_sd": 0.0, "tolerance": 0.32371200850504067, "passes": true}, "odd.women.b1936_1940.aime_p25": {"truth": 287.75, "filled": 287.75, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.20139115911902655, "passes": true}, "odd.women.b1936_1940.aime_p50": {"truth": 786.0, "filled": 786.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.12524071173214943, "passes": true}, "odd.women.b1936_1940.aime_p75": {"truth": 1566.0, "filled": 1565.0, "gap": -0.0006387735764947777, "seed_sd": 0.0, "tolerance": 0.10110924143579125, "passes": true}, "odd.women.b1936_1940.aime_p90": {"truth": 2438.7000000000007, "filled": 2435.7000000000007, "gap": -0.0012309208841250197, "seed_sd": 0.0, "tolerance": 0.08967006892401257, "passes": true}, "odd.women.b1936_1945.aime_p10": {"truth": 98.0, "filled": 98.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.1995625804481066, "passes": true}, "odd.women.b1936_1945.aime_p25": {"truth": 345.0, "filled": 344.0, "gap": -0.0029027596579620507, "seed_sd": 0.0, "tolerance": 0.11108088073804002, "passes": true}, "odd.women.b1936_1945.aime_p50": {"truth": 932.0, "filled": 930.0, "gap": -0.002148228538289665, "seed_sd": 0.0, "tolerance": 0.07343389253317807, "passes": true}, "odd.women.b1936_1945.aime_p75": {"truth": 1865.0, "filled": 1863.0, "gap": -0.0010729614763267392, "seed_sd": 0.0, "tolerance": 0.06283602704614394, "passes": true}, "odd.women.b1936_1945.aime_p90": {"truth": 2920.0, "filled": 2924.2000000000007, "gap": 0.001437322721010048, "seed_sd": 0.0, "tolerance": 0.0536209031123137, "passes": true}, "odd.women.b1941_1945.aime_p10": {"truth": 115.0, "filled": 116.0, "gap": 0.008658062743114314, "seed_sd": 0.0, "tolerance": 0.2647761737592696, "passes": true}, "odd.women.b1941_1945.aime_p25": {"truth": 399.0, "filled": 398.0, "gap": -0.0025094116054260596, "seed_sd": 0.0, "tolerance": 0.15691889792263147, "passes": true}, "odd.women.b1941_1945.aime_p50": {"truth": 1077.0, "filled": 1075.0, "gap": -0.001858736594625654, "seed_sd": 0.0, "tolerance": 0.08797933476607916, "passes": true}, "odd.women.b1941_1945.aime_p75": {"truth": 2116.0, "filled": 2116.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.07205053130731136, "passes": true}, "odd.women.b1941_1945.aime_p90": {"truth": 3268.6000000000004, "filled": 3270.6000000000004, "gap": 0.0006116956393338313, "seed_sd": 0.0, "tolerance": 0.06548294022251817, "passes": true}, "odd.women.b1946_1955.paime_p10": {"truth": 158.0, "filled": 159.0, "gap": 0.00630916919326463, "seed_sd": 0.0, "tolerance": 0.1085059910489472, "passes": true}, "odd.women.b1946_1955.paime_p25": {"truth": 519.0, "filled": 519.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.062889445073137, "passes": true}, "odd.women.b1946_1955.paime_p50": {"truth": 1313.0, "filled": 1313.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.04313158587991037, "passes": true}, "odd.women.b1946_1955.paime_p75": {"truth": 2486.0, "filled": 2486.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.03158401312013406, "passes": true}, "odd.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3779.0, "gap": -0.0002645852641443014, "seed_sd": 0.0, "tolerance": 0.02998098721497899, "passes": true}, "odd.women.b1946_1980.paime_p10": {"truth": 109.0, "filled": 108.0, "gap": -0.009216655104923532, "seed_sd": 0.0, "tolerance": 0.0500438197853722, "passes": true}, "odd.women.b1946_1980.paime_p25": {"truth": 309.0, "filled": 310.0, "gap": 0.003231020581446309, "seed_sd": 0.0, "tolerance": 0.030021121307637916, "passes": true}, "odd.women.b1946_1980.paime_p50": {"truth": 758.0, "filled": 758.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.02113414008122686, "passes": true}, "odd.women.b1946_1980.paime_p75": {"truth": 1590.0, "filled": 1590.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.018434974631993433, "passes": true}, "odd.women.b1946_1980.paime_p90": {"truth": 2696.0, "filled": 2697.0, "gap": 0.00037085110753221073, "seed_sd": 0.0, "tolerance": 0.018162919937725473, "passes": true}, "odd.women.b1956_1965.paime_p10": {"truth": 140.0, "filled": 140.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0987946317785998, "passes": true}, "odd.women.b1956_1965.paime_p25": {"truth": 426.0, "filled": 425.0, "gap": -0.0023501773449536856, "seed_sd": 0.0, "tolerance": 0.05702004966656188, "passes": true}, "odd.women.b1956_1965.paime_p50": {"truth": 1000.0, "filled": 1000.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0353321966023362, "passes": true}, "odd.women.b1956_1965.paime_p75": {"truth": 1842.0, "filled": 1846.0, "gap": 0.0021691982475449123, "seed_sd": 0.0, "tolerance": 0.026005856561860215, "passes": true}, "odd.women.b1956_1965.paime_p90": {"truth": 2793.0, "filled": 2795.0999999999985, "gap": 0.0007515971793115028, "seed_sd": 0.0, "tolerance": 0.027293077620429543, "passes": true}, "odd.women.b1966_1980.paime_p10": {"truth": 80.0, "filled": 80.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.06283109225080236, "passes": true}, "odd.women.b1966_1980.paime_p25": {"truth": 211.0, "filled": 211.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.033157126538046186, "passes": true}, "odd.women.b1966_1980.paime_p50": {"truth": 463.0, "filled": 464.0, "gap": 0.0021574981400211968, "seed_sd": 0.0, "tolerance": 0.02601990880890944, "passes": true}, "odd.women.b1966_1980.paime_p75": {"truth": 866.75, "filled": 867.0, "gap": 0.0002883922154088836, "seed_sd": 0.0, "tolerance": 0.02735691355887302, "passes": true}, "odd.women.b1966_1980.paime_p90": {"truth": 1387.0, "filled": 1388.0, "gap": 0.0007207207519188685, "seed_sd": 0.0, "tolerance": 0.027821321010049294, "passes": true}}}}, "primary": {"passes": false, "n_gating": 183, "n_failing": 7, "cells": {"odd.men.a22_29.atcap": {"truth": 0.014676604027739744, "filled": 0.015478991505216875, "gap": 0.053229054633069595, "seed_sd": 0.00022562078995838478, "tolerance": 0.18023370223284765, "passes": true}, "odd.men.a22_29.level": {"truth": 0.23657185599421368, "filled": 0.23717255395952153, "gap": 0.0025359593680274184, "seed_sd": 0.0002515217198283508, "tolerance": 0.02171146931539837, "passes": true}, "odd.men.a22_29.q10": {"truth": 0.040229885057471264, "filled": 0.041617456321049816, "gap": 0.03390957352930313, "seed_sd": 0.0003830112083455766, "tolerance": 0.07329266358707544, "passes": true}, "odd.men.a22_29.q50": {"truth": 0.24222222222222223, "filled": 0.24384832532234682, "gap": 0.006690836030639913, "seed_sd": 0.0005141419280494356, "tolerance": 0.022980444394274636, "passes": true}, "odd.men.a22_29.q90": {"truth": 0.5661157024793388, "filled": 0.5690728618295566, "gap": 0.005209999699271828, "seed_sd": 0.0011631471597934602, "tolerance": 0.02353059621893112, "passes": true}, "odd.men.a22_29.r1": {"truth": 0.8114342592359591, "filled": 0.7918666865747263, "gap": -0.01956757266123288, "seed_sd": 0.0008536254079533948, "tolerance": 0.007273490161268873, "passes": false}, "odd.men.a22_29.r2": {"truth": 0.7090594261869381, "filled": 0.7086612870078752, "gap": -0.0003981391790628397, "seed_sd": 0.0017514876624430678, "tolerance": 0.014356564488812985, "passes": true}, "odd.men.a22_29.r3": {"truth": 0.6020834292042946, "filled": 0.5897635473854892, "gap": -0.01231988181880539, "seed_sd": 0.001092216462324124, "tolerance": 0.013280307041936976, "passes": true}, "odd.men.a22_29.r4": {"truth": 0.5990582702554191, "filled": 0.5975865098462816, "gap": -0.0014717604091375458, "seed_sd": 0.001832401877059017, "tolerance": 0.02119963449400435, "passes": true}, "odd.men.a22_29.wint": {"truth": 0.08288543140028289, "filled": 0.09262022630834513, "gap": 0.11104823478796844, "seed_sd": 0.0025397861808929443}, "odd.men.a22_29.zexit": {"truth": 0.4099387400566883, "filled": 0.42119411173082194, "gap": 0.027086066376702744, "seed_sd": 0.003339605159177347, "tolerance": 0.052311207438163365, "passes": true}, "odd.men.a22_29.zint": {"truth": 0.027589420573962787, "filled": 0.032563791496609575, "gap": 0.16576859410855027, "seed_sd": 0.00038799743953147253, "tolerance": 0.12140160693630424, "passes": false}, "odd.men.a22_74.atcap": {"truth": 0.10371778137953785, "filled": 0.10405772152852044, "gap": 0.0032721899121184173, "seed_sd": 0.0001421100987836538, "tolerance": 0.03690628859850606, "passes": true}, "odd.men.a22_74.level": {"truth": 0.35325148977531445, "filled": 0.3536267889985615, "gap": 0.0010618497406973404, "seed_sd": 0.00015811254712585015, "tolerance": 0.01106822547283481, "passes": true}, "odd.men.a22_74.q10": {"truth": 0.06111111111111111, "filled": 0.06104982070649271, "gap": -0.0010034371684821686, "seed_sd": 0.0001901280444885172, "tolerance": 0.03462864626687699, "passes": true}, "odd.men.a22_74.q50": {"truth": 0.37052341597796146, "filled": 0.3714999618524454, "gap": 0.0026321177134208673, "seed_sd": 0.00027634709713943287, "tolerance": 0.012463773427145929, "passes": true}, "odd.men.a22_74.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.003451823067723808, "passes": true}, "odd.men.a22_74.r1": {"truth": 0.8935990156167083, "filled": 0.8910980192174813, "gap": -0.0025009963992269624, "seed_sd": 0.0001975630980355699, "tolerance": 0.00258049589094515, "passes": true}, "odd.men.a22_74.r2": {"truth": 0.8291027164330249, "filled": 0.8277733291882811, "gap": -0.0013293872447438515, "seed_sd": 0.00042543786212773815, "tolerance": 0.004576681103770273, "passes": true}, "odd.men.a22_74.r3": {"truth": 0.7806492619651291, "filled": 0.7768047594054334, "gap": -0.003844502559695706, "seed_sd": 0.00021680874531272687, "tolerance": 0.004722018684826323, "passes": true}, "odd.men.a22_74.r4": {"truth": 0.7414979369803346, "filled": 0.7366250578680585, "gap": -0.004872879112276074, "seed_sd": 0.0006218303381177339, "tolerance": 0.007000982741779954, "passes": true}, "odd.men.a22_74.wint": {"truth": 0.05745341614906832, "filled": 0.059355590062111795, "gap": 0.03257183917090156, "seed_sd": 0.0007236709175504945, "tolerance": 0.07337302886151653, "passes": true}, "odd.men.a22_74.zexit": {"truth": 0.44752843148294114, "filled": 0.44796616372030174, "gap": 0.0009776324158056182, "seed_sd": 0.001535188345791419, "tolerance": 0.01916127382099423, "passes": true}, "odd.men.a22_74.zint": {"truth": 0.019892968610837895, "filled": 0.02071057380286708, "gap": 0.04027804843040528, "seed_sd": 0.0001501775466059639, "tolerance": 0.05481211657295162, "passes": true}, "odd.men.a30_44.atcap": {"truth": 0.10574805846620577, "filled": 0.10532608891605424, "gap": -0.003998311660010412, "seed_sd": 0.00031281429807503787, "tolerance": 0.049095580366705964, "passes": true}, "odd.men.a30_44.level": {"truth": 0.3982899917120792, "filled": 0.3988284449313848, "gap": 0.0013509994912968004, "seed_sd": 0.0002028287101855178, "tolerance": 0.015220021463125325, "passes": true}, "odd.men.a30_44.q10": {"truth": 0.08706467661691543, "filled": 0.08610589761196305, "gap": -0.011073345529330147, "seed_sd": 0.0004749422573459383, "tolerance": 0.047669064141889934, "passes": true}, "odd.men.a30_44.q50": {"truth": 0.41333333333333333, "filled": 0.41458838788433666, "gap": 0.0030318216812166288, "seed_sd": 0.00034599819975770856, "tolerance": 0.016660415997544607, "passes": true}, "odd.men.a30_44.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0037405831290436837, "passes": true}, "odd.men.a30_44.r1": {"truth": 0.9000861304840623, "filled": 0.8998266022341059, "gap": -0.00025952824995634227, "seed_sd": 0.0004413185664793287, "tolerance": 0.003556208950302802, "passes": true}, "odd.men.a30_44.r2": {"truth": 0.8479233224100629, "filled": 0.8483810977099816, "gap": 0.0004577752999187501, "seed_sd": 0.000997098561784646, "tolerance": 0.006500650570758907, "passes": true}, "odd.men.a30_44.r3": {"truth": 0.8030903893396322, "filled": 0.7992399426798853, "gap": -0.0038504466597468756, "seed_sd": 0.00048283750320468534, "tolerance": 0.006087370619188444, "passes": true}, "odd.men.a30_44.r4": {"truth": 0.7869108820616492, "filled": 0.7810755971849026, "gap": -0.00583528487674656, "seed_sd": 0.0010191541897079349, "tolerance": 0.009218001294372115, "passes": true}, "odd.men.a30_44.wint": {"truth": 0.07239145006330527, "filled": 0.07317904222834587, "gap": 0.010820872247405244, "seed_sd": 0.0017316235445208798, "tolerance": 0.12001713872675526, "passes": true}, "odd.men.a30_44.zexit": {"truth": 0.4342024282437196, "filled": 0.4297935433335199, "gap": -0.01020588827807345, "seed_sd": 0.002303556443411403, "tolerance": 0.031953074094816486, "passes": true}, "odd.men.a30_44.zint": {"truth": 0.017933882760031997, "filled": 0.017023794020798837, "gap": -0.05207980146296265, "seed_sd": 0.00021688098899260444, "tolerance": 0.08989768311292742, "passes": true}, "odd.men.a45_59.atcap": {"truth": 0.15826463477931516, "filled": 0.15934572375910133, "gap": 0.0068076693717182835, "seed_sd": 0.0003390418371717721, "tolerance": 0.04711609868499565, "passes": true}, "odd.men.a45_59.level": {"truth": 0.4279852808128974, "filled": 0.42780971681639934, "gap": -0.00041029452017671275, "seed_sd": 0.00024109620070850895, "tolerance": 0.015834008317210692, "passes": true}, "odd.men.a45_59.q10": {"truth": 0.08868501529051988, "filled": 0.08761303120469977, "gap": -0.012161193147874894, "seed_sd": 0.00042934976933140727, "tolerance": 0.055701610600039544, "passes": true}, "odd.men.a45_59.q50": {"truth": 0.4701492537313433, "filled": 0.4705012588693065, "gap": 0.0007484291980476288, "seed_sd": 0.000414975117828889, "tolerance": 0.020385938992396834, "passes": true}, "odd.men.a45_59.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0}, "odd.men.a45_59.r1": {"truth": 0.9077661818197654, "filled": 0.9078114869048537, "gap": 4.530508508826525e-05, "seed_sd": 0.00034430581774453007, "tolerance": 0.004032009924506552, "passes": true}, "odd.men.a45_59.r2": {"truth": 0.850421444941529, "filled": 0.8484356528621042, "gap": -0.001985792079424842, "seed_sd": 0.0006903711344554875, "tolerance": 0.0074262368998509335, "passes": true}, "odd.men.a45_59.r3": {"truth": 0.8146289743583418, "filled": 0.8121562172502159, "gap": -0.0024727571081258892, "seed_sd": 0.00036683181976336675, "tolerance": 0.007198396861260139, "passes": true}, "odd.men.a45_59.r4": {"truth": 0.7598303885400162, "filled": 0.756280816003992, "gap": -0.0035495725360241703, "seed_sd": 0.0014138669035669439, "tolerance": 0.012107705260459355, "passes": true}, "odd.men.a45_59.wint": {"truth": 0.05825168693530555, "filled": 0.0575464145476726, "gap": -0.012181220579572383, "seed_sd": 0.0013702186370924433, "tolerance": 0.13344110142481172, "passes": true}, "odd.men.a45_59.zexit": {"truth": 0.45185706741770815, "filled": 0.45060728744939266, "gap": -0.0027697066604916998, "seed_sd": 0.002801554802201966, "tolerance": 0.036314919811016276, "passes": true}, "odd.men.a45_59.zint": {"truth": 0.015835938476928848, "filled": 0.01581418031761911, "gap": -0.0013749182351325828, "seed_sd": 0.00022747480774057554, "tolerance": 0.09287476169914739, "passes": true}, "odd.men.a60_74.atcap": {"truth": 0.10024796315975912, "filled": 0.09946648841993258, "gap": -0.00782596073680164, "seed_sd": 0.0004015553201233456, "tolerance": 0.12472307183935011, "passes": true}, "odd.men.a60_74.level": {"truth": 0.20453937198442498, "filled": 0.20540302537010527, "gap": 0.004213541556197242, "seed_sd": 0.00035587347264424323, "tolerance": 0.0522939828282486, "passes": true}, "odd.men.a60_74.q10": {"truth": 0.01529051987767584, "filled": 0.015350881208514536, "gap": 0.003939859587278605, "seed_sd": 0.00029774282812945784, "tolerance": 0.20186846491463123, "passes": true}, "odd.men.a60_74.q50": {"truth": 0.21666666666666667, "filled": 0.2233566033417258, "gap": 0.030409538134932967, "seed_sd": 0.001070762123976171, "tolerance": 0.07446457229432804, "passes": true}, "odd.men.a60_74.q90": {"truth": 1.0, "filled": 0.9959601739528496, "gap": -0.004048008188114695, "seed_sd": 0.0009839142641652522, "tolerance": 0.04069424979546094, "passes": true}, "odd.men.a60_74.r1": {"truth": 0.8712782553777247, "filled": 0.8633097014979182, "gap": -0.00796855387980655, "seed_sd": 0.001064453719918088, "tolerance": 0.009750532797445479, "passes": true}, "odd.men.a60_74.r2": {"truth": 0.7888015745113253, "filled": 0.7830136813243864, "gap": -0.005787893186938842, "seed_sd": 0.0018046466076060675, "tolerance": 0.020757725353477027, "passes": true}, "odd.men.a60_74.r3": {"truth": 0.7283746799072385, "filled": 0.7250001781709396, "gap": -0.0033745017362988294, "seed_sd": 0.000782393180230533, "tolerance": 0.02064690204069302, "passes": true}, "odd.men.a60_74.r4": {"truth": 0.6781549627202678, "filled": 0.6692115763481665, "gap": -0.008943386372101236, "seed_sd": 0.0036651414112914395, "tolerance": 0.03820786124425089, "passes": true}, "odd.men.a60_74.wint": {"truth": 0.03084255319148936, "filled": 0.03232170212765957, "gap": 0.04684356352844654, "seed_sd": 0.0008910456258281209}, "odd.men.a60_74.zexit": {"truth": 0.5047452182800409, "filled": 0.5044313038399767, "gap": -0.0006221200024142393, "seed_sd": 0.003042999000388235, "tolerance": 0.04434932710092839, "passes": true}, "odd.men.a60_74.zint": {"truth": 0.027132152458672565, "filled": 0.0301380441207314, "gap": 0.10506883573756909, "seed_sd": 0.0007776869923929043, "tolerance": 0.17758159695027936, "passes": true}, "odd.men.b1936_1940.aime_p10": {"truth": 346.0, "filled": 345.15, "gap": -0.002459669908239981, "seed_sd": 1.2258187382102497, "tolerance": 0.30817736367768905, "passes": true}, "odd.men.b1936_1940.aime_p25": {"truth": 1131.0, "filled": 1129.85, "gap": -0.001017316583746819, "seed_sd": 1.1367080817685316, "tolerance": 0.15436566295506535, "passes": true}, "odd.men.b1936_1940.aime_p50": {"truth": 2430.0, "filled": 2428.85, "gap": -0.00047336304741829593, "seed_sd": 1.1821033884786183, "tolerance": 0.0822816707411384, "passes": true}, "odd.men.b1936_1940.aime_p75": {"truth": 3582.0, "filled": 3580.4, "gap": -0.0004467776238730181, "seed_sd": 1.729009451741235, "tolerance": 0.04915519601813279, "passes": true}, "odd.men.b1936_1940.aime_p90": {"truth": 4382.0, "filled": 4378.0, "gap": -0.0009132420726043478, "seed_sd": 2.2710998958306754, "tolerance": 0.03801502836217192, "passes": true}, "odd.men.b1936_1945.aime_p10": {"truth": 342.8000000000002, "filled": 340.81000000000006, "gap": -0.005822049475948887, "seed_sd": 0.6820248490611044, "tolerance": 0.19387427864829543, "passes": true}, "odd.men.b1936_1945.aime_p25": {"truth": 1152.0, "filled": 1153.625, "gap": 0.0014095963299034509, "seed_sd": 1.190698600778021, "tolerance": 0.09928513681983075, "passes": true}, "odd.men.b1936_1945.aime_p50": {"truth": 2625.0, "filled": 2623.35, "gap": -0.0006287690624144915, "seed_sd": 1.598519051464429, "tolerance": 0.05013200755321697, "passes": true}, "odd.men.b1936_1945.aime_p75": {"truth": 3991.5, "filled": 3994.9, "gap": 0.0008514475121224052, "seed_sd": 2.1860803664140644, "tolerance": 0.030282584009309735, "passes": true}, "odd.men.b1936_1945.aime_p90": {"truth": 5075.0, "filled": 5068.030000000001, "gap": -0.001374342991608657, "seed_sd": 2.1406098491588508, "tolerance": 0.029246187918705275, "passes": true}, "odd.men.b1941_1945.aime_p10": {"truth": 338.0, "filled": 336.46000000000004, "gap": -0.004566624192004376, "seed_sd": 1.8222022534920688, "tolerance": 0.2434370939708618, "passes": true}, "odd.men.b1941_1945.aime_p25": {"truth": 1174.25, "filled": 1176.5, "gap": 0.0019142832603122883, "seed_sd": 1.8009500416812174, "tolerance": 0.15717605214643726, "passes": true}, "odd.men.b1941_1945.aime_p50": {"truth": 2821.5, "filled": 2822.425, "gap": 0.0003277860737984639, "seed_sd": 2.341136700970616, "tolerance": 0.06794319334096698, "passes": true}, "odd.men.b1941_1945.aime_p75": {"truth": 4421.0, "filled": 4422.65, "gap": 0.00037314910000851853, "seed_sd": 2.286113686909224, "tolerance": 0.039611439643930886, "passes": true}, "odd.men.b1941_1945.aime_p90": {"truth": 5528.300000000001, "filled": 5531.785000000001, "gap": 0.0006301940926025651, "seed_sd": 2.464970374541801, "tolerance": 0.024092287379584104, "passes": true}, "odd.men.b1946_1955.paime_p10": {"truth": 308.0, "filled": 308.0, "gap": 0.0, "seed_sd": 0.6488856845230502, "tolerance": 0.11641286886749184, "passes": true}, "odd.men.b1946_1955.paime_p25": {"truth": 1026.0, "filled": 1027.9, "gap": 0.0018501392881615786, "seed_sd": 1.5861240410775606, "tolerance": 0.07923354406873329, "passes": true}, "odd.men.b1946_1955.paime_p50": {"truth": 2614.0, "filled": 2611.8, "gap": -0.0008419763978597672, "seed_sd": 2.375311613906246, "tolerance": 0.03884145918889322, "passes": true}, "odd.men.b1946_1955.paime_p75": {"truth": 4390.0, "filled": 4389.45, "gap": -0.00012529258682825173, "seed_sd": 2.874113135523631, "tolerance": 0.02464882648778213, "passes": true}, "odd.men.b1946_1955.paime_p90": {"truth": 5810.0, "filled": 5808.47, "gap": -0.0002633737503892064, "seed_sd": 3.151123442102177, "tolerance": 0.01935089604023114, "passes": true}, "odd.men.b1946_1980.paime_p10": {"truth": 175.0, "filled": 174.95, "gap": -0.00028575510981720953, "seed_sd": 0.5104177855340405, "tolerance": 0.05432313062129484, "passes": true}, "odd.men.b1946_1980.paime_p25": {"truth": 487.0, "filled": 487.15, "gap": 0.00030796078876083044, "seed_sd": 0.48936048492959283, "tolerance": 0.03064646902486962, "passes": true}, "odd.men.b1946_1980.paime_p50": {"truth": 1245.0, "filled": 1246.425, "gap": 0.0011439237828883009, "seed_sd": 0.5910516942393091, "tolerance": 0.025757396517852603, "passes": true}, "odd.men.b1946_1980.paime_p75": {"truth": 2644.0, "filled": 2645.525, "gap": 0.0005766113374088278, "seed_sd": 1.4439073450372888, "tolerance": 0.0206417529534006, "passes": true}, "odd.men.b1946_1980.paime_p90": {"truth": 4310.0, "filled": 4311.220000000001, "gap": 0.0002830225903398542, "seed_sd": 1.667206929977776, "tolerance": 0.01797004074133569, "passes": true}, "odd.men.b1956_1965.paime_p10": {"truth": 253.0, "filled": 252.76000000000005, "gap": -0.0009490668222653653, "seed_sd": 0.9483503682988224, "tolerance": 0.10325340800222761, "passes": true}, "odd.men.b1956_1965.paime_p25": {"truth": 765.0, "filled": 767.9, "gap": 0.00378368250996175, "seed_sd": 1.252366181526625, "tolerance": 0.06714959239176221, "passes": true}, "odd.men.b1956_1965.paime_p50": {"truth": 1764.0, "filled": 1765.1, "gap": 0.0006233884194966066, "seed_sd": 1.9708400559953587, "tolerance": 0.03783036836459788, "passes": true}, "odd.men.b1956_1965.paime_p75": {"truth": 2958.0, "filled": 2956.9, "gap": -0.00037194204895474314, "seed_sd": 2.125038699338017, "tolerance": 0.02582244416868233, "passes": true}, "odd.men.b1956_1965.paime_p90": {"truth": 4111.0, "filled": 4111.450000000001, "gap": 0.00010945642732984595, "seed_sd": 1.6693837501499207, "tolerance": 0.022664845011086624, "passes": true}, "odd.men.b1966_1980.paime_p10": {"truth": 112.0, "filled": 112.4, "gap": 0.0035650661644970327, "seed_sd": 0.5982430416161189, "tolerance": 0.07763263566160054, "passes": true}, "odd.men.b1966_1980.paime_p25": {"truth": 303.25, "filled": 303.7625, "gap": 0.0016885982472416572, "seed_sd": 0.558869865277946, "tolerance": 0.04145124387911718, "passes": true}, "odd.men.b1966_1980.paime_p50": {"truth": 655.0, "filled": 656.55, "gap": 0.0023636166697622585, "seed_sd": 0.8413648060396307, "tolerance": 0.02859495381248372, "passes": true}, "odd.men.b1966_1980.paime_p75": {"truth": 1225.0, "filled": 1226.95, "gap": 0.0015905711055381744, "seed_sd": 1.0990426455975697, "tolerance": 0.024541637055534683, "passes": true}, "odd.men.b1966_1980.paime_p90": {"truth": 1888.9000000000015, "filled": 1889.7400000000002, "gap": 0.0004446044152581763, "seed_sd": 2.1283549367626273, "tolerance": 0.025554311796701996, "passes": true}, "odd.women.a22_29.atcap": {"truth": 0.005728386357627567, "filled": 0.005840506534186164, "gap": 0.019383650297911004, "seed_sd": 0.00014710922581249458}, "odd.women.a22_29.level": {"truth": 0.1854665923300802, "filled": 0.18547670162135138, "gap": 5.450585811095365e-05, "seed_sd": 0.000224942861263494, "tolerance": 0.018517427407013797, "passes": true}, "odd.women.a22_29.q10": {"truth": 0.028735632183908046, "filled": 0.02897917143511101, "gap": 0.00843945336115448, "seed_sd": 0.0002739223160825707, "tolerance": 0.0684466769976442, "passes": true}, "odd.women.a22_29.q50": {"truth": 0.191131498470948, "filled": 0.19147020675974674, "gap": 0.0017705534118206412, "seed_sd": 0.00042356535871290117, "tolerance": 0.021849929026002995, "passes": true}, "odd.women.a22_29.q90": {"truth": 0.46005509641873277, "filled": 0.46155634393835354, "gap": 0.003257878063804065, "seed_sd": 0.0007612439450409235, "tolerance": 0.020751504651450387, "passes": true}, "odd.women.a22_29.r1": {"truth": 0.7942016511246733, "filled": 0.7811893738584953, "gap": -0.01301227726617804, "seed_sd": 0.0009183067288461948, "tolerance": 0.006445547562698701, "passes": false}, "odd.women.a22_29.r2": {"truth": 0.6731585499668502, "filled": 0.6832118075212995, "gap": 0.010053257554449302, "seed_sd": 0.0015380755525540044, "tolerance": 0.013111971249341917, "passes": true}, "odd.women.a22_29.r3": {"truth": 0.5535614164842262, "filled": 0.5458382326973334, "gap": -0.00772318378689274, "seed_sd": 0.001022663276913233, "tolerance": 0.012054115899361516, "passes": true}, "odd.women.a22_29.r4": {"truth": 0.552312691565311, "filled": 0.5543544199888721, "gap": 0.0020417284235610955, "seed_sd": 0.0022237898651966616, "tolerance": 0.01982762523559322, "passes": true}, "odd.women.a22_29.wint": {"truth": 0.08501885498800137, "filled": 0.09672608844703463, "gap": 0.12901009825016718, "seed_sd": 0.002577622041541449, "tolerance": 0.15037337545812676, "passes": true}, "odd.women.a22_29.zexit": {"truth": 0.41757828810020875, "filled": 0.4201440501043841, "gap": 0.006125585793858135, "seed_sd": 0.0032598959894505576, "tolerance": 0.041287246240518535, "passes": true}, "odd.women.a22_29.zint": {"truth": 0.0315547703180212, "filled": 0.03659054770318021, "gap": 0.14806517134934172, "seed_sd": 0.0004775866131467123, "tolerance": 0.09342121518593281, "passes": false}, "odd.women.a22_74.atcap": {"truth": 0.029398307023564402, "filled": 0.02945745381492562, "gap": 0.002009890295517458, "seed_sd": 0.00011259923212691128, "tolerance": 0.06709882226274566, "passes": true}, "odd.women.a22_74.level": {"truth": 0.2372854094489719, "filled": 0.23753619967326758, "gap": 0.001056355661994468, "seed_sd": 8.268291240265756e-05, "tolerance": 0.010902725750872375, "passes": true}, "odd.women.a22_74.q10": {"truth": 0.03777777777777778, "filled": 0.037823300526436246, "gap": 0.0012042884885090643, "seed_sd": 0.0001235564277909059, "tolerance": 0.031129845660342752, "passes": true}, "odd.women.a22_74.q50": {"truth": 0.24875621890547264, "filled": 0.24890363927672238, "gap": 0.0005924543566777629, "seed_sd": 0.00017148945504414184, "tolerance": 0.012092425838785583, "passes": true}, "odd.women.a22_74.q90": {"truth": 0.6444444444444445, "filled": 0.6455617608911275, "gap": 0.0017322656611419296, "seed_sd": 0.0005288221041851096, "tolerance": 0.013338685039058783, "passes": true}, "odd.women.a22_74.r1": {"truth": 0.8812684347612898, "filled": 0.8785632456320756, "gap": -0.002705189129214247, "seed_sd": 0.00030066051021595305, "tolerance": 0.002475922564645967, "passes": false}, "odd.women.a22_74.r2": {"truth": 0.8056395091372208, "filled": 0.8086345514864973, "gap": 0.0029950423492765, "seed_sd": 0.0005474787133861371, "tolerance": 0.004772836686607259, "passes": true}, "odd.women.a22_74.r3": {"truth": 0.7468316241328095, "filled": 0.7425035732187585, "gap": -0.004328050914050974, "seed_sd": 0.0002996302075456965, "tolerance": 0.005171487109756781, "passes": true}, "odd.women.a22_74.r4": {"truth": 0.7116744894807431, "filled": 0.7070081700012946, "gap": -0.004666319479448511, "seed_sd": 0.0008952863437053086, "tolerance": 0.007666262733486471, "passes": true}, "odd.women.a22_74.wint": {"truth": 0.051544874036178946, "filled": 0.05278242947314182, "gap": 0.023725591419143655, "seed_sd": 0.0007715150904738836, "tolerance": 0.07133033012317215, "passes": true}, "odd.women.a22_74.zexit": {"truth": 0.45845752128015943, "filled": 0.4576647226063578, "gap": -0.0017307709249479997, "seed_sd": 0.0011185395892386197, "tolerance": 0.016039678883419197, "passes": true}, "odd.women.a22_74.zint": {"truth": 0.022201124403524123, "filled": 0.023460056043049, "gap": 0.05515629569622549, "seed_sd": 0.00019832742657188788, "tolerance": 0.04476234327525859, "passes": false}, "odd.women.a30_44.atcap": {"truth": 0.034619662054697874, "filled": 0.03434400495858832, "gap": -0.007994312835381212, "seed_sd": 0.0001719755329787735, "tolerance": 0.08510646139536093, "passes": true}, "odd.women.a30_44.level": {"truth": 0.2579924798566703, "filled": 0.25856952575486103, "gap": 0.002234179563932237, "seed_sd": 0.0001654878317954069, "tolerance": 0.01578879525774269, "passes": true}, "odd.women.a30_44.q10": {"truth": 0.04228855721393035, "filled": 0.04248416876478217, "gap": 0.004614972463625744, "seed_sd": 0.0002927857689515677, "tolerance": 0.04573918824793483, "passes": true}, "odd.women.a30_44.q50": {"truth": 0.26859504132231404, "filled": 0.26862745098039215, "gap": 0.00012065637080271863, "seed_sd": 0.00022554139509321257, "tolerance": 0.018488627942446087, "passes": true}, "odd.women.a30_44.q90": {"truth": 0.6716417910447762, "filled": 0.6719058518348973, "gap": 0.0003930799103709637, "seed_sd": 0.0005591699861576255, "tolerance": 0.018806950089013175, "passes": true}, "odd.women.a30_44.r1": {"truth": 0.888241419643356, "filled": 0.888364648373031, "gap": 0.00012322872967496235, "seed_sd": 0.00037398027635182656, "tolerance": 0.0034876380830913202, "passes": true}, "odd.women.a30_44.r2": {"truth": 0.8242277470219269, "filled": 0.8261260370432127, "gap": 0.0018982900212858311, "seed_sd": 0.0008564417910465683, "tolerance": 0.007078708407960211, "passes": true}, "odd.women.a30_44.r3": {"truth": 0.7685608654079126, "filled": 0.7647437596543007, "gap": -0.003817105753611827, "seed_sd": 0.00045573186883397765, "tolerance": 0.0073196290204859075, "passes": true}, "odd.women.a30_44.r4": {"truth": 0.7490435327097834, "filled": 0.7421544589183829, "gap": -0.0068890737914004685, "seed_sd": 0.001251592453325158, "tolerance": 0.010187042430903364, "passes": true}, "odd.women.a30_44.wint": {"truth": 0.06073485056210584, "filled": 0.06228544008774335, "gap": 0.02521001436438297, "seed_sd": 0.0012284904999587422, "tolerance": 0.10093703169083498, "passes": true}, "odd.women.a30_44.zexit": {"truth": 0.45066799061202384, "filled": 0.45179860985737497, "gap": 0.002505621451870721, "seed_sd": 0.002515913111566333, "tolerance": 0.024719366356404402, "passes": true}, "odd.women.a30_44.zint": {"truth": 0.022222222222222223, "filled": 0.022268645656248635, "gap": 0.002086875490999507, "seed_sd": 0.0003599125241648141, "tolerance": 0.07029190354044747, "passes": true}, "odd.women.a45_59.atcap": {"truth": 0.04053483605515179, "filled": 0.04080148917927712, "gap": 0.006556826330295085, "seed_sd": 0.00019608552018686722, "tolerance": 0.09107631963722376, "passes": true}, "odd.women.a45_59.level": {"truth": 0.28104586423928846, "filled": 0.2810774396033896, "gap": 0.00011234319558894867, "seed_sd": 0.00015511405161526974, "tolerance": 0.01605667435242628, "passes": true}, "odd.women.a45_59.q10": {"truth": 0.053516819571865444, "filled": 0.052449683375295666, "gap": -0.020141690886250174, "seed_sd": 0.00032583523000722845, "tolerance": 0.051646333169991114, "passes": true}, "odd.women.a45_59.q50": {"truth": 0.29338842975206614, "filled": 0.2937277790493629, "gap": 0.0011559869409125678, "seed_sd": 0.0002448197822313442, "tolerance": 0.0182941133196311, "passes": true}, "odd.women.a45_59.q90": {"truth": 0.7300275482093664, "filled": 0.7307142748149844, "gap": 0.0009402437109500283, "seed_sd": 0.0008138649714838994, "tolerance": 0.0208569023153089, "passes": true}, "odd.women.a45_59.r1": {"truth": 0.9101006329634369, "filled": 0.9087391750153332, "gap": -0.001361457948103717, "seed_sd": 0.000419730811063702, "tolerance": 0.0037361334549905496, "passes": true}, "odd.women.a45_59.r2": {"truth": 0.8498173267452248, "filled": 0.8515147642421521, "gap": 0.001697437496927301, "seed_sd": 0.0008879958346065681, "tolerance": 0.0074184353060767995, "passes": true}, "odd.women.a45_59.r3": {"truth": 0.8096130207660378, "filled": 0.8055548329310239, "gap": -0.004058187835013882, "seed_sd": 0.0005770848358180316, "tolerance": 0.007740061426920219, "passes": true}, "odd.women.a45_59.r4": {"truth": 0.7620167777938601, "filled": 0.7581822504292892, "gap": -0.0038345273645709055, "seed_sd": 0.0014198400183539773, "tolerance": 0.012512985825002505, "passes": true}, "odd.women.a45_59.wint": {"truth": 0.050115932427956277, "filled": 0.046954289499834385, "gap": -0.06516440543993252, "seed_sd": 0.001114473359463049, "tolerance": 0.1271041647107106, "passes": true}, "odd.women.a45_59.zexit": {"truth": 0.4693136110029843, "filled": 0.46718080965356173, "gap": -0.004554869713656262, "seed_sd": 0.002121600873384182, "tolerance": 0.030024476078798885, "passes": true}, "odd.women.a45_59.zint": {"truth": 0.01632776600720909, "filled": 0.01605233469994222, "gap": -0.017012791468189903, "seed_sd": 0.0003180980237889735, "tolerance": 0.09017841563442361, "passes": true}, "odd.women.a60_74.atcap": {"truth": 0.01812919896640827, "filled": 0.01870662717455454, "gap": 0.03135401441666019, "seed_sd": 0.00040142732607189924}, "odd.women.a60_74.level": {"truth": 0.12926943691615103, "filled": 0.1293852858038575, "gap": 0.0008957802450684227, "seed_sd": 0.00036005331017849995, "tolerance": 0.04408494999775865, "passes": true}, "odd.women.a60_74.q10": {"truth": 0.017412935323383085, "filled": 0.016837567711909668, "gap": -0.03360077619576973, "seed_sd": 0.0003065216899310214, "tolerance": 0.13013445963696416, "passes": true}, "odd.women.a60_74.q50": {"truth": 0.15517241379310345, "filled": 0.15723735408560308, "gap": 0.0132196274053622, "seed_sd": 0.0008103021569256636, "tolerance": 0.05365060769995596, "passes": true}, "odd.women.a60_74.q90": {"truth": 0.5360696517412935, "filled": 0.5337356374456398, "gap": -0.00436344449308268, "seed_sd": 0.0018056770118273073, "tolerance": 0.04960336282899433, "passes": true}, "odd.women.a60_74.r1": {"truth": 0.8610627385333514, "filled": 0.8496491895289273, "gap": -0.01141354900442404, "seed_sd": 0.001253637642873998, "tolerance": 0.009751654703991025, "passes": false}, "odd.women.a60_74.r2": {"truth": 0.7722738686836694, "filled": 0.7795050897584435, "gap": 0.00723122107477403, "seed_sd": 0.003077248521934921, "tolerance": 0.022174026993056747, "passes": true}, "odd.women.a60_74.r3": {"truth": 0.7141078828415282, "filled": 0.7020798077569574, "gap": -0.01202807508457071, "seed_sd": 0.00218141956853534, "tolerance": 0.018479646240433655, "passes": true}, "odd.women.a60_74.r4": {"truth": 0.6501204938024134, "filled": 0.6374646350510534, "gap": -0.012655858751359994, "seed_sd": 0.005366315338935656, "tolerance": 0.036314307016843086, "passes": true}, "odd.women.a60_74.wint": {"truth": 0.023107482596493787, "filled": 0.023204067500091116, "gap": 0.00417109958221884, "seed_sd": 0.0009728381028477894}, "odd.women.a60_74.zexit": {"truth": 0.5155483759303063, "filled": 0.5075809150175965, "gap": -0.01557500512456278, "seed_sd": 0.004491939020397144, "tolerance": 0.03896610501214408, "passes": true}, "odd.women.a60_74.zint": {"truth": 0.022290698541288734, "filled": 0.026766233443502895, "gap": 0.1829716612979282, "seed_sd": 0.0008912211575906045}, "odd.women.b1936_1940.aime_p10": {"truth": 82.29999999999995, "filled": 83.05, "gap": 0.009071728376279786, "seed_sd": 0.22360679774997896, "tolerance": 0.32371200850504067, "passes": true}, "odd.women.b1936_1940.aime_p25": {"truth": 287.75, "filled": 287.8375, "gap": 0.0003040371817455423, "seed_sd": 0.48851951423609846, "tolerance": 0.20139115911902655, "passes": true}, "odd.women.b1936_1940.aime_p50": {"truth": 786.0, "filled": 785.975, "gap": -3.1807121617433154e-05, "seed_sd": 0.7340407273943894, "tolerance": 0.12524071173214943, "passes": true}, "odd.women.b1936_1940.aime_p75": {"truth": 1566.0, "filled": 1566.675, "gap": 0.00043094161408152587, "seed_sd": 1.8745174817732921, "tolerance": 0.10110924143579125, "passes": true}, "odd.women.b1936_1940.aime_p90": {"truth": 2438.7000000000007, "filled": 2436.605, "gap": -0.0008594334627067823, "seed_sd": 3.0194936835939212, "tolerance": 0.08967006892401257, "passes": true}, "odd.women.b1936_1945.aime_p10": {"truth": 98.0, "filled": 98.13000000000002, "gap": 0.0013256515478303754, "seed_sd": 0.3197038102929258, "tolerance": 0.1995625804481066, "passes": true}, "odd.women.b1936_1945.aime_p25": {"truth": 345.0, "filled": 344.8, "gap": -0.0005798782418224846, "seed_sd": 0.6958523739384594, "tolerance": 0.11108088073804002, "passes": true}, "odd.women.b1936_1945.aime_p50": {"truth": 932.0, "filled": 930.0, "gap": -0.002148228538289665, "seed_sd": 1.2565617248750864, "tolerance": 0.07343389253317807, "passes": true}, "odd.women.b1936_1945.aime_p75": {"truth": 1865.0, "filled": 1863.5, "gap": -0.0008046131586025851, "seed_sd": 1.3178930553209385, "tolerance": 0.06283602704614394, "passes": true}, "odd.women.b1936_1945.aime_p90": {"truth": 2920.0, "filled": 2923.59, "gap": 0.001228696897506154, "seed_sd": 2.6663398922591086, "tolerance": 0.0536209031123137, "passes": true}, "odd.women.b1941_1945.aime_p10": {"truth": 115.0, "filled": 115.56000000000002, "gap": 0.004857747234784604, "seed_sd": 0.5716089940730976, "tolerance": 0.2647761737592696, "passes": true}, "odd.women.b1941_1945.aime_p25": {"truth": 399.0, "filled": 398.65, "gap": -0.0008775779413596752, "seed_sd": 0.8750939799154207, "tolerance": 0.15691889792263147, "passes": true}, "odd.women.b1941_1945.aime_p50": {"truth": 1077.0, "filled": 1074.3, "gap": -0.002510111483892352, "seed_sd": 1.218281792655455, "tolerance": 0.08797933476607916, "passes": true}, "odd.women.b1941_1945.aime_p75": {"truth": 2116.0, "filled": 2117.2, "gap": 0.0005669470056419712, "seed_sd": 2.647739850474263, "tolerance": 0.07205053130731136, "passes": true}, "odd.women.b1941_1945.aime_p90": {"truth": 3268.6000000000004, "filled": 3270.2, "gap": 0.0004893864415294047, "seed_sd": 4.079215610874249, "tolerance": 0.06548294022251817, "passes": true}, "odd.women.b1946_1955.paime_p10": {"truth": 158.0, "filled": 158.1, "gap": 0.0006327111884596448, "seed_sd": 0.6407232755171874, "tolerance": 0.1085059910489472, "passes": true}, "odd.women.b1946_1955.paime_p25": {"truth": 519.0, "filled": 519.5375, "gap": 0.001035109561267511, "seed_sd": 0.7533914548297761, "tolerance": 0.062889445073137, "passes": true}, "odd.women.b1946_1955.paime_p50": {"truth": 1313.0, "filled": 1313.1, "gap": 7.615856216336425e-05, "seed_sd": 1.3337718577107005, "tolerance": 0.04313158587991037, "passes": true}, "odd.women.b1946_1955.paime_p75": {"truth": 2486.0, "filled": 2487.075, "gap": 0.0004323280934812601, "seed_sd": 1.3280872672658541, "tolerance": 0.03158401312013406, "passes": true}, "odd.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3780.7699999999995, "gap": 0.00020368295892225774, "seed_sd": 3.2444446447169453, "tolerance": 0.02998098721497899, "passes": true}, "odd.women.b1946_1980.paime_p10": {"truth": 109.0, "filled": 107.9, "gap": -0.010143009965054794, "seed_sd": 0.30779350562554625, "tolerance": 0.0500438197853722, "passes": true}, "odd.women.b1946_1980.paime_p25": {"truth": 309.0, "filled": 309.2, "gap": 0.0006470398155213886, "seed_sd": 0.523148363780597, "tolerance": 0.030021121307637916, "passes": true}, "odd.women.b1946_1980.paime_p50": {"truth": 758.0, "filled": 757.575, "gap": -0.0005608432590138435, "seed_sd": 0.6742442162583931, "tolerance": 0.02113414008122686, "passes": true}, "odd.women.b1946_1980.paime_p75": {"truth": 1590.0, "filled": 1588.9, "gap": -0.0006920633199563042, "seed_sd": 0.7880689256524124, "tolerance": 0.018434974631993433, "passes": true}, "odd.women.b1946_1980.paime_p90": {"truth": 2696.0, "filled": 2699.1399999999994, "gap": 0.0011640107039063707, "seed_sd": 1.7863223025034836, "tolerance": 0.018162919937725473, "passes": true}, "odd.women.b1956_1965.paime_p10": {"truth": 140.0, "filled": 139.7, "gap": -0.0021451563463878998, "seed_sd": 0.47016234598162726, "tolerance": 0.0987946317785998, "passes": true}, "odd.women.b1956_1965.paime_p25": {"truth": 426.0, "filled": 425.3, "gap": -0.0016445440097827557, "seed_sd": 0.8645047258706176, "tolerance": 0.05702004966656188, "passes": true}, "odd.women.b1956_1965.paime_p50": {"truth": 1000.0, "filled": 999.975, "gap": -2.500031250463053e-05, "seed_sd": 1.2190915513308735, "tolerance": 0.0353321966023362, "passes": true}, "odd.women.b1956_1965.paime_p75": {"truth": 1842.0, "filled": 1845.2, "gap": 0.0017357348684132745, "seed_sd": 1.538112308540638, "tolerance": 0.026005856561860215, "passes": true}, "odd.women.b1956_1965.paime_p90": {"truth": 2793.0, "filled": 2796.055, "gap": 0.0010932081735646193, "seed_sd": 2.1636895876208633, "tolerance": 0.027293077620429543, "passes": true}, "odd.women.b1966_1980.paime_p10": {"truth": 80.0, "filled": 79.5, "gap": -0.006269613013595077, "seed_sd": 0.512989176042577, "tolerance": 0.06283109225080236, "passes": true}, "odd.women.b1966_1980.paime_p25": {"truth": 211.0, "filled": 210.6, "gap": -0.0018975337761917288, "seed_sd": 0.5982430416161189, "tolerance": 0.033157126538046186, "passes": true}, "odd.women.b1966_1980.paime_p50": {"truth": 463.0, "filled": 463.45, "gap": 0.0009714502356077404, "seed_sd": 0.8255779474818965, "tolerance": 0.02601990880890944, "passes": true}, "odd.women.b1966_1980.paime_p75": {"truth": 866.75, "filled": 867.0875, "gap": 0.0003893098450840071, "seed_sd": 0.840093196584384, "tolerance": 0.02735691355887302, "passes": true}, "odd.women.b1966_1980.paime_p90": {"truth": 1387.0, "filled": 1389.2, "gap": 0.0015849005550876427, "seed_sd": 1.5761378513048248, "tolerance": 0.027821321010049294, "passes": true}}, "tier": "improves"}, "alternative": {"passes": false, "n_gating": 183, "n_failing": 46, "cells": {"odd.men.a22_29.atcap": {"truth": 0.014676604027739744, "filled": 0.014777786673743196, "gap": 0.006870489703507232, "seed_sd": 0.0001771275607299105, "tolerance": 0.18023370223284765, "passes": true}, "odd.men.a22_29.level": {"truth": 0.23657185599421368, "filled": 0.23663321104590285, "gap": 0.00025931696977288254, "seed_sd": 0.0002391145887235585, "tolerance": 0.02171146931539837, "passes": true}, "odd.men.a22_29.q10": {"truth": 0.040229885057471264, "filled": 0.04246680662035942, "gap": 0.0541126212544607, "seed_sd": 0.00023223418208336734, "tolerance": 0.07329266358707544, "passes": true}, "odd.men.a22_29.q50": {"truth": 0.24222222222222223, "filled": 0.24479072242975236, "gap": 0.010548072901813477, "seed_sd": 0.00041499148868458826, "tolerance": 0.022980444394274636, "passes": true}, "odd.men.a22_29.q90": {"truth": 0.5661157024793388, "filled": 0.5691991060972214, "gap": 0.005431817106252845, "seed_sd": 0.000776009792142812, "tolerance": 0.02353059621893112, "passes": true}, "odd.men.a22_29.r1": {"truth": 0.8114342592359591, "filled": 0.7956558165756714, "gap": -0.015778442660287717, "seed_sd": 0.000538390871650334, "tolerance": 0.007273490161268873, "passes": false}, "odd.men.a22_29.r2": {"truth": 0.7090594261869381, "filled": 0.6903291482211632, "gap": -0.018730277965774866, "seed_sd": 0.001555470519385471, "tolerance": 0.014356564488812985, "passes": false}, "odd.men.a22_29.r3": {"truth": 0.6020834292042946, "filled": 0.5785067458786555, "gap": -0.023576683325639114, "seed_sd": 0.0007072870652359764, "tolerance": 0.013280307041936976, "passes": false}, "odd.men.a22_29.r4": {"truth": 0.5990582702554191, "filled": 0.5615431784424237, "gap": -0.037515091812995394, "seed_sd": 0.0016355341838879228, "tolerance": 0.02119963449400435, "passes": false}, "odd.men.a22_29.wint": {"truth": 0.08288543140028289, "filled": 0.06141796322489392, "gap": -0.29975695663693624, "seed_sd": 0.0022284434581811073}, "odd.men.a22_29.zexit": {"truth": 0.4099387400566883, "filled": 0.4131343147115296, "gap": 0.00776502326956241, "seed_sd": 0.002840187990001759, "tolerance": 0.052311207438163365, "passes": true}, "odd.men.a22_29.zint": {"truth": 0.027589420573962787, "filled": 0.03422566442780474, "gap": 0.21554339780599507, "seed_sd": 0.0004781100182703536, "tolerance": 0.12140160693630424, "passes": false}, "odd.men.a22_74.atcap": {"truth": 0.10371778137953785, "filled": 0.10473678694403343, "gap": 0.009776841924077129, "seed_sd": 0.00015417691554360955, "tolerance": 0.03690628859850606, "passes": true}, "odd.men.a22_74.level": {"truth": 0.35325148977531445, "filled": 0.3531990207432427, "gap": -0.00014854269729291936, "seed_sd": 0.0001000050086771046, "tolerance": 0.01106822547283481, "passes": true}, "odd.men.a22_74.q10": {"truth": 0.06111111111111111, "filled": 0.06439628414809703, "gap": 0.05236223099161608, "seed_sd": 0.0001374553093986069, "tolerance": 0.03462864626687699, "passes": false}, "odd.men.a22_74.q50": {"truth": 0.37052341597796146, "filled": 0.3743119269609451, "gap": 0.010172835353512988, "seed_sd": 0.0003016483959067802, "tolerance": 0.012463773427145929, "passes": true}, "odd.men.a22_74.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.003451823067723808, "passes": true}, "odd.men.a22_74.r1": {"truth": 0.8935990156167083, "filled": 0.8912007740159659, "gap": -0.0023982416007424234, "seed_sd": 0.00018848733136768257, "tolerance": 0.00258049589094515, "passes": true}, "odd.men.a22_74.r2": {"truth": 0.8291027164330249, "filled": 0.8070684084035232, "gap": -0.02203430802950168, "seed_sd": 0.000390606153237376, "tolerance": 0.004576681103770273, "passes": false}, "odd.men.a22_74.r3": {"truth": 0.7806492619651291, "filled": 0.7637546915867583, "gap": -0.01689457037837072, "seed_sd": 0.00023760130078809586, "tolerance": 0.004722018684826323, "passes": false}, "odd.men.a22_74.r4": {"truth": 0.7414979369803346, "filled": 0.7061540022562445, "gap": -0.03534393472409014, "seed_sd": 0.0007006280998860182, "tolerance": 0.007000982741779954, "passes": false}, "odd.men.a22_74.wint": {"truth": 0.05745341614906832, "filled": 0.027112318840579706, "gap": -0.7509862711587996, "seed_sd": 0.0005542995081891025, "tolerance": 0.07337302886151653, "passes": false}, "odd.men.a22_74.zexit": {"truth": 0.44752843148294114, "filled": 0.44908700221446535, "gap": 0.0034765680865107562, "seed_sd": 0.001594744130375705, "tolerance": 0.01916127382099423, "passes": true}, "odd.men.a22_74.zint": {"truth": 0.019892968610837895, "filled": 0.024561302963886585, "gap": 0.2108058209935062, "seed_sd": 0.00015894269446237608, "tolerance": 0.05481211657295162, "passes": false}, "odd.men.a30_44.atcap": {"truth": 0.10574805846620577, "filled": 0.10662077371152032, "gap": 0.008218909988384926, "seed_sd": 0.0002907042942532108, "tolerance": 0.049095580366705964, "passes": true}, "odd.men.a30_44.level": {"truth": 0.3982899917120792, "filled": 0.39874622791535386, "gap": 0.0011448319207326696, "seed_sd": 0.00016511436265352354, "tolerance": 0.015220021463125325, "passes": true}, "odd.men.a30_44.q10": {"truth": 0.08706467661691543, "filled": 0.08896815888583662, "gap": 0.021627288538648592, "seed_sd": 0.0003609981444106532, "tolerance": 0.047669064141889934, "passes": true}, "odd.men.a30_44.q50": {"truth": 0.41333333333333333, "filled": 0.41762659698724747, "gap": 0.010333354712889542, "seed_sd": 0.00047442460909717036, "tolerance": 0.016660415997544607, "passes": true}, "odd.men.a30_44.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.0037405831290436837, "passes": true}, "odd.men.a30_44.r1": {"truth": 0.9000861304840623, "filled": 0.8982106532629379, "gap": -0.0018754772211243553, "seed_sd": 0.00027892984330348577, "tolerance": 0.003556208950302802, "passes": true}, "odd.men.a30_44.r2": {"truth": 0.8479233224100629, "filled": 0.8246521380478662, "gap": -0.023271184362196662, "seed_sd": 0.0006272798022775334, "tolerance": 0.006500650570758907, "passes": false}, "odd.men.a30_44.r3": {"truth": 0.8030903893396322, "filled": 0.7843304370743531, "gap": -0.018759952265279045, "seed_sd": 0.0004185258212879459, "tolerance": 0.006087370619188444, "passes": false}, "odd.men.a30_44.r4": {"truth": 0.7869108820616492, "filled": 0.7484063801783494, "gap": -0.03850450188329979, "seed_sd": 0.0006994124324364131, "tolerance": 0.009218001294372115, "passes": false}, "odd.men.a30_44.wint": {"truth": 0.07239145006330527, "filled": 0.03310866165189543, "gap": -0.782293269893291, "seed_sd": 0.0008974621170717661, "tolerance": 0.12001713872675526, "passes": false}, "odd.men.a30_44.zexit": {"truth": 0.4342024282437196, "filled": 0.43137833603759856, "gap": -0.006525334996968168, "seed_sd": 0.002601768462594161, "tolerance": 0.031953074094816486, "passes": true}, "odd.men.a30_44.zint": {"truth": 0.017933882760031997, "filled": 0.02123248934943166, "gap": 0.1688407098487641, "seed_sd": 0.00022396503563602355, "tolerance": 0.08989768311292742, "passes": false}, "odd.men.a45_59.atcap": {"truth": 0.15826463477931516, "filled": 0.15946433822763031, "gap": 0.007551776839737512, "seed_sd": 0.000344680339561423, "tolerance": 0.04711609868499565, "passes": true}, "odd.men.a45_59.level": {"truth": 0.4279852808128974, "filled": 0.4275779517090131, "gap": -0.0009521894330581926, "seed_sd": 0.0001983681644962252, "tolerance": 0.015834008317210692, "passes": true}, "odd.men.a45_59.q10": {"truth": 0.08868501529051988, "filled": 0.09235107265412808, "gap": 0.04050638360257164, "seed_sd": 0.00043726182355950465, "tolerance": 0.055701610600039544, "passes": true}, "odd.men.a45_59.q50": {"truth": 0.4701492537313433, "filled": 0.47394056916236876, "gap": 0.008031726897409053, "seed_sd": 0.0004982713854334642, "tolerance": 0.020385938992396834, "passes": true}, "odd.men.a45_59.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0}, "odd.men.a45_59.r1": {"truth": 0.9077661818197654, "filled": 0.9062895753124117, "gap": -0.001476606507353706, "seed_sd": 0.0002991092950177627, "tolerance": 0.004032009924506552, "passes": true}, "odd.men.a45_59.r2": {"truth": 0.850421444941529, "filled": 0.8269450270263959, "gap": -0.023476417915133108, "seed_sd": 0.0006042911387057817, "tolerance": 0.0074262368998509335, "passes": false}, "odd.men.a45_59.r3": {"truth": 0.8146289743583418, "filled": 0.7978400340916917, "gap": -0.01678894026665012, "seed_sd": 0.0004995869149426763, "tolerance": 0.007198396861260139, "passes": false}, "odd.men.a45_59.r4": {"truth": 0.7598303885400162, "filled": 0.7277735022660579, "gap": -0.03205688627395831, "seed_sd": 0.0013551668525231687, "tolerance": 0.012107705260459355, "passes": false}, "odd.men.a45_59.wint": {"truth": 0.05825168693530555, "filled": 0.02120773131028173, "gap": -1.010407252645218, "seed_sd": 0.0007690594119414791, "tolerance": 0.13344110142481172, "passes": false}, "odd.men.a45_59.zexit": {"truth": 0.45185706741770815, "filled": 0.4509065305403978, "gap": -0.002105838628690182, "seed_sd": 0.003266441784644715, "tolerance": 0.036314919811016276, "passes": true}, "odd.men.a45_59.zint": {"truth": 0.015835938476928848, "filled": 0.01935900962861073, "gap": 0.20087598076997093, "seed_sd": 0.0002885353100226522, "tolerance": 0.09287476169914739, "passes": false}, "odd.men.a60_74.atcap": {"truth": 0.10024796315975912, "filled": 0.10304852526017169, "gap": 0.02755324794681613, "seed_sd": 0.0004932495167821012, "tolerance": 0.12472307183935011, "passes": true}, "odd.men.a60_74.level": {"truth": 0.20453937198442498, "filled": 0.2035442319076033, "gap": -0.004877147917293545, "seed_sd": 0.0003456308860662253, "tolerance": 0.0522939828282486, "passes": true}, "odd.men.a60_74.q10": {"truth": 0.01529051987767584, "filled": 0.01800191108137369, "gap": 0.16324490292884608, "seed_sd": 0.00021772331220310686, "tolerance": 0.20186846491463123, "passes": true}, "odd.men.a60_74.q50": {"truth": 0.21666666666666667, "filled": 0.22468282133340836, "gap": 0.03632965048242465, "seed_sd": 0.001131712401025331, "tolerance": 0.07446457229432804, "passes": true}, "odd.men.a60_74.q90": {"truth": 1.0, "filled": 1.0, "gap": 0.0, "seed_sd": 0.0, "tolerance": 0.04069424979546094, "passes": true}, "odd.men.a60_74.r1": {"truth": 0.8712782553777247, "filled": 0.8740493797312835, "gap": 0.0027711243535587515, "seed_sd": 0.0008060871217724458, "tolerance": 0.009750532797445479, "passes": true}, "odd.men.a60_74.r2": {"truth": 0.7888015745113253, "filled": 0.7688138264043076, "gap": -0.019987748107017644, "seed_sd": 0.002381034523972206, "tolerance": 0.020757725353477027, "passes": true}, "odd.men.a60_74.r3": {"truth": 0.7283746799072385, "filled": 0.7151581255769487, "gap": -0.013216554330289787, "seed_sd": 0.0011687895450421084, "tolerance": 0.02064690204069302, "passes": true}, "odd.men.a60_74.r4": {"truth": 0.6781549627202678, "filled": 0.6469288632171912, "gap": -0.031226099503076532, "seed_sd": 0.003228826412160021, "tolerance": 0.03820786124425089, "passes": true}, "odd.men.a60_74.wint": {"truth": 0.03084255319148936, "filled": 0.010389787234042554, "gap": -1.088072006632677, "seed_sd": 0.000692980423590686}, "odd.men.a60_74.zexit": {"truth": 0.5047452182800409, "filled": 0.5156543534335911, "gap": 0.021382899633966224, "seed_sd": 0.0028521576213372778, "tolerance": 0.04434932710092839, "passes": true}, "odd.men.a60_74.zint": {"truth": 0.027132152458672565, "filled": 0.038515072358762184, "gap": 0.3503301923053983, "seed_sd": 0.0007221867926525434, "tolerance": 0.17758159695027936, "passes": false}, "odd.men.b1936_1940.aime_p10": {"truth": 346.0, "filled": 346.45, "gap": 0.0012997330156654385, "seed_sd": 1.1459310165698642, "tolerance": 0.30817736367768905, "passes": true}, "odd.men.b1936_1940.aime_p25": {"truth": 1131.0, "filled": 1130.3, "gap": -0.0006191129194350609, "seed_sd": 0.9233805168766387, "tolerance": 0.15436566295506535, "passes": true}, "odd.men.b1936_1940.aime_p50": {"truth": 2430.0, "filled": 2427.9, "gap": -0.0008645711648282983, "seed_sd": 1.2937094768634554, "tolerance": 0.0822816707411384, "passes": true}, "odd.men.b1936_1940.aime_p75": {"truth": 3582.0, "filled": 3580.8, "gap": -0.0003350645030515409, "seed_sd": 1.3218806379747876, "tolerance": 0.04915519601813279, "passes": true}, "odd.men.b1936_1940.aime_p90": {"truth": 4382.0, "filled": 4376.85, "gap": -0.0011759535997271087, "seed_sd": 2.084403234046921, "tolerance": 0.03801502836217192, "passes": true}, "odd.men.b1936_1945.aime_p10": {"truth": 342.8000000000002, "filled": 341.72, "gap": -0.0031554984402069053, "seed_sd": 0.77974354758472, "tolerance": 0.19387427864829543, "passes": true}, "odd.men.b1936_1945.aime_p25": {"truth": 1152.0, "filled": 1154.125, "gap": 0.0018429188369548655, "seed_sd": 0.8251794063302713, "tolerance": 0.09928513681983075, "passes": true}, "odd.men.b1936_1945.aime_p50": {"truth": 2625.0, "filled": 2623.7, "gap": -0.0004953607661262183, "seed_sd": 1.3803127029389888, "tolerance": 0.05013200755321697, "passes": true}, "odd.men.b1936_1945.aime_p75": {"truth": 3991.5, "filled": 3995.55, "gap": 0.00101414172870129, "seed_sd": 1.6535448364999934, "tolerance": 0.030282584009309735, "passes": true}, "odd.men.b1936_1945.aime_p90": {"truth": 5075.0, "filled": 5069.499999999999, "gap": -0.0010843315173527657, "seed_sd": 3.0803280754866273, "tolerance": 0.029246187918705275, "passes": true}, "odd.men.b1941_1945.aime_p10": {"truth": 338.0, "filled": 337.69, "gap": -0.0009175806116727969, "seed_sd": 1.1247923785022893, "tolerance": 0.2434370939708618, "passes": true}, "odd.men.b1941_1945.aime_p25": {"truth": 1174.25, "filled": 1177.95, "gap": 0.0031459935818887175, "seed_sd": 2.067289096988207, "tolerance": 0.15717605214643726, "passes": true}, "odd.men.b1941_1945.aime_p50": {"truth": 2821.5, "filled": 2824.2, "gap": 0.0009564802259571792, "seed_sd": 2.816305904437304, "tolerance": 0.06794319334096698, "passes": true}, "odd.men.b1941_1945.aime_p75": {"truth": 4421.0, "filled": 4423.125, "gap": 0.00048054500380700915, "seed_sd": 2.5707105304913167, "tolerance": 0.039611439643930886, "passes": true}, "odd.men.b1941_1945.aime_p90": {"truth": 5528.300000000001, "filled": 5530.155000000001, "gap": 0.0003354899065737271, "seed_sd": 3.074080949521515, "tolerance": 0.024092287379584104, "passes": true}, "odd.men.b1946_1955.paime_p10": {"truth": 308.0, "filled": 309.55, "gap": 0.00501984699166691, "seed_sd": 0.998683343734455, "tolerance": 0.11641286886749184, "passes": true}, "odd.men.b1946_1955.paime_p25": {"truth": 1026.0, "filled": 1031.0, "gap": 0.0048614582862454014, "seed_sd": 1.2565617248750864, "tolerance": 0.07923354406873329, "passes": true}, "odd.men.b1946_1955.paime_p50": {"truth": 2614.0, "filled": 2612.45, "gap": -0.0005931368502301027, "seed_sd": 1.8202082009311031, "tolerance": 0.03884145918889322, "passes": true}, "odd.men.b1946_1955.paime_p75": {"truth": 4390.0, "filled": 4390.2, "gap": 4.55570488231416e-05, "seed_sd": 2.912766816330444, "tolerance": 0.02464882648778213, "passes": true}, "odd.men.b1946_1955.paime_p90": {"truth": 5810.0, "filled": 5804.210000000001, "gap": -0.0009970545529416341, "seed_sd": 2.14178970907387, "tolerance": 0.01935089604023114, "passes": true}, "odd.men.b1946_1980.paime_p10": {"truth": 175.0, "filled": 176.85, "gap": 0.01051594172797543, "seed_sd": 0.36634754853252327, "tolerance": 0.05432313062129484, "passes": true}, "odd.men.b1946_1980.paime_p25": {"truth": 487.0, "filled": 487.45, "gap": 0.0009235979926911497, "seed_sd": 0.6048053188292994, "tolerance": 0.03064646902486962, "passes": true}, "odd.men.b1946_1980.paime_p50": {"truth": 1245.0, "filled": 1246.15, "gap": 0.0009232684356144105, "seed_sd": 0.7451598203705947, "tolerance": 0.025757396517852603, "passes": true}, "odd.men.b1946_1980.paime_p75": {"truth": 2644.0, "filled": 2645.6125, "gap": 0.0006096855109705146, "seed_sd": 1.2888055465593347, "tolerance": 0.0206417529534006, "passes": true}, "odd.men.b1946_1980.paime_p90": {"truth": 4310.0, "filled": 4309.555000000002, "gap": -0.00010325359032847814, "seed_sd": 2.004068230795734, "tolerance": 0.01797004074133569, "passes": true}, "odd.men.b1956_1965.paime_p10": {"truth": 253.0, "filled": 255.37000000000006, "gap": 0.009323985168177451, "seed_sd": 0.5777451632719952, "tolerance": 0.10325340800222761, "passes": true}, "odd.men.b1956_1965.paime_p25": {"truth": 765.0, "filled": 768.75, "gap": 0.00488998529419149, "seed_sd": 1.2085223687584246, "tolerance": 0.06714959239176221, "passes": true}, "odd.men.b1956_1965.paime_p50": {"truth": 1764.0, "filled": 1764.85, "gap": 0.0004817433534656246, "seed_sd": 1.496487114615601, "tolerance": 0.03783036836459788, "passes": true}, "odd.men.b1956_1965.paime_p75": {"truth": 2958.0, "filled": 2957.75, "gap": -8.452013697279881e-05, "seed_sd": 1.4823523268955432, "tolerance": 0.02582244416868233, "passes": true}, "odd.men.b1956_1965.paime_p90": {"truth": 4111.0, "filled": 4108.740000000001, "gap": -0.0005498957526519632, "seed_sd": 1.215860102756237, "tolerance": 0.022664845011086624, "passes": true}, "odd.men.b1966_1980.paime_p10": {"truth": 112.0, "filled": 114.15, "gap": 0.019014501680710616, "seed_sd": 0.36634754853252327, "tolerance": 0.07763263566160054, "passes": true}, "odd.men.b1966_1980.paime_p25": {"truth": 303.25, "filled": 304.2125, "gap": 0.0031689225440505453, "seed_sd": 0.6138478724621562, "tolerance": 0.04145124387911718, "passes": true}, "odd.men.b1966_1980.paime_p50": {"truth": 655.0, "filled": 655.25, "gap": 0.0003816065682640257, "seed_sd": 0.8506963092234007, "tolerance": 0.02859495381248372, "passes": true}, "odd.men.b1966_1980.paime_p75": {"truth": 1225.0, "filled": 1225.025, "gap": 2.0407955021894963e-05, "seed_sd": 1.0159905718584514, "tolerance": 0.024541637055534683, "passes": true}, "odd.men.b1966_1980.paime_p90": {"truth": 1888.9000000000015, "filled": 1888.94, "gap": 2.117612180541073e-05, "seed_sd": 1.428064571737837, "tolerance": 0.025554311796701996, "passes": true}, "odd.women.a22_29.atcap": {"truth": 0.005728386357627567, "filled": 0.00547841056691332, "gap": -0.044618861731397175, "seed_sd": 0.0001198351137429579}, "odd.women.a22_29.level": {"truth": 0.1854665923300802, "filled": 0.1848740910687206, "gap": -0.0031997660178675336, "seed_sd": 0.00014689392351929534, "tolerance": 0.018517427407013797, "passes": true}, "odd.women.a22_29.q10": {"truth": 0.028735632183908046, "filled": 0.029796942509710787, "gap": 0.03626789570059907, "seed_sd": 0.00021225931773054583, "tolerance": 0.0684466769976442, "passes": true}, "odd.women.a22_29.q50": {"truth": 0.191131498470948, "filled": 0.19297325015068054, "gap": 0.0095899142162017, "seed_sd": 0.0003559220591245615, "tolerance": 0.021849929026002995, "passes": true}, "odd.women.a22_29.q90": {"truth": 0.46005509641873277, "filled": 0.4622275517880918, "gap": 0.004711049029350156, "seed_sd": 0.0005400193806653005, "tolerance": 0.020751504651450387, "passes": true}, "odd.women.a22_29.r1": {"truth": 0.7942016511246733, "filled": 0.7745630021286282, "gap": -0.01963864899604517, "seed_sd": 0.0007292428154612855, "tolerance": 0.006445547562698701, "passes": false}, "odd.women.a22_29.r2": {"truth": 0.6731585499668502, "filled": 0.6642985372572703, "gap": -0.008860012709579923, "seed_sd": 0.002126151717425611, "tolerance": 0.013111971249341917, "passes": true}, "odd.women.a22_29.r3": {"truth": 0.5535614164842262, "filled": 0.5329081040600008, "gap": -0.020653312424225412, "seed_sd": 0.0009899564896844986, "tolerance": 0.012054115899361516, "passes": false}, "odd.women.a22_29.r4": {"truth": 0.552312691565311, "filled": 0.5278673657012054, "gap": -0.024445325864105638, "seed_sd": 0.002194118197599944, "tolerance": 0.01982762523559322, "passes": false}, "odd.women.a22_29.wint": {"truth": 0.08501885498800137, "filled": 0.06794309221803221, "gap": -0.22420257962872592, "seed_sd": 0.0017381669686975256, "tolerance": 0.15037337545812676, "passes": false}, "odd.women.a22_29.zexit": {"truth": 0.41757828810020875, "filled": 0.42427974947807934, "gap": 0.015920981053845762, "seed_sd": 0.0035450250782922098, "tolerance": 0.041287246240518535, "passes": true}, "odd.women.a22_29.zint": {"truth": 0.0315547703180212, "filled": 0.04034628975265017, "gap": 0.24577466265365944, "seed_sd": 0.0006563413066339625, "tolerance": 0.09342121518593281, "passes": false}, "odd.women.a22_74.atcap": {"truth": 0.029398307023564402, "filled": 0.029927931331664083, "gap": 0.017855114137902195, "seed_sd": 0.00013923057619168755, "tolerance": 0.06709882226274566, "passes": true}, "odd.women.a22_74.level": {"truth": 0.2372854094489719, "filled": 0.2373323106690793, "gap": 0.00019763788106530455, "seed_sd": 6.692992230174344e-05, "tolerance": 0.010902725750872375, "passes": true}, "odd.women.a22_74.q10": {"truth": 0.03777777777777778, "filled": 0.03953723702579737, "gap": 0.045521897075397, "seed_sd": 0.00010489045382469636, "tolerance": 0.031129845660342752, "passes": false}, "odd.women.a22_74.q50": {"truth": 0.24875621890547264, "filled": 0.25114321485161784, "gap": 0.009549977161435352, "seed_sd": 0.00023603789043680378, "tolerance": 0.012092425838785583, "passes": true}, "odd.women.a22_74.q90": {"truth": 0.6444444444444445, "filled": 0.6478378057479859, "gap": 0.005251746052140682, "seed_sd": 0.00023512489065456287, "tolerance": 0.013338685039058783, "passes": true}, "odd.women.a22_74.r1": {"truth": 0.8812684347612898, "filled": 0.8790561126410884, "gap": -0.002212322120201393, "seed_sd": 0.00026366261525432043, "tolerance": 0.002475922564645967, "passes": true}, "odd.women.a22_74.r2": {"truth": 0.8056395091372208, "filled": 0.7906943070261213, "gap": -0.01494520211109951, "seed_sd": 0.0008216520277802765, "tolerance": 0.004772836686607259, "passes": false}, "odd.women.a22_74.r3": {"truth": 0.7468316241328095, "filled": 0.7322134148251858, "gap": -0.014618209307623697, "seed_sd": 0.00042896857078218335, "tolerance": 0.005171487109756781, "passes": false}, "odd.women.a22_74.r4": {"truth": 0.7116744894807431, "filled": 0.6823372442061806, "gap": -0.029337245274562496, "seed_sd": 0.0007616203326649829, "tolerance": 0.007666262733486471, "passes": false}, "odd.women.a22_74.wint": {"truth": 0.051544874036178946, "filled": 0.027846837562797884, "gap": -0.6157333613583016, "seed_sd": 0.0002980240871252773, "tolerance": 0.07133033012317215, "passes": false}, "odd.women.a22_74.zexit": {"truth": 0.45845752128015943, "filled": 0.4576462554649855, "gap": -0.0017711225471119807, "seed_sd": 0.0008638884333265164, "tolerance": 0.016039678883419197, "passes": true}, "odd.women.a22_74.zint": {"truth": 0.022201124403524123, "filled": 0.027159357807590257, "gap": 0.2015787212211242, "seed_sd": 0.0002523189210600705, "tolerance": 0.04476234327525859, "passes": false}, "odd.women.a30_44.atcap": {"truth": 0.034619662054697874, "filled": 0.035343450511844496, "gap": 0.020691311561050973, "seed_sd": 0.0002127099479691062, "tolerance": 0.08510646139536093, "passes": true}, "odd.women.a30_44.level": {"truth": 0.2579924798566703, "filled": 0.25834447971623264, "gap": 0.0013634503886146287, "seed_sd": 0.00012777256821587513, "tolerance": 0.01578879525774269, "passes": true}, "odd.women.a30_44.q10": {"truth": 0.04228855721393035, "filled": 0.0434532780200243, "gap": 0.027169757945184614, "seed_sd": 0.0002946128390438379, "tolerance": 0.04573918824793483, "passes": true}, "odd.women.a30_44.q50": {"truth": 0.26859504132231404, "filled": 0.27048982232809066, "gap": 0.0070296494531243425, "seed_sd": 0.0003572641637749294, "tolerance": 0.018488627942446087, "passes": true}, "odd.women.a30_44.q90": {"truth": 0.6716417910447762, "filled": 0.6776760023832323, "gap": 0.008944151770306275, "seed_sd": 0.0008011122590598357, "tolerance": 0.018806950089013175, "passes": true}, "odd.women.a30_44.r1": {"truth": 0.888241419643356, "filled": 0.8878263947476558, "gap": -0.0004150248957002223, "seed_sd": 0.00039085301367594664, "tolerance": 0.0034876380830913202, "passes": true}, "odd.women.a30_44.r2": {"truth": 0.8242277470219269, "filled": 0.8080048131468264, "gap": -0.016222933875100543, "seed_sd": 0.0013833436447721692, "tolerance": 0.007078708407960211, "passes": false}, "odd.women.a30_44.r3": {"truth": 0.7685608654079126, "filled": 0.7530075711034663, "gap": -0.015553294304446297, "seed_sd": 0.0006708836561040368, "tolerance": 0.0073196290204859075, "passes": false}, "odd.women.a30_44.r4": {"truth": 0.7490435327097834, "filled": 0.7166554517324724, "gap": -0.03238808097731105, "seed_sd": 0.0012795706934268887, "tolerance": 0.010187042430903364, "passes": false}, "odd.women.a30_44.wint": {"truth": 0.06073485056210584, "filled": 0.033955305730737594, "gap": -0.5814725551358548, "seed_sd": 0.0006491314836425788, "tolerance": 0.10093703169083498, "passes": false}, "odd.women.a30_44.zexit": {"truth": 0.45066799061202384, "filled": 0.44936021845098406, "gap": -0.0029060713169714036, "seed_sd": 0.0018066337869472745, "tolerance": 0.024719366356404402, "passes": true}, "odd.women.a30_44.zint": {"truth": 0.022222222222222223, "filled": 0.02548624733010767, "gap": 0.13704619707907684, "seed_sd": 0.0002748908459707359, "tolerance": 0.07029190354044747, "passes": false}, "odd.women.a45_59.atcap": {"truth": 0.04053483605515179, "filled": 0.041239707200188894, "gap": 0.01723980531931124, "seed_sd": 0.00018965501967751098, "tolerance": 0.09107631963722376, "passes": true}, "odd.women.a45_59.level": {"truth": 0.28104586423928846, "filled": 0.28119635395726095, "gap": 0.0005353198913060631, "seed_sd": 0.00014247082170896653, "tolerance": 0.01605667435242628, "passes": true}, "odd.women.a45_59.q10": {"truth": 0.053516819571865444, "filled": 0.05545805092900992, "gap": 0.03563090683069614, "seed_sd": 0.0004748767234848939, "tolerance": 0.051646333169991114, "passes": true}, "odd.women.a45_59.q50": {"truth": 0.29338842975206614, "filled": 0.2964274540543556, "gap": 0.010305084280428867, "seed_sd": 0.0003713143604795996, "tolerance": 0.0182941133196311, "passes": true}, "odd.women.a45_59.q90": {"truth": 0.7300275482093664, "filled": 0.7337269335985186, "gap": 0.005054663622374556, "seed_sd": 0.0005886018592810698, "tolerance": 0.0208569023153089, "passes": true}, "odd.women.a45_59.r1": {"truth": 0.9101006329634369, "filled": 0.9108974471764941, "gap": 0.0007968142130572176, "seed_sd": 0.0003800420009505249, "tolerance": 0.0037361334549905496, "passes": true}, "odd.women.a45_59.r2": {"truth": 0.8498173267452248, "filled": 0.8361682096210815, "gap": -0.013649117124143295, "seed_sd": 0.0008553583208083107, "tolerance": 0.0074184353060767995, "passes": false}, "odd.women.a45_59.r3": {"truth": 0.8096130207660378, "filled": 0.7977960800478768, "gap": -0.011816940718160973, "seed_sd": 0.0005535582972400586, "tolerance": 0.007740061426920219, "passes": false}, "odd.women.a45_59.r4": {"truth": 0.7620167777938601, "filled": 0.7367946054768152, "gap": -0.025222172317044933, "seed_sd": 0.0014331459374298417, "tolerance": 0.012512985825002505, "passes": false}, "odd.women.a45_59.wint": {"truth": 0.050115932427956277, "filled": 0.018719774759854254, "gap": -0.9847585311516158, "seed_sd": 0.00039460658224803785, "tolerance": 0.1271041647107106, "passes": false}, "odd.women.a45_59.zexit": {"truth": 0.4693136110029843, "filled": 0.46855780459322693, "gap": -0.0016117488193031493, "seed_sd": 0.0026195994047660633, "tolerance": 0.030024476078798885, "passes": true}, "odd.women.a45_59.zint": {"truth": 0.01632776600720909, "filled": 0.01969952948298159, "gap": 0.18772765670992042, "seed_sd": 0.0003049152557496679, "tolerance": 0.09017841563442361, "passes": false}, "odd.women.a60_74.atcap": {"truth": 0.01812919896640827, "filled": 0.018630583537738197, "gap": 0.02728066557727482, "seed_sd": 0.00039508547268232713}, "odd.women.a60_74.level": {"truth": 0.12926943691615103, "filled": 0.12904856374827697, "gap": -0.0017100877301055029, "seed_sd": 0.0001935279226056811, "tolerance": 0.04408494999775865, "passes": true}, "odd.women.a60_74.q10": {"truth": 0.017412935323383085, "filled": 0.018677153233438732, "gap": 0.07008768527188503, "seed_sd": 0.00034170082521718935, "tolerance": 0.13013445963696416, "passes": true}, "odd.women.a60_74.q50": {"truth": 0.15517241379310345, "filled": 0.1576036237180233, "gap": 0.015546324522169197, "seed_sd": 0.0005376073055390707, "tolerance": 0.05365060769995596, "passes": true}, "odd.women.a60_74.q90": {"truth": 0.5360696517412935, "filled": 0.5359236469864845, "gap": -0.0002723986351127472, "seed_sd": 0.0013064682810661346, "tolerance": 0.04960336282899433, "passes": true}, "odd.women.a60_74.r1": {"truth": 0.8610627385333514, "filled": 0.8641934680473181, "gap": 0.003130729513966757, "seed_sd": 0.0009278283970724514, "tolerance": 0.009751654703991025, "passes": true}, "odd.women.a60_74.r2": {"truth": 0.7722738686836694, "filled": 0.7442124191252916, "gap": -0.02806144955837786, "seed_sd": 0.0023964065139079997, "tolerance": 0.022174026993056747, "passes": false}, "odd.women.a60_74.r3": {"truth": 0.7141078828415282, "filled": 0.6932584179481365, "gap": -0.020849464893391678, "seed_sd": 0.001709358820841769, "tolerance": 0.018479646240433655, "passes": false}, "odd.women.a60_74.r4": {"truth": 0.6501204938024134, "filled": 0.6102725160155678, "gap": -0.03984797778684568, "seed_sd": 0.004143724959812022, "tolerance": 0.036314307016843086, "passes": false}, "odd.women.a60_74.wint": {"truth": 0.023107482596493787, "filled": 0.008455734956445676, "gap": -1.0053115827709154, "seed_sd": 0.000566243984687916}, "odd.women.a60_74.zexit": {"truth": 0.5155483759303063, "filled": 0.5055270293659493, "gap": -0.019629634203051416, "seed_sd": 0.003325690297144084, "tolerance": 0.03896610501214408, "passes": true}, "odd.women.a60_74.zint": {"truth": 0.022290698541288734, "filled": 0.033288188663303596, "gap": 0.40103315359017433, "seed_sd": 0.0006465836607581495}, "odd.women.b1936_1940.aime_p10": {"truth": 82.29999999999995, "filled": 83.05, "gap": 0.009071728376279786, "seed_sd": 0.22360679774997896, "tolerance": 0.32371200850504067, "passes": true}, "odd.women.b1936_1940.aime_p25": {"truth": 287.75, "filled": 288.15, "gap": 0.0013891302806836592, "seed_sd": 0.36634754853252327, "tolerance": 0.20139115911902655, "passes": true}, "odd.women.b1936_1940.aime_p50": {"truth": 786.0, "filled": 785.9, "gap": -0.0001272345570777489, "seed_sd": 0.7181848464596079, "tolerance": 0.12524071173214943, "passes": true}, "odd.women.b1936_1940.aime_p75": {"truth": 1566.0, "filled": 1566.225, "gap": 0.00014366784020136691, "seed_sd": 1.2822164198638313, "tolerance": 0.10110924143579125, "passes": true}, "odd.women.b1936_1940.aime_p90": {"truth": 2438.7000000000007, "filled": 2438.2400000000007, "gap": -0.00018864287908559874, "seed_sd": 2.977353116267416, "tolerance": 0.08967006892401257, "passes": true}, "odd.women.b1936_1945.aime_p10": {"truth": 98.0, "filled": 98.54, "gap": 0.005495078445244772, "seed_sd": 0.5030433694811536, "tolerance": 0.1995625804481066, "passes": true}, "odd.women.b1936_1945.aime_p25": {"truth": 345.0, "filled": 344.55, "gap": -0.0013051992281427616, "seed_sd": 0.6048053188292994, "tolerance": 0.11108088073804002, "passes": true}, "odd.women.b1936_1945.aime_p50": {"truth": 932.0, "filled": 930.85, "gap": -0.001234667467684858, "seed_sd": 0.9880869341680844, "tolerance": 0.07343389253317807, "passes": true}, "odd.women.b1936_1945.aime_p75": {"truth": 1865.0, "filled": 1865.0, "gap": 0.0, "seed_sd": 1.3764944032233706, "tolerance": 0.06283602704614394, "passes": true}, "odd.women.b1936_1945.aime_p90": {"truth": 2920.0, "filled": 2925.88, "gap": 0.002011673856783247, "seed_sd": 1.7355721034622027, "tolerance": 0.0536209031123137, "passes": true}, "odd.women.b1941_1945.aime_p10": {"truth": 115.0, "filled": 115.96000000000001, "gap": 0.008313175690204844, "seed_sd": 0.6003507746572602, "tolerance": 0.2647761737592696, "passes": true}, "odd.women.b1941_1945.aime_p25": {"truth": 399.0, "filled": 399.05, "gap": 0.00012530543215394374, "seed_sd": 0.9445132413883327, "tolerance": 0.15691889792263147, "passes": true}, "odd.women.b1941_1945.aime_p50": {"truth": 1077.0, "filled": 1074.3, "gap": -0.002510111483892352, "seed_sd": 1.0809352675491621, "tolerance": 0.08797933476607916, "passes": true}, "odd.women.b1941_1945.aime_p75": {"truth": 2116.0, "filled": 2119.05, "gap": 0.0014403610475932638, "seed_sd": 2.0124611797498106, "tolerance": 0.07205053130731136, "passes": true}, "odd.women.b1941_1945.aime_p90": {"truth": 3268.6000000000004, "filled": 3272.19, "gap": 0.0010977268374308125, "seed_sd": 3.0741537132407077, "tolerance": 0.06548294022251817, "passes": true}, "odd.women.b1946_1955.paime_p10": {"truth": 158.0, "filled": 158.7, "gap": 0.004420594505396558, "seed_sd": 0.6569466853317862, "tolerance": 0.1085059910489472, "passes": true}, "odd.women.b1946_1955.paime_p25": {"truth": 519.0, "filled": 518.7375, "gap": -0.0005059082968452699, "seed_sd": 0.9370832406995656, "tolerance": 0.062889445073137, "passes": true}, "odd.women.b1946_1955.paime_p50": {"truth": 1313.0, "filled": 1313.25, "gap": 0.00019038553127526114, "seed_sd": 1.019545822516343, "tolerance": 0.04313158587991037, "passes": true}, "odd.women.b1946_1955.paime_p75": {"truth": 2486.0, "filled": 2486.425, "gap": 0.0001709427496789928, "seed_sd": 1.709301180692212, "tolerance": 0.03158401312013406, "passes": true}, "odd.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3782.2200000000003, "gap": 0.0005871291932191269, "seed_sd": 2.638699839334961, "tolerance": 0.02998098721497899, "passes": true}, "odd.women.b1946_1980.paime_p10": {"truth": 109.0, "filled": 108.15500000000002, "gap": -0.007782498813678096, "seed_sd": 0.36487200351269256, "tolerance": 0.0500438197853722, "passes": true}, "odd.women.b1946_1980.paime_p25": {"truth": 309.0, "filled": 308.55, "gap": -0.001457372130669654, "seed_sd": 0.5104177855340405, "tolerance": 0.030021121307637916, "passes": true}, "odd.women.b1946_1980.paime_p50": {"truth": 758.0, "filled": 757.325, "gap": -0.0008908980511055375, "seed_sd": 0.46665100224336475, "tolerance": 0.02113414008122686, "passes": true}, "odd.women.b1946_1980.paime_p75": {"truth": 1590.0, "filled": 1589.4, "gap": -0.000377429708198207, "seed_sd": 0.7539370349250519, "tolerance": 0.018434974631993433, "passes": true}, "odd.women.b1946_1980.paime_p90": {"truth": 2696.0, "filled": 2698.4, "gap": 0.0008898117152424945, "seed_sd": 1.729009451741235, "tolerance": 0.018162919937725473, "passes": true}, "odd.women.b1956_1965.paime_p10": {"truth": 140.0, "filled": 139.79500000000002, "gap": -0.0014653588283035646, "seed_sd": 0.6056531924077901, "tolerance": 0.0987946317785998, "passes": true}, "odd.women.b1956_1965.paime_p25": {"truth": 426.0, "filled": 425.075, "gap": -0.002173722325821359, "seed_sd": 1.0390405493328425, "tolerance": 0.05702004966656188, "passes": true}, "odd.women.b1956_1965.paime_p50": {"truth": 1000.0, "filled": 999.575, "gap": -0.0004250903380960125, "seed_sd": 1.0166379058341997, "tolerance": 0.0353321966023362, "passes": true}, "odd.women.b1956_1965.paime_p75": {"truth": 1842.0, "filled": 1844.7625, "gap": 0.0014986050861729439, "seed_sd": 1.298974798183471, "tolerance": 0.026005856561860215, "passes": true}, "odd.women.b1956_1965.paime_p90": {"truth": 2793.0, "filled": 2794.3599999999997, "gap": 0.00048681310202169925, "seed_sd": 2.08286240291426, "tolerance": 0.027293077620429543, "passes": true}, "odd.women.b1966_1980.paime_p10": {"truth": 80.0, "filled": 80.05, "gap": 0.0006248047688428571, "seed_sd": 0.22360679774997896, "tolerance": 0.06283109225080236, "passes": true}, "odd.women.b1966_1980.paime_p25": {"truth": 211.0, "filled": 210.35, "gap": -0.0030853234395369356, "seed_sd": 0.5871429486123998, "tolerance": 0.033157126538046186, "passes": true}, "odd.women.b1966_1980.paime_p50": {"truth": 463.0, "filled": 462.8, "gap": -0.0004320587667123732, "seed_sd": 0.7677718959499145, "tolerance": 0.02601990880890944, "passes": true}, "odd.women.b1966_1980.paime_p75": {"truth": 866.75, "filled": 867.1, "gap": 0.0004037258179820924, "seed_sd": 0.7181848464596079, "tolerance": 0.02735691355887302, "passes": true}, "odd.women.b1966_1980.paime_p90": {"truth": 1387.0, "filled": 1389.1649999999997, "gap": 0.0015597058812399922, "seed_sd": 1.3819418069857066, "tolerance": 0.027821321010049294, "passes": true}}, "tier": "not_adopted"}, "adopted": "primary"}, "pre": {"dropped_undefined_truth": [], "truth": {"pre.men.b1930_1934.aime_p10": {"value": 319.29999999999995, "events": 12005}, "pre.men.b1930_1934.aime_p25": {"value": 961.0, "events": 12005}, "pre.men.b1930_1934.aime_p50": {"value": 1871.0, "events": 12005}, "pre.men.b1930_1934.aime_p75": {"value": 2604.0, "events": 12005}, "pre.men.b1930_1934.aime_p90": {"value": 3048.7000000000007, "events": 12005}, "pre.men.b1930_1934.pzero": {"value": 0.23849477516714046, "events": 48408}, "pre.men.b1930_1934.plevel": {"value": 0.5292246059138269, "events": 154565}, "pre.men.b1930_1934.pr_in": {"value": 0.6408121492387024, "events": 9749}, "pre.men.b1930_1934.pr_cross": {"value": 0.6005766012618865, "events": 9749}, "pre.men.b1935_1939.aime_p10": {"value": 342.0, "events": 12618}, "pre.men.b1935_1939.aime_p25": {"value": 1099.5, "events": 12618}, "pre.men.b1935_1939.aime_p50": {"value": 2295.0, "events": 12618}, "pre.men.b1935_1939.aime_p75": {"value": 3360.0, "events": 12618}, "pre.men.b1935_1939.aime_p90": {"value": 4087.0, "events": 12618}, "pre.men.b1935_1939.pzero": {"value": 0.21315539421109053, "events": 34995}, "pre.men.b1935_1939.plevel": {"value": 0.47885857925887765, "events": 129181}, "pre.men.b1935_1939.pr_in": {"value": 0.5082293220171369, "events": 9949}, "pre.men.b1935_1939.pr_cross": {"value": 0.5480805317128244, "events": 10010}, "pre.men.b1940_1945.aime_p10": {"value": 343.0, "events": 18739}, "pre.men.b1940_1945.aime_p25": {"value": 1178.0, "events": 18739}, "pre.men.b1940_1945.aime_p50": {"value": 2795.0, "events": 18739}, "pre.men.b1940_1945.aime_p75": {"value": 4361.5, "events": 18739}, "pre.men.b1940_1945.aime_p90": {"value": 5432.800000000003, "events": 18739}, "pre.men.b1940_1945.pzero": {"value": 0.21918652571223796, "events": 30582}, "pre.men.b1940_1945.plevel": {"value": 0.3760217560197671, "events": 108943}, "pre.men.b1940_1945.pr_in": {"value": 0.36011521796896784, "events": 12582}, "pre.men.b1940_1945.pr_cross": {"value": 0.376062044153939, "events": 14181}, "pre.men.b1930_1945.aime_p10": {"value": 335.0, "events": 43362}, "pre.men.b1930_1945.aime_p25": {"value": 1080.0, "events": 43362}, "pre.men.b1930_1945.aime_p50": {"value": 2266.0, "events": 43362}, "pre.men.b1930_1945.aime_p75": {"value": 3412.75, "events": 43362}, "pre.men.b1930_1945.aime_p90": {"value": 4641.0, "events": 43362}, "pre.men.b1930_1945.pzero": {"value": 0.22496713863351978, "events": 113985}, "pre.men.b1930_1945.plevel": {"value": 0.47071653085260085, "events": 392689}, "pre.men.b1930_1945.pr_in": {"value": 0.5486408320203577, "events": 32280}, "pre.men.b1930_1945.pr_cross": {"value": 0.5031824436185776, "events": 33940}, "pre.men.b1946_1955.yzero": {"value": 0.41447106660067734, "events": 128130}, "pre.men.b1946_1955.ylevel": {"value": 0.1479729717947154, "events": 181011}, "pre.men.b1946_1955.yr_cross": {"value": 0.4294012571477206, "events": 32399}, "pre.men.b1946_1955.paime_p10": {"value": 305.0, "events": 44105}, "pre.men.b1946_1955.paime_p25": {"value": 1022.0, "events": 44105}, "pre.men.b1946_1955.paime_p50": {"value": 2610.0, "events": 44105}, "pre.men.b1946_1955.paime_p75": {"value": 4389.0, "events": 44105}, "pre.men.b1946_1955.paime_p90": {"value": 5809.0, "events": 44105}, "pre.men.b1956_1965.yzero": {"value": 0.4166095146652036, "events": 145790}, "pre.men.b1956_1965.ylevel": {"value": 0.09639046670103465, "events": 204154}, "pre.men.b1956_1965.yr_cross": {"value": 0.40595517510382484, "events": 35165}, "pre.men.b1956_1965.paime_p10": {"value": 249.0, "events": 49926}, "pre.men.b1956_1965.paime_p25": {"value": 760.0, "events": 49926}, "pre.men.b1956_1965.paime_p50": {"value": 1761.0, "events": 49926}, "pre.men.b1956_1965.paime_p75": {"value": 2956.0, "events": 49926}, "pre.men.b1956_1965.paime_p90": {"value": 4109.0, "events": 49926}, "pre.men.b1966_1980.yzero": {"value": 0.41574858870643483, "events": 180950}, "pre.men.b1966_1980.ylevel": {"value": 0.055118361796504756, "events": 254289}, "pre.men.b1966_1980.yr_cross": {"value": 0.4093896749871457, "events": 45353}, "pre.men.b1966_1980.paime_p10": {"value": 104.0, "events": 62071}, "pre.men.b1966_1980.paime_p25": {"value": 297.0, "events": 62071}, "pre.men.b1966_1980.paime_p50": {"value": 648.0, "events": 62071}, "pre.men.b1966_1980.paime_p75": {"value": 1218.0, "events": 62071}, "pre.men.b1966_1980.paime_p90": {"value": 1883.0, "events": 62071}, "pre.men.b1946_1980.yzero": {"value": 0.415663002913214, "events": 454870}, "pre.men.b1946_1980.ylevel": {"value": 0.09454735400371912, "events": 639454}, "pre.men.b1946_1980.yr_cross": {"value": 0.49823745301361005, "events": 112917}, "pre.men.b1946_1980.paime_p10": {"value": 168.0, "events": 156102}, "pre.men.b1946_1980.paime_p25": {"value": 480.0, "events": 156102}, "pre.men.b1946_1980.paime_p50": {"value": 1237.0, "events": 156102}, "pre.men.b1946_1980.paime_p75": {"value": 2636.0, "events": 156102}, "pre.men.b1946_1980.paime_p90": {"value": 4302.0, "events": 156102}, "pre.women.b1930_1934.aime_p10": {"value": 51.40000000000009, "events": 10532}, "pre.women.b1930_1934.aime_p25": {"value": 192.0, "events": 10532}, "pre.women.b1930_1934.aime_p50": {"value": 526.0, "events": 10532}, "pre.women.b1930_1934.aime_p75": {"value": 1055.0, "events": 10532}, "pre.women.b1930_1934.aime_p90": {"value": 1663.0, "events": 10532}, "pre.women.b1930_1934.pzero": {"value": 0.5514059876498446, "events": 80419}, "pre.women.b1930_1934.plevel": {"value": 0.17987838861322633, "events": 80419}, "pre.women.b1930_1934.pr_in": {"value": 0.5486860492872828, "events": 3125}, "pre.women.b1930_1934.pr_cross": {"value": 0.5714979296198577, "events": 3542}, "pre.women.b1935_1939.aime_p10": {"value": 75.0, "events": 11626}, "pre.women.b1935_1939.aime_p25": {"value": 267.0, "events": 11626}, "pre.women.b1935_1939.aime_p50": {"value": 726.0, "events": 11626}, "pre.women.b1935_1939.aime_p75": {"value": 1448.0, "events": 11626}, "pre.women.b1935_1939.aime_p90": {"value": 2293.8999999999996, "events": 11626}, "pre.women.b1935_1939.pzero": {"value": 0.5152445437812253, "events": 73741}, "pre.women.b1935_1939.plevel": {"value": 0.18459645468504016, "events": 73741}, "pre.women.b1935_1939.pr_in": {"value": 0.48588537175161917, "events": 3458}, "pre.women.b1935_1939.pr_cross": {"value": 0.5177687037806673, "events": 3675}, "pre.women.b1940_1945.aime_p10": {"value": 110.0, "events": 17687}, "pre.women.b1940_1945.aime_p25": {"value": 386.0, "events": 17687}, "pre.women.b1940_1945.aime_p50": {"value": 1041.0, "events": 17687}, "pre.women.b1940_1945.aime_p75": {"value": 2058.0, "events": 17687}, "pre.women.b1940_1945.aime_p90": {"value": 3179.7000000000007, "events": 17687}, "pre.women.b1940_1945.pzero": {"value": 0.45477910425062734, "events": 59808}, "pre.women.b1940_1945.plevel": {"value": 0.19012308454067162, "events": 71702}, "pre.women.b1940_1945.pr_in": {"value": 0.19634726483554435, "events": 5880}, "pre.women.b1940_1945.pr_cross": {"value": 0.3145537439002571, "events": 6220}, "pre.women.b1930_1945.aime_p10": {"value": 79.0, "events": 39845}, "pre.women.b1930_1945.aime_p25": {"value": 280.0, "events": 39845}, "pre.women.b1930_1945.aime_p50": {"value": 771.0, "events": 39845}, "pre.women.b1930_1945.aime_p75": {"value": 1574.0, "events": 39845}, "pre.women.b1930_1945.aime_p90": {"value": 2560.0, "events": 39845}, "pre.women.b1930_1945.pzero": {"value": 0.5120706676834471, "events": 225862}, "pre.women.b1930_1945.plevel": {"value": 0.18433938803699407, "events": 225862}, "pre.women.b1930_1945.pr_in": {"value": 0.372715069734525, "events": 12463}, "pre.women.b1930_1945.pr_cross": {"value": 0.4395941263612976, "events": 13437}, "pre.women.b1946_1955.yzero": {"value": 0.5347643649882455, "events": 136154}, "pre.women.b1946_1955.ylevel": {"value": 0.09059452101103667, "events": 136154}, "pre.women.b1946_1955.yr_cross": {"value": 0.34158903474392094, "events": 22200}, "pre.women.b1946_1955.paime_p10": {"value": 156.0, "events": 41718}, "pre.women.b1946_1955.paime_p25": {"value": 516.0, "events": 41718}, "pre.women.b1946_1955.paime_p50": {"value": 1311.0, "events": 41718}, "pre.women.b1946_1955.paime_p75": {"value": 2484.0, "events": 41718}, "pre.women.b1946_1955.paime_p90": {"value": 3780.0, "events": 41718}, "pre.women.b1956_1965.yzero": {"value": 0.47511142887567126, "events": 156162}, "pre.women.b1956_1965.ylevel": {"value": 0.06347343380872424, "events": 172523}, "pre.women.b1956_1965.yr_cross": {"value": 0.3538451043271927, "events": 28194}, "pre.women.b1956_1965.paime_p10": {"value": 136.0, "events": 46845}, "pre.women.b1956_1965.paime_p25": {"value": 422.0, "events": 46845}, "pre.women.b1956_1965.paime_p50": {"value": 996.0, "events": 46845}, "pre.women.b1956_1965.paime_p75": {"value": 1839.0, "events": 46845}, "pre.women.b1956_1965.paime_p90": {"value": 2790.5999999999985, "events": 46845}, "pre.women.b1966_1980.yzero": {"value": 0.42453709903219916, "events": 174368}, "pre.women.b1966_1980.ylevel": {"value": 0.04398552907354234, "events": 236357}, "pre.women.b1966_1980.yr_cross": {"value": 0.318767197977468, "events": 40457}, "pre.women.b1966_1980.paime_p10": {"value": 73.0, "events": 58514}, "pre.women.b1966_1980.paime_p25": {"value": 205.0, "events": 58514}, "pre.women.b1966_1980.paime_p50": {"value": 458.0, "events": 58514}, "pre.women.b1966_1980.paime_p75": {"value": 861.0, "events": 58514}, "pre.women.b1966_1980.paime_p90": {"value": 1382.5999999999985, "events": 58514}, "pre.women.b1946_1980.yzero": {"value": 0.4719000529035934, "events": 487032}, "pre.women.b1946_1980.ylevel": {"value": 0.06340849534928693, "events": 545034}, "pre.women.b1946_1980.yr_cross": {"value": 0.43636860215681333, "events": 90851}, "pre.women.b1946_1980.paime_p10": {"value": 103.0, "events": 147077}, "pre.women.b1946_1980.paime_p25": {"value": 304.0, "events": 147077}, "pre.women.b1946_1980.paime_p50": {"value": 752.0, "events": 147077}, "pre.women.b1946_1980.paime_p75": {"value": 1583.75, "events": 147077}, "pre.women.b1946_1980.paime_p90": {"value": 2690.0, "events": 147077}}, "current_rule": {"fallback": {"passes": false, "n_gating": 136, "n_failing": 131, "cells": {"pre.men.b1930_1934.aime_p10": {"truth": 319.29999999999995, "filled": 103.29999999999995, "gap": -1.1284937235951435, "seed_sd": 0.0, "tolerance": 0.3785445178728429, "passes": false}, "pre.men.b1930_1934.aime_p25": {"truth": 961.0, "filled": 509.75, "gap": -0.6340539995157277, "seed_sd": 0.0, "tolerance": 0.1748310883030068, "passes": false}, "pre.men.b1930_1934.aime_p50": {"truth": 1871.0, "filled": 1274.0, "gap": -0.3843114901419806, "seed_sd": 0.0, "tolerance": 0.08482500842259981, "passes": false}, "pre.men.b1930_1934.aime_p75": {"truth": 2604.0, "filled": 1990.0, "gap": -0.26891408560992147, "seed_sd": 0.0, "tolerance": 0.0526345782788754, "passes": false}, "pre.men.b1930_1934.aime_p90": {"truth": 3048.7000000000007, "filled": 2479.0, "gap": -0.20686001719645475, "seed_sd": 0.0, "tolerance": 0.032507815796954, "passes": false}, "pre.men.b1930_1934.plevel": {"truth": 0.5292246059138269, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.05399925993701722, "passes": false}, "pre.men.b1930_1934.pr_cross": {"truth": 0.6005766012618865, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.10805017504629746, "passes": false}, "pre.men.b1930_1934.pr_in": {"truth": 0.6408121492387024, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.08754712042010183, "passes": false}, "pre.men.b1930_1934.pzero": {"truth": 0.23849477516714046, "filled": 1.0, "gap": 1.433407875949719, "seed_sd": 0.0, "tolerance": 0.12848404727056906, "passes": false}, "pre.men.b1930_1945.aime_p10": {"truth": 335.0, "filled": 155.0, "gap": -0.7707054149058195, "seed_sd": 0.0, "tolerance": 0.17154652244393, "passes": false}, "pre.men.b1930_1945.aime_p25": {"truth": 1080.0, "filled": 716.0, "gap": -0.4110361531576201, "seed_sd": 0.0, "tolerance": 0.08803344717643322, "passes": false}, "pre.men.b1930_1945.aime_p50": {"truth": 2266.0, "filled": 1813.0, "gap": -0.22303323083310111, "seed_sd": 0.0, "tolerance": 0.042293553277777104, "passes": false}, "pre.men.b1930_1945.aime_p75": {"truth": 3412.75, "filled": 3075.0, "gap": -0.10421351664246892, "seed_sd": 0.0, "tolerance": 0.0339810042777025, "passes": false}, "pre.men.b1930_1945.aime_p90": {"truth": 4641.0, "filled": 4481.0, "gap": -0.03508362445503366, "seed_sd": 0.0, "tolerance": 0.03097131693794459, "passes": false}, "pre.men.b1930_1945.plevel": {"truth": 0.47071653085260085, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.029173903026433003, "passes": false}, "pre.men.b1930_1945.pr_cross": {"truth": 0.5031824436185776, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.04396339753603818, "passes": false}, "pre.men.b1930_1945.pr_in": {"truth": 0.5486408320203577, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.04095810136062007, "passes": false}, "pre.men.b1930_1945.pzero": {"truth": 0.22496713863351978, "filled": 1.0, "gap": 1.4918009379618222, "seed_sd": 0.0, "tolerance": 0.06285985273036603, "passes": false}, "pre.men.b1935_1939.aime_p10": {"truth": 342.0, "filled": 150.0, "gap": -0.8241754429663493, "seed_sd": 0.0, "tolerance": 0.35458282617348863, "passes": false}, "pre.men.b1935_1939.aime_p25": {"truth": 1099.5, "filled": 706.0, "gap": -0.44299557250157395, "seed_sd": 0.0, "tolerance": 0.17611744524582296, "passes": false}, "pre.men.b1935_1939.aime_p50": {"truth": 2295.0, "filled": 1845.0, "gap": -0.21825356602001822, "seed_sd": 0.0, "tolerance": 0.08805365646562159, "passes": false}, "pre.men.b1935_1939.aime_p75": {"truth": 3360.0, "filled": 2934.0, "gap": -0.13557429425432233, "seed_sd": 0.0, "tolerance": 0.047075663502220186, "passes": false}, "pre.men.b1935_1939.aime_p90": {"truth": 4087.0, "filled": 3747.0, "gap": -0.08685568477059036, "seed_sd": 0.0, "tolerance": 0.03677617093870473, "passes": false}, "pre.men.b1935_1939.plevel": {"truth": 0.47885857925887765, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.0505548909117149, "passes": false}, "pre.men.b1935_1939.pr_cross": {"truth": 0.5480805317128244, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.08965478073758665, "passes": false}, "pre.men.b1935_1939.pr_in": {"truth": 0.5082293220171369, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.08171176478999366, "passes": false}, "pre.men.b1935_1939.pzero": {"truth": 0.21315539421109053, "filled": 1.0, "gap": 1.5457338289783504, "seed_sd": 0.0, "tolerance": 0.12406583731865686, "passes": false}, "pre.men.b1940_1945.aime_p10": {"truth": 343.0, "filled": 211.0, "gap": -0.4858723136898728, "seed_sd": 0.0, "tolerance": 0.2436998893100363, "passes": false}, "pre.men.b1940_1945.aime_p25": {"truth": 1178.0, "filled": 965.0, "gap": -0.19944526287254583, "seed_sd": 0.0, "tolerance": 0.14802915955392618, "passes": false}, "pre.men.b1940_1945.aime_p50": {"truth": 2795.0, "filled": 2571.0, "gap": -0.08353717832331053, "seed_sd": 0.0, "tolerance": 0.07497681768693197, "passes": false}, "pre.men.b1940_1945.aime_p75": {"truth": 4361.5, "filled": 4219.0, "gap": -0.033217901748935574, "seed_sd": 0.0, "tolerance": 0.04190766386026753, "passes": true}, "pre.men.b1940_1945.aime_p90": {"truth": 5432.800000000003, "filled": 5366.0, "gap": -0.01237190281401368, "seed_sd": 0.0, "tolerance": 0.023132300299791037, "passes": true}, "pre.men.b1940_1945.plevel": {"truth": 0.3760217560197671, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.04021166929479513, "passes": false}, "pre.men.b1940_1945.pr_cross": {"truth": 0.376062044153939, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.05919050745324543, "passes": false}, "pre.men.b1940_1945.pr_in": {"truth": 0.36011521796896784, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.06500235072966114, "passes": false}, "pre.men.b1940_1945.pzero": {"truth": 0.21918652571223796, "filled": 1.0, "gap": 1.5178321960885375, "seed_sd": 0.0, "tolerance": 0.09503409717716321, "passes": false}, "pre.men.b1946_1955.paime_p10": {"truth": 305.0, "filled": 230.0, "gap": -0.2822324676842163, "seed_sd": 0.0, "tolerance": 0.11504556883383053, "passes": false}, "pre.men.b1946_1955.paime_p25": {"truth": 1022.0, "filled": 925.5, "gap": -0.09918263875009803, "seed_sd": 0.0, "tolerance": 0.08011617890879942, "passes": false}, "pre.men.b1946_1955.paime_p50": {"truth": 2610.0, "filled": 2482.0, "gap": -0.050285534552186206, "seed_sd": 0.0, "tolerance": 0.038612530439093365, "passes": false}, "pre.men.b1946_1955.paime_p75": {"truth": 4389.0, "filled": 4277.0, "gap": -0.025849581461324433, "seed_sd": 0.0, "tolerance": 0.022937547887772285, "passes": false}, "pre.men.b1946_1955.paime_p90": {"truth": 5809.0, "filled": 5730.0, "gap": -0.013692908283747585, "seed_sd": 0.0, "tolerance": 0.018317056116449435, "passes": true}, "pre.men.b1946_1955.ylevel": {"truth": 0.1479729717947154, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.02562210127460557, "passes": false}, "pre.men.b1946_1955.yr_cross": {"truth": 0.4294012571477206, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.03213928274663533, "passes": false}, "pre.men.b1946_1955.yzero": {"truth": 0.41447106660067734, "filled": 1.0, "gap": 0.8807521099778143, "seed_sd": 0.0, "tolerance": 0.021706485686521018, "passes": false}, "pre.men.b1946_1980.paime_p10": {"truth": 168.0, "filled": 121.0, "gap": -0.3281734338065174, "seed_sd": 0.0, "tolerance": 0.055691835388134804, "passes": false}, "pre.men.b1946_1980.paime_p25": {"truth": 480.0, "filled": 400.0, "gap": -0.1823215567939549, "seed_sd": 0.0, "tolerance": 0.03034406597699225, "passes": false}, "pre.men.b1946_1980.paime_p50": {"truth": 1237.0, "filled": 1127.0, "gap": -0.09312985835271093, "seed_sd": 0.0, "tolerance": 0.02474263079924987, "passes": false}, "pre.men.b1946_1980.paime_p75": {"truth": 2636.0, "filled": 2493.0, "gap": -0.055775812098840305, "seed_sd": 0.0, "tolerance": 0.02024689542128028, "passes": false}, "pre.men.b1946_1980.paime_p90": {"truth": 4302.0, "filled": 4169.899999999994, "gap": -0.031187976137720952, "seed_sd": 0.0, "tolerance": 0.01712820101125616, "passes": false}, "pre.men.b1946_1980.ylevel": {"truth": 0.09454735400371912, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.01520534060461241, "passes": false}, "pre.men.b1946_1980.yr_cross": {"truth": 0.49823745301361005, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.014632893176957621, "passes": false}, "pre.men.b1946_1980.yzero": {"truth": 0.415663002913214, "filled": 1.0, "gap": 0.877880436171331, "seed_sd": 0.0, "tolerance": 0.011384493587046594, "passes": false}, "pre.men.b1956_1965.paime_p10": {"truth": 249.0, "filled": 194.0, "gap": -0.24959473740137916, "seed_sd": 0.0, "tolerance": 0.11340392915219548, "passes": false}, "pre.men.b1956_1965.paime_p25": {"truth": 760.0, "filled": 679.0, "gap": -0.11269730572168069, "seed_sd": 0.0, "tolerance": 0.06443128496619711, "passes": false}, "pre.men.b1956_1965.paime_p50": {"truth": 1761.0, "filled": 1630.0, "gap": -0.07730181469539765, "seed_sd": 0.0, "tolerance": 0.03738718984350382, "passes": false}, "pre.men.b1956_1965.paime_p75": {"truth": 2956.0, "filled": 2793.0, "gap": -0.05672071612291507, "seed_sd": 0.0, "tolerance": 0.02599848784066106, "passes": false}, "pre.men.b1956_1965.paime_p90": {"truth": 4109.0, "filled": 3938.0, "gap": -0.04250670968433923, "seed_sd": 0.0, "tolerance": 0.022251423931640178, "passes": false}, "pre.men.b1956_1965.ylevel": {"truth": 0.09639046670103465, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.027201666962537053, "passes": false}, "pre.men.b1956_1965.yr_cross": {"truth": 0.40595517510382484, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.033463015007033206, "passes": false}, "pre.men.b1956_1965.yzero": {"truth": 0.4166095146652036, "filled": 1.0, "gap": 0.8756059115653633, "seed_sd": 0.0, "tolerance": 0.022693736955041323, "passes": false}, "pre.men.b1966_1980.paime_p10": {"truth": 104.0, "filled": 70.0, "gap": -0.39589565709201313, "seed_sd": 0.0, "tolerance": 0.07746758178850374, "passes": false}, "pre.men.b1966_1980.paime_p25": {"truth": 297.0, "filled": 233.0, "gap": -0.24269368523699963, "seed_sd": 0.0, "tolerance": 0.03917957142016487, "passes": false}, "pre.men.b1966_1980.paime_p50": {"truth": 648.0, "filled": 553.0, "gap": -0.15853269482993948, "seed_sd": 0.0, "tolerance": 0.029131134938009468, "passes": false}, "pre.men.b1966_1980.paime_p75": {"truth": 1218.0, "filled": 1107.0, "gap": -0.09555651556120548, "seed_sd": 0.0, "tolerance": 0.028246877693415603, "passes": false}, "pre.men.b1966_1980.paime_p90": {"truth": 1883.0, "filled": 1758.4000000000015, "gap": -0.06846194500779479, "seed_sd": 0.0, "tolerance": 0.02461284698314315, "passes": false}, "pre.men.b1966_1980.ylevel": {"truth": 0.055118361796504756, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.021382204828984532, "passes": false}, "pre.men.b1966_1980.yr_cross": {"truth": 0.4093896749871457, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.02246113667973179, "passes": false}, "pre.men.b1966_1980.yzero": {"truth": 0.41574858870643483, "filled": 1.0, "gap": 0.8776745554874777, "seed_sd": 0.0, "tolerance": 0.01675184764087755, "passes": false}, "pre.women.b1930_1934.aime_p10": {"truth": 51.40000000000009, "filled": 11.0, "gap": -1.5417428996627507, "seed_sd": 0.0, "tolerance": 0.3914251373153828, "passes": false}, "pre.women.b1930_1934.aime_p25": {"truth": 192.0, "filled": 83.0, "gap": -0.8386547642311832, "seed_sd": 0.0, "tolerance": 0.17419751750648213, "passes": false}, "pre.women.b1930_1934.aime_p50": {"truth": 526.0, "filled": 348.0, "gap": -0.4130987329632356, "seed_sd": 0.0, "tolerance": 0.10499905152764143, "passes": false}, "pre.women.b1930_1934.aime_p75": {"truth": 1055.0, "filled": 819.0, "gap": -0.2532119620570974, "seed_sd": 0.0, "tolerance": 0.0942504614966514, "passes": false}, "pre.women.b1930_1934.aime_p90": {"truth": 1663.0, "filled": 1328.6000000000004, "gap": -0.22449744396178684, "seed_sd": 0.0, "tolerance": 0.09146081269677721, "passes": false}, "pre.women.b1930_1934.plevel": {"truth": 0.17987838861322633, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.0866710956881841, "passes": false}, "pre.women.b1930_1934.pr_cross": {"truth": 0.5714979296198577, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.10849315671129696, "passes": false}, "pre.women.b1930_1934.pr_in": {"truth": 0.5486860492872828, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.1270201928993476, "passes": false}, "pre.women.b1930_1934.pzero": {"truth": 0.5514059876498446, "filled": 1.0, "gap": 0.5952839214563962, "seed_sd": 0.0, "tolerance": 0.04703504786975244, "passes": false}, "pre.women.b1930_1945.aime_p10": {"truth": 79.0, "filled": 27.0, "gap": -1.0736109864626924, "seed_sd": 0.0, "tolerance": 0.16621909484652816, "passes": false}, "pre.women.b1930_1945.aime_p25": {"truth": 280.0, "filled": 164.0, "gap": -0.5349231753450505, "seed_sd": 0.0, "tolerance": 0.09520232121757725, "passes": false}, "pre.women.b1930_1945.aime_p50": {"truth": 771.0, "filled": 601.0, "gap": -0.24909343902812164, "seed_sd": 0.0, "tolerance": 0.06218129686271377, "passes": false}, "pre.women.b1930_1945.aime_p75": {"truth": 1574.0, "filled": 1367.0, "gap": -0.14100159225339937, "seed_sd": 0.0, "tolerance": 0.0540210895745143, "passes": false}, "pre.women.b1930_1945.aime_p90": {"truth": 2560.0, "filled": 2353.0, "gap": -0.08431614874624582, "seed_sd": 0.0, "tolerance": 0.04855891958559449, "passes": false}, "pre.women.b1930_1945.plevel": {"truth": 0.18433938803699407, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.050000040503342155, "passes": false}, "pre.women.b1930_1945.pr_cross": {"truth": 0.4395941263612976, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.0677006808491353, "passes": false}, "pre.women.b1930_1945.pr_in": {"truth": 0.372715069734525, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.07412154624977427, "passes": false}, "pre.women.b1930_1945.pzero": {"truth": 0.5120706676834471, "filled": 1.0, "gap": 0.6692926406476696, "seed_sd": 0.0, "tolerance": 0.02913422225552515, "passes": false}, "pre.women.b1935_1939.aime_p10": {"truth": 75.0, "filled": 22.0, "gap": -1.226445660177994, "seed_sd": 0.0, "tolerance": 0.3512121935176347, "passes": false}, "pre.women.b1935_1939.aime_p25": {"truth": 267.0, "filled": 146.0, "gap": -0.6036420366919133, "seed_sd": 0.0, "tolerance": 0.180762984617868, "passes": false}, "pre.women.b1935_1939.aime_p50": {"truth": 726.0, "filled": 548.0, "gap": -0.28127472787678, "seed_sd": 0.0, "tolerance": 0.11655697880988479, "passes": false}, "pre.women.b1935_1939.aime_p75": {"truth": 1448.0, "filled": 1217.75, "gap": -0.1731784002590091, "seed_sd": 0.0, "tolerance": 0.08830721415636139, "passes": false}, "pre.women.b1935_1939.aime_p90": {"truth": 2293.8999999999996, "filled": 2012.8999999999996, "gap": -0.1306769574530957, "seed_sd": 0.0, "tolerance": 0.08570712841284106, "passes": false}, "pre.women.b1935_1939.plevel": {"truth": 0.18459645468504016, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.09407530912273358, "passes": false}, "pre.women.b1935_1939.pr_cross": {"truth": 0.5177687037806673, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.1216792156910919, "passes": false}, "pre.women.b1935_1939.pr_in": {"truth": 0.48588537175161917, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.13268326765003252, "passes": false}, "pre.women.b1935_1939.pzero": {"truth": 0.5152445437812253, "filled": 1.0, "gap": 0.6631136487266858, "seed_sd": 0.0, "tolerance": 0.0510656392812699, "passes": false}, "pre.women.b1940_1945.aime_p10": {"truth": 110.0, "filled": 57.0, "gap": -0.6574290979578663, "seed_sd": 0.0, "tolerance": 0.26939110963753704, "passes": false}, "pre.women.b1940_1945.aime_p25": {"truth": 386.0, "filled": 283.0, "gap": -0.3103904718215933, "seed_sd": 0.0, "tolerance": 0.14001979176307694, "passes": false}, "pre.women.b1940_1945.aime_p50": {"truth": 1041.0, "filled": 903.0, "gap": -0.1422145151979839, "seed_sd": 0.0, "tolerance": 0.08949151188066314, "passes": false}, "pre.women.b1940_1945.aime_p75": {"truth": 2058.0, "filled": 1914.0, "gap": -0.0725393443810951, "seed_sd": 0.0, "tolerance": 0.06627004271644896, "passes": false}, "pre.women.b1940_1945.aime_p90": {"truth": 3179.7000000000007, "filled": 3045.0, "gap": -0.04328595155732273, "seed_sd": 0.0, "tolerance": 0.06421030653828377, "passes": true}, "pre.women.b1940_1945.plevel": {"truth": 0.19012308454067162, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.06996499056922599, "passes": false}, "pre.women.b1940_1945.pr_cross": {"truth": 0.3145537439002571, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.10514700123361043, "passes": false}, "pre.women.b1940_1945.pr_in": {"truth": 0.19634726483554435, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.10970200704526734, "passes": false}, "pre.women.b1940_1945.pzero": {"truth": 0.45477910425062734, "filled": 1.0, "gap": 0.7879434630807212, "seed_sd": 0.0, "tolerance": 0.04701699812231318, "passes": false}, "pre.women.b1946_1955.paime_p10": {"truth": 156.0, "filled": 117.0, "gap": -0.2876820724517808, "seed_sd": 0.0, "tolerance": 0.1092839592303832, "passes": false}, "pre.women.b1946_1955.paime_p25": {"truth": 516.0, "filled": 457.0, "gap": -0.12142337458735764, "seed_sd": 0.0, "tolerance": 0.06230795442282464, "passes": false}, "pre.women.b1946_1955.paime_p50": {"truth": 1311.0, "filled": 1228.0, "gap": -0.06540337505661231, "seed_sd": 0.0, "tolerance": 0.0404504662154056, "passes": false}, "pre.women.b1946_1955.paime_p75": {"truth": 2484.0, "filled": 2391.0, "gap": -0.03815847559504526, "seed_sd": 0.0, "tolerance": 0.0286716465849327, "passes": false}, "pre.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3699.300000000003, "gap": -0.02158039706903736, "seed_sd": 0.0, "tolerance": 0.03102321311868942, "passes": true}, "pre.women.b1946_1955.ylevel": {"truth": 0.09059452101103667, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.029569090149685853, "passes": false}, "pre.women.b1946_1955.yr_cross": {"truth": 0.34158903474392094, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.03669253843323441, "passes": false}, "pre.women.b1946_1955.yzero": {"truth": 0.5347643649882455, "filled": 1.0, "gap": 0.6259290683823043, "seed_sd": 0.0, "tolerance": 0.015053483015288157, "passes": false}, "pre.women.b1946_1980.paime_p10": {"truth": 103.0, "filled": 71.0, "gap": -0.3720491111883204, "seed_sd": 0.0, "tolerance": 0.05217708504283977, "passes": false}, "pre.women.b1946_1980.paime_p25": {"truth": 304.0, "filled": 248.0, "gap": -0.2035989552412394, "seed_sd": 0.0, "tolerance": 0.031237246580028546, "passes": false}, "pre.women.b1946_1980.paime_p50": {"truth": 752.0, "filled": 672.0, "gap": -0.11247798342669046, "seed_sd": 0.0, "tolerance": 0.021471785135136603, "passes": false}, "pre.women.b1946_1980.paime_p75": {"truth": 1583.75, "filled": 1484.0, "gap": -0.06505430790802347, "seed_sd": 0.0, "tolerance": 0.021247962622687265, "passes": false}, "pre.women.b1946_1980.paime_p90": {"truth": 2690.0, "filled": 2588.0, "gap": -0.038655816975094126, "seed_sd": 0.0, "tolerance": 0.019286406429246394, "passes": false}, "pre.women.b1946_1980.ylevel": {"truth": 0.06340849534928693, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.015450091379218515, "passes": false}, "pre.women.b1946_1980.yr_cross": {"truth": 0.43636860215681333, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.015684351176846988, "passes": false}, "pre.women.b1946_1980.yzero": {"truth": 0.4719000529035934, "filled": 1.0, "gap": 0.7509880681421656, "seed_sd": 0.0, "tolerance": 0.008556191699814752, "passes": false}, "pre.women.b1956_1965.paime_p10": {"truth": 136.0, "filled": 106.0, "gap": -0.24921579162398544, "seed_sd": 0.0, "tolerance": 0.10309020846799896, "passes": false}, "pre.women.b1956_1965.paime_p25": {"truth": 422.0, "filled": 367.0, "gap": -0.1396434659814414, "seed_sd": 0.0, "tolerance": 0.05400647590482275, "passes": false}, "pre.women.b1956_1965.paime_p50": {"truth": 996.0, "filled": 912.0, "gap": -0.0881072675102672, "seed_sd": 0.0, "tolerance": 0.03402127173688933, "passes": false}, "pre.women.b1956_1965.paime_p75": {"truth": 1839.0, "filled": 1727.0, "gap": -0.06283614645764235, "seed_sd": 0.0, "tolerance": 0.026722598015179132, "passes": false}, "pre.women.b1956_1965.paime_p90": {"truth": 2790.5999999999985, "filled": 2662.0, "gap": -0.047178906503049234, "seed_sd": 0.0, "tolerance": 0.026506809111477153, "passes": false}, "pre.women.b1956_1965.ylevel": {"truth": 0.06347343380872424, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.026809792541125404, "passes": false}, "pre.women.b1956_1965.yr_cross": {"truth": 0.3538451043271927, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.03047360671843986, "passes": false}, "pre.women.b1956_1965.yzero": {"truth": 0.47511142887567126, "filled": 1.0, "gap": 0.7442059153520724, "seed_sd": 0.0, "tolerance": 0.017102901274216892, "passes": false}, "pre.women.b1966_1980.paime_p10": {"truth": 73.0, "filled": 47.0, "gap": -0.4403118394383325, "seed_sd": 0.0, "tolerance": 0.0627268236453548, "passes": false}, "pre.women.b1966_1980.paime_p25": {"truth": 205.0, "filled": 156.0, "gap": -0.27315397188887136, "seed_sd": 0.0, "tolerance": 0.039676687855380324, "passes": false}, "pre.women.b1966_1980.paime_p50": {"truth": 458.0, "filled": 384.0, "gap": -0.17622663152645845, "seed_sd": 0.0, "tolerance": 0.02838346665574009, "passes": false}, "pre.women.b1966_1980.paime_p75": {"truth": 861.0, "filled": 772.0, "gap": -0.10910995440295412, "seed_sd": 0.0, "tolerance": 0.023922104075604994, "passes": false}, "pre.women.b1966_1980.paime_p90": {"truth": 1382.5999999999985, "filled": 1281.5999999999985, "gap": -0.07585648719707017, "seed_sd": 0.0, "tolerance": 0.02489092426663599, "passes": false}, "pre.women.b1966_1980.ylevel": {"truth": 0.04398552907354234, "filled": 0.0, "gap": "-inf", "seed_sd": 0.0, "tolerance": 0.017769920520911725, "passes": false}, "pre.women.b1966_1980.yr_cross": {"truth": 0.318767197977468, "filled": "nan", "gap": "nan", "seed_sd": 0.0, "tolerance": 0.02158530789673962, "passes": false}, "pre.women.b1966_1980.yzero": {"truth": 0.42453709903219916, "filled": 1.0, "gap": 0.8567558823917126, "seed_sd": 0.0, "tolerance": 0.014833297517636257, "passes": false}}}}, "alternative": {"passes": false, "n_gating": 136, "n_failing": 50, "cells": {"pre.men.b1930_1934.aime_p10": {"truth": 319.29999999999995, "filled": 333.45500000000004, "gap": 0.04337682399699627, "seed_sd": 7.198280130844114, "tolerance": 0.3785445178728429, "passes": true}, "pre.men.b1930_1934.aime_p25": {"truth": 961.0, "filled": 915.0, "gap": -0.04905034369477157, "seed_sd": 4.867994291394828, "tolerance": 0.1748310883030068, "passes": true}, "pre.men.b1930_1934.aime_p50": {"truth": 1871.0, "filled": 1756.225, "gap": -0.06330642816879539, "seed_sd": 4.537780004409762, "tolerance": 0.08482500842259981, "passes": true}, "pre.men.b1930_1934.aime_p75": {"truth": 2604.0, "filled": 2493.6375, "gap": -0.04330623648985288, "seed_sd": 4.3607663196664, "tolerance": 0.0526345782788754, "passes": true}, "pre.men.b1930_1934.aime_p90": {"truth": 3048.7000000000007, "filled": 2965.025, "gap": -0.027829806132161572, "seed_sd": 3.0473241065771446, "tolerance": 0.032507815796954, "passes": true}, "pre.men.b1930_1934.plevel": {"truth": 0.5292246059138269, "filled": 0.4512559722229019, "gap": -0.15937818319317099, "seed_sd": 0.00167539442003081, "tolerance": 0.05399925993701722, "passes": false}, "pre.men.b1930_1934.pr_cross": {"truth": 0.6005766012618865, "filled": 0.47070779788851286, "gap": -0.12986880337337364, "seed_sd": 0.00949471493032198, "tolerance": 0.10805017504629746, "passes": false}, "pre.men.b1930_1934.pr_in": {"truth": 0.6408121492387024, "filled": 0.5053907062038611, "gap": -0.13542144303484138, "seed_sd": 0.009558967224263285, "tolerance": 0.08754712042010183, "passes": false}, "pre.men.b1930_1934.pzero": {"truth": 0.23849477516714046, "filled": 0.19747749700699108, "gap": -0.18872276438791413, "seed_sd": 0.001329875025424285, "tolerance": 0.12848404727056906, "passes": false}, "pre.men.b1930_1945.aime_p10": {"truth": 335.0, "filled": 331.65500000000003, "gap": -0.010035259832665844, "seed_sd": 3.554311688491496, "tolerance": 0.17154652244393, "passes": true}, "pre.men.b1930_1945.aime_p25": {"truth": 1080.0, "filled": 1031.675, "gap": -0.0457773461558757, "seed_sd": 2.610832696367397, "tolerance": 0.08803344717643322, "passes": true}, "pre.men.b1930_1945.aime_p50": {"truth": 2266.0, "filled": 2180.175, "gap": -0.03861101379734322, "seed_sd": 2.852860983201545, "tolerance": 0.042293553277777104, "passes": true}, "pre.men.b1930_1945.aime_p75": {"truth": 3412.75, "filled": 3355.925, "gap": -0.016790977579084654, "seed_sd": 2.1644556870152436, "tolerance": 0.0339810042777025, "passes": true}, "pre.men.b1930_1945.aime_p90": {"truth": 4641.0, "filled": 4621.67, "gap": -0.004173748619134443, "seed_sd": 3.457120797791944, "tolerance": 0.03097131693794459, "passes": true}, "pre.men.b1930_1945.plevel": {"truth": 0.47071653085260085, "filled": 0.39204916712598037, "gap": -0.18286880924423354, "seed_sd": 0.0011927952381529446, "tolerance": 0.029173903026433003, "passes": false}, "pre.men.b1930_1945.pr_cross": {"truth": 0.5031824436185776, "filled": 0.4377212528977751, "gap": -0.06546119072080248, "seed_sd": 0.004959927945707018, "tolerance": 0.04396339753603818, "passes": false}, "pre.men.b1930_1945.pr_in": {"truth": 0.5486408320203577, "filled": 0.5051279148500414, "gap": -0.043512917170316356, "seed_sd": 0.004839756413913183, "tolerance": 0.04095810136062007, "passes": false}, "pre.men.b1930_1945.pzero": {"truth": 0.22496713863351978, "filled": 0.20559610321429558, "gap": -0.0900407608567817, "seed_sd": 0.0010776916227014178, "tolerance": 0.06285985273036603, "passes": false}, "pre.men.b1935_1939.aime_p10": {"truth": 342.0, "filled": 338.3, "gap": -0.010877661275942252, "seed_sd": 6.018392861186361, "tolerance": 0.35458282617348863, "passes": true}, "pre.men.b1935_1939.aime_p25": {"truth": 1099.5, "filled": 1038.55, "gap": -0.05703002147269842, "seed_sd": 6.176994670379337, "tolerance": 0.17611744524582296, "passes": true}, "pre.men.b1935_1939.aime_p50": {"truth": 2295.0, "filled": 2208.45, "gap": -0.03844193151511899, "seed_sd": 4.784569496115392, "tolerance": 0.08805365646562159, "passes": true}, "pre.men.b1935_1939.aime_p75": {"truth": 3360.0, "filled": 3280.05, "gap": -0.024082307792808066, "seed_sd": 4.784569496115392, "tolerance": 0.047075663502220186, "passes": true}, "pre.men.b1935_1939.aime_p90": {"truth": 4087.0, "filled": 4046.85, "gap": -0.009872403867532853, "seed_sd": 3.030980386815982, "tolerance": 0.03677617093870473, "passes": true}, "pre.men.b1935_1939.plevel": {"truth": 0.47885857925887765, "filled": 0.39017091820212124, "gap": -0.20482041726805789, "seed_sd": 0.0017982338283550564, "tolerance": 0.0505548909117149, "passes": false}, "pre.men.b1935_1939.pr_cross": {"truth": 0.5480805317128244, "filled": 0.4469978607767592, "gap": -0.10108267093606527, "seed_sd": 0.0067716623452240094, "tolerance": 0.08965478073758665, "passes": false}, "pre.men.b1935_1939.pr_in": {"truth": 0.5082293220171369, "filled": 0.45769779075131967, "gap": -0.05053153126581722, "seed_sd": 0.007181281967088367, "tolerance": 0.08171176478999366, "passes": true}, "pre.men.b1935_1939.pzero": {"truth": 0.21315539421109053, "filled": 0.2055349137510964, "gap": -0.036405533899280806, "seed_sd": 0.0014509009835175304, "tolerance": 0.12406583731865686, "passes": true}, "pre.men.b1940_1945.aime_p10": {"truth": 343.0, "filled": 326.0300000000001, "gap": -0.050741045493352566, "seed_sd": 2.2750592820500675, "tolerance": 0.2436998893100363, "passes": true}, "pre.men.b1940_1945.aime_p25": {"truth": 1178.0, "filled": 1148.925, "gap": -0.02499136264466273, "seed_sd": 4.693991119674061, "tolerance": 0.14802915955392618, "passes": true}, "pre.men.b1940_1945.aime_p50": {"truth": 2795.0, "filled": 2747.4, "gap": -0.017177096698095085, "seed_sd": 4.546832327074126, "tolerance": 0.07497681768693197, "passes": true}, "pre.men.b1940_1945.aime_p75": {"truth": 4361.5, "filled": 4334.75, "gap": -0.006152096448492017, "seed_sd": 2.4468024246479643, "tolerance": 0.04190766386026753, "passes": true}, "pre.men.b1940_1945.aime_p90": {"truth": 5432.800000000003, "filled": 5423.68, "gap": -0.0016801029698925163, "seed_sd": 2.50170468196981, "tolerance": 0.023132300299791037, "passes": true}, "pre.men.b1940_1945.plevel": {"truth": 0.3760217560197671, "filled": 0.3081285833193937, "gap": -0.1991298293071967, "seed_sd": 0.001180427121821751, "tolerance": 0.04021166929479513, "passes": false}, "pre.men.b1940_1945.pr_cross": {"truth": 0.376062044153939, "filled": 0.3566365358232604, "gap": -0.01942550833067863, "seed_sd": 0.007971024138862238, "tolerance": 0.05919050745324543, "passes": true}, "pre.men.b1940_1945.pr_in": {"truth": 0.36011521796896784, "filled": 0.3780255602140736, "gap": 0.017910342245105737, "seed_sd": 0.006428697396647492, "tolerance": 0.06500235072966114, "passes": true}, "pre.men.b1940_1945.pzero": {"truth": 0.21918652571223796, "filled": 0.2174785880666547, "gap": -0.007822682881506893, "seed_sd": 0.0016370070296969266, "tolerance": 0.09503409717716321, "passes": true}, "pre.men.b1946_1955.paime_p10": {"truth": 305.0, "filled": 297.55, "gap": -0.024729498516550485, "seed_sd": 1.7614288458371994, "tolerance": 0.11504556883383053, "passes": true}, "pre.men.b1946_1955.paime_p25": {"truth": 1022.0, "filled": 1012.25, "gap": -0.009585915851379134, "seed_sd": 1.7206180040292134, "tolerance": 0.08011617890879942, "passes": true}, "pre.men.b1946_1955.paime_p50": {"truth": 2610.0, "filled": 2589.5, "gap": -0.007885414452792894, "seed_sd": 3.137212977016048, "tolerance": 0.038612530439093365, "passes": true}, "pre.men.b1946_1955.paime_p75": {"truth": 4389.0, "filled": 4372.0, "gap": -0.0038808403918046963, "seed_sd": 2.724160904127902, "tolerance": 0.022937547887772285, "passes": true}, "pre.men.b1946_1955.paime_p90": {"truth": 5809.0, "filled": 5803.150000000001, "gap": -0.0010075654370478304, "seed_sd": 2.4639506146796553, "tolerance": 0.018317056116449435, "passes": true}, "pre.men.b1946_1955.ylevel": {"truth": 0.1479729717947154, "filled": 0.12961802758851015, "gap": -0.13243775807020652, "seed_sd": 0.00048550237036035763, "tolerance": 0.02562210127460557, "passes": false}, "pre.men.b1946_1955.yr_cross": {"truth": 0.4294012571477206, "filled": 0.42755734896351827, "gap": -0.0018439081842023253, "seed_sd": 0.003862677810953365, "tolerance": 0.03213928274663533, "passes": true}, "pre.men.b1946_1955.yzero": {"truth": 0.41447106660067734, "filled": 0.410343014999628, "gap": -0.010009737128729324, "seed_sd": 0.0015004624993788292, "tolerance": 0.021706485686521018, "passes": true}, "pre.men.b1946_1980.paime_p10": {"truth": 168.0, "filled": 185.3, "gap": 0.09801215388805495, "seed_sd": 0.6569466853317862, "tolerance": 0.055691835388134804, "passes": false}, "pre.men.b1946_1980.paime_p25": {"truth": 480.0, "filled": 505.15, "gap": 0.05106931097180478, "seed_sd": 0.7451598203705946, "tolerance": 0.03034406597699225, "passes": false}, "pre.men.b1946_1980.paime_p50": {"truth": 1237.0, "filled": 1256.15, "gap": 0.015362394256442258, "seed_sd": 0.9880869341680844, "tolerance": 0.02474263079924987, "passes": true}, "pre.men.b1946_1980.paime_p75": {"truth": 2636.0, "filled": 2630.7, "gap": -0.002012646168978449, "seed_sd": 2.4083189157584592, "tolerance": 0.02024689542128028, "passes": true}, "pre.men.b1946_1980.paime_p90": {"truth": 4302.0, "filled": 4293.74, "gap": -0.001921882826248833, "seed_sd": 2.0808399113924523, "tolerance": 0.01712820101125616, "passes": true}, "pre.men.b1946_1980.ylevel": {"truth": 0.09454735400371912, "filled": 0.09735423430573367, "gap": 0.029255417006314843, "seed_sd": 0.00024988348955974567, "tolerance": 0.01520534060461241, "passes": false}, "pre.men.b1946_1980.yr_cross": {"truth": 0.49823745301361005, "filled": 0.48664760466065066, "gap": -0.011589848352959398, "seed_sd": 0.002057777438744933, "tolerance": 0.014632893176957621, "passes": true}, "pre.men.b1946_1980.yzero": {"truth": 0.415663002913214, "filled": 0.44072148650673837, "gap": 0.058538283253072976, "seed_sd": 0.0006898565225956868, "tolerance": 0.011384493587046594, "passes": false}, "pre.men.b1956_1965.paime_p10": {"truth": 249.0, "filled": 260.71500000000003, "gap": 0.04597496021884595, "seed_sd": 1.8411309453416889, "tolerance": 0.11340392915219548, "passes": true}, "pre.men.b1956_1965.paime_p25": {"truth": 760.0, "filled": 769.0625, "gap": 0.011853807304998298, "seed_sd": 1.7638083050156288, "tolerance": 0.06443128496619711, "passes": true}, "pre.men.b1956_1965.paime_p50": {"truth": 1761.0, "filled": 1757.275, "gap": -0.002117515766600242, "seed_sd": 2.2211601994405865, "tolerance": 0.03738718984350382, "passes": true}, "pre.men.b1956_1965.paime_p75": {"truth": 2956.0, "filled": 2938.2625, "gap": -0.006018582831216257, "seed_sd": 2.4781611922425144, "tolerance": 0.02599848784066106, "passes": true}, "pre.men.b1956_1965.paime_p90": {"truth": 4109.0, "filled": 4099.719999999999, "gap": -0.0022610112059897602, "seed_sd": 3.206014085401028, "tolerance": 0.022251423931640178, "passes": true}, "pre.men.b1956_1965.ylevel": {"truth": 0.09639046670103465, "filled": 0.09251066477986337, "gap": -0.041083370793583374, "seed_sd": 0.00026376763707092367, "tolerance": 0.027201666962537053, "passes": false}, "pre.men.b1956_1965.yr_cross": {"truth": 0.40595517510382484, "filled": 0.3794224340718494, "gap": -0.02653274103197545, "seed_sd": 0.0031287878272003673, "tolerance": 0.033463015007033206, "passes": true}, "pre.men.b1956_1965.yzero": {"truth": 0.4166095146652036, "filled": 0.44938318702420965, "gap": 0.07572657958344886, "seed_sd": 0.0010074550971289882, "tolerance": 0.022693736955041323, "passes": false}, "pre.men.b1966_1980.paime_p10": {"truth": 104.0, "filled": 126.23000000000002, "gap": 0.1937147406233981, "seed_sd": 0.9205833390902217, "tolerance": 0.07746758178850374, "passes": false}, "pre.men.b1966_1980.paime_p25": {"truth": 297.0, "filled": 327.25, "gap": 0.0969922659873097, "seed_sd": 1.164157703189193, "tolerance": 0.03917957142016487, "passes": false}, "pre.men.b1966_1980.paime_p50": {"truth": 648.0, "filled": 691.7, "gap": 0.06526163925426509, "seed_sd": 1.2607433062326867, "tolerance": 0.029131134938009468, "passes": false}, "pre.men.b1966_1980.paime_p75": {"truth": 1218.0, "filled": 1261.4, "gap": 0.03501204595970808, "seed_sd": 1.353358395757909, "tolerance": 0.028246877693415603, "passes": false}, "pre.men.b1966_1980.paime_p90": {"truth": 1883.0, "filled": 1922.9300000000003, "gap": 0.02098381481301992, "seed_sd": 3.638550897209153, "tolerance": 0.02461284698314315, "passes": true}, "pre.men.b1966_1980.ylevel": {"truth": 0.055118361796504756, "filled": 0.07833231020180534, "gap": 0.35147725854645673, "seed_sd": 0.0003552933419435695, "tolerance": 0.021382204828984532, "passes": false}, "pre.men.b1966_1980.yr_cross": {"truth": 0.4093896749871457, "filled": 0.3745708705928209, "gap": -0.034818804394324776, "seed_sd": 0.0034799046915726293, "tolerance": 0.02246113667973179, "passes": false}, "pre.men.b1966_1980.yzero": {"truth": 0.41574858870643483, "filled": 0.45533442545360137, "gap": 0.09095142647226717, "seed_sd": 0.0007615661255219844, "tolerance": 0.01675184764087755, "passes": false}, "pre.women.b1930_1934.aime_p10": {"truth": 51.40000000000009, "filled": 69.41000000000001, "gap": 0.30039277689037114, "seed_sd": 1.6045658537272247, "tolerance": 0.3914251373153828, "passes": true}, "pre.women.b1930_1934.aime_p25": {"truth": 192.0, "filled": 202.675, "gap": 0.054108338845989756, "seed_sd": 2.0537193474023505, "tolerance": 0.17419751750648213, "passes": true}, "pre.women.b1930_1934.aime_p50": {"truth": 526.0, "filled": 520.6, "gap": -0.010319220177245292, "seed_sd": 3.0847673289381503, "tolerance": 0.10499905152764143, "passes": true}, "pre.women.b1930_1934.aime_p75": {"truth": 1055.0, "filled": 1030.425, "gap": -0.02356942843204557, "seed_sd": 3.7250044153983013, "tolerance": 0.0942504614966514, "passes": true}, "pre.women.b1930_1934.aime_p90": {"truth": 1663.0, "filled": 1599.4800000000002, "gap": -0.038944623789000765, "seed_sd": 6.872799169265325, "tolerance": 0.09146081269677721, "passes": true}, "pre.women.b1930_1934.plevel": {"truth": 0.17987838861322633, "filled": 0.166275116265377, "gap": -0.07863726033585361, "seed_sd": 0.0012300386992293107, "tolerance": 0.0866710956881841, "passes": true}, "pre.women.b1930_1934.pr_cross": {"truth": 0.5714979296198577, "filled": 0.4748597988631834, "gap": -0.09663813075667427, "seed_sd": 0.01059299078241411, "tolerance": 0.10849315671129696, "passes": true}, "pre.women.b1930_1934.pr_in": {"truth": 0.5486860492872828, "filled": 0.4506119091531285, "gap": -0.09807414013415433, "seed_sd": 0.010727888439518692, "tolerance": 0.1270201928993476, "passes": true}, "pre.women.b1930_1934.pzero": {"truth": 0.5514059876498446, "filled": 0.4412112523637662, "gap": -0.22294756647395808, "seed_sd": 0.0023214886803956196, "tolerance": 0.04703504786975244, "passes": false}, "pre.women.b1930_1945.aime_p10": {"truth": 79.0, "filled": 89.05, "gap": 0.11975015726864946, "seed_sd": 1.1459310165698642, "tolerance": 0.16621909484652816, "passes": true}, "pre.women.b1930_1945.aime_p25": {"truth": 280.0, "filled": 276.85, "gap": -0.011313759900272835, "seed_sd": 1.7252002172135514, "tolerance": 0.09520232121757725, "passes": true}, "pre.women.b1930_1945.aime_p50": {"truth": 771.0, "filled": 742.05, "gap": -0.038271747221502395, "seed_sd": 1.8202082009311031, "tolerance": 0.06218129686271377, "passes": true}, "pre.women.b1930_1945.aime_p75": {"truth": 1574.0, "filled": 1527.0, "gap": -0.030315123758716034, "seed_sd": 2.9019050004400464, "tolerance": 0.0540210895745143, "passes": true}, "pre.women.b1930_1945.aime_p90": {"truth": 2560.0, "filled": 2502.3, "gap": -0.022796949557932322, "seed_sd": 3.388836161031659, "tolerance": 0.04855891958559449, "passes": true}, "pre.women.b1930_1945.plevel": {"truth": 0.18433938803699407, "filled": 0.15443381522319435, "gap": -0.17701293468413826, "seed_sd": 0.0007451874725392632, "tolerance": 0.050000040503342155, "passes": false}, "pre.women.b1930_1945.pr_cross": {"truth": 0.4395941263612976, "filled": 0.38852744136167233, "gap": -0.051066684999625245, "seed_sd": 0.005151871755167338, "tolerance": 0.0677006808491353, "passes": true}, "pre.women.b1930_1945.pr_in": {"truth": 0.372715069734525, "filled": 0.34945624096577543, "gap": -0.023258828768749573, "seed_sd": 0.006480298607911677, "tolerance": 0.07412154624977427, "passes": true}, "pre.women.b1930_1945.pzero": {"truth": 0.5120706676834471, "filled": 0.42671014627380915, "gap": -0.18235766995166947, "seed_sd": 0.0016892997635133166, "tolerance": 0.02913422225552515, "passes": false}, "pre.women.b1935_1939.aime_p10": {"truth": 75.0, "filled": 87.8, "gap": 0.15757338710476088, "seed_sd": 1.7044832524535805, "tolerance": 0.3512121935176347, "passes": true}, "pre.women.b1935_1939.aime_p25": {"truth": 267.0, "filled": 264.3625, "gap": -0.00992739104138085, "seed_sd": 2.5150533634602334, "tolerance": 0.180762984617868, "passes": true}, "pre.women.b1935_1939.aime_p50": {"truth": 726.0, "filled": 693.825, "gap": -0.04533024749930359, "seed_sd": 2.749521489469146, "tolerance": 0.11655697880988479, "passes": true}, "pre.women.b1935_1939.aime_p75": {"truth": 1448.0, "filled": 1392.45, "gap": -0.03911850843160991, "seed_sd": 4.525948577515634, "tolerance": 0.08830721415636139, "passes": true}, "pre.women.b1935_1939.aime_p90": {"truth": 2293.8999999999996, "filled": 2210.7949999999996, "gap": -0.03690124642828341, "seed_sd": 7.21989028929614, "tolerance": 0.08570712841284106, "passes": true}, "pre.women.b1935_1939.plevel": {"truth": 0.18459645468504016, "filled": 0.15058972619927874, "gap": -0.20361302258404357, "seed_sd": 0.0011346805174700786, "tolerance": 0.09407530912273358, "passes": false}, "pre.women.b1935_1939.pr_cross": {"truth": 0.5177687037806673, "filled": 0.4330974292581584, "gap": -0.08467127452250889, "seed_sd": 0.008445900072306159, "tolerance": 0.1216792156910919, "passes": true}, "pre.women.b1935_1939.pr_in": {"truth": 0.48588537175161917, "filled": 0.3844627635977695, "gap": -0.10142260815384968, "seed_sd": 0.012464065495063868, "tolerance": 0.13268326765003252, "passes": true}, "pre.women.b1935_1939.pzero": {"truth": 0.5152445437812253, "filled": 0.43101465947935835, "gap": -0.1784995280169943, "seed_sd": 0.0024295006714432823, "tolerance": 0.0510656392812699, "passes": false}, "pre.women.b1940_1945.aime_p10": {"truth": 110.0, "filled": 109.88000000000002, "gap": -0.0010915045653430155, "seed_sd": 1.3820655784577995, "tolerance": 0.26939110963753704, "passes": true}, "pre.women.b1940_1945.aime_p25": {"truth": 386.0, "filled": 366.6375, "gap": -0.051463747964931805, "seed_sd": 2.6251253102922236, "tolerance": 0.14001979176307694, "passes": true}, "pre.women.b1940_1945.aime_p50": {"truth": 1041.0, "filled": 1006.05, "gap": -0.034150017401113786, "seed_sd": 2.6102026539594285, "tolerance": 0.08949151188066314, "passes": true}, "pre.women.b1940_1945.aime_p75": {"truth": 2058.0, "filled": 2005.95, "gap": -0.02561687340707941, "seed_sd": 3.0753690407562644, "tolerance": 0.06627004271644896, "passes": true}, "pre.women.b1940_1945.aime_p90": {"truth": 3179.7000000000007, "filled": 3143.15, "gap": -0.011561370939153548, "seed_sd": 4.8068481850049665, "tolerance": 0.06421030653828377, "passes": true}, "pre.women.b1940_1945.plevel": {"truth": 0.19012308454067162, "filled": 0.14273877017557055, "gap": -0.2866554981221232, "seed_sd": 0.000855197366426852, "tolerance": 0.06996499056922599, "passes": false}, "pre.women.b1940_1945.pr_cross": {"truth": 0.3145537439002571, "filled": 0.31177953224975996, "gap": -0.002774211650497127, "seed_sd": 0.010577622662610095, "tolerance": 0.10514700123361043, "passes": true}, "pre.women.b1940_1945.pr_in": {"truth": 0.19634726483554435, "filled": 0.2748579888301229, "gap": 0.07851072399457856, "seed_sd": 0.00870178568609491, "tolerance": 0.10970200704526734, "passes": true}, "pre.women.b1940_1945.pzero": {"truth": 0.45477910425062734, "filled": 0.40196372899399285, "gap": -0.12344995773614209, "seed_sd": 0.0026140225069639154, "tolerance": 0.04701699812231318, "passes": false}, "pre.women.b1946_1955.paime_p10": {"truth": 156.0, "filled": 155.46999999999997, "gap": -0.003403220287887976, "seed_sd": 0.8933084573650976, "tolerance": 0.1092839592303832, "passes": true}, "pre.women.b1946_1955.paime_p25": {"truth": 516.0, "filled": 513.2, "gap": -0.0054411327400814, "seed_sd": 1.1516578439248717, "tolerance": 0.06230795442282464, "passes": true}, "pre.women.b1946_1955.paime_p50": {"truth": 1311.0, "filled": 1302.675, "gap": -0.006370362155514009, "seed_sd": 1.5241477342401797, "tolerance": 0.0404504662154056, "passes": true}, "pre.women.b1946_1955.paime_p75": {"truth": 2484.0, "filled": 2469.95, "gap": -0.005672256551217281, "seed_sd": 1.8771478925557026, "tolerance": 0.0286716465849327, "passes": true}, "pre.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3774.0200000000013, "gap": -0.0015832632821144443, "seed_sd": 2.8162311508004154, "tolerance": 0.03102321311868942, "passes": true}, "pre.women.b1946_1955.ylevel": {"truth": 0.09059452101103667, "filled": 0.08651283195139477, "gap": -0.046100987533894244, "seed_sd": 0.00039502547096333595, "tolerance": 0.029569090149685853, "passes": false}, "pre.women.b1946_1955.yr_cross": {"truth": 0.34158903474392094, "filled": 0.38173946433515016, "gap": 0.04015042959122922, "seed_sd": 0.00490792684904385, "tolerance": 0.03669253843323441, "passes": false}, "pre.women.b1946_1955.yzero": {"truth": 0.5347643649882455, "filled": 0.49023495161554864, "gap": -0.08694144132228299, "seed_sd": 0.0011450982333459034, "tolerance": 0.015053483015288157, "passes": false}, "pre.women.b1946_1980.paime_p10": {"truth": 103.0, "filled": 114.75, "gap": 0.10802686071101864, "seed_sd": 0.5501196042201808, "tolerance": 0.05217708504283977, "passes": false}, "pre.women.b1946_1980.paime_p25": {"truth": 304.0, "filled": 319.85, "gap": 0.050824434489924464, "seed_sd": 0.6708203932499369, "tolerance": 0.031237246580028546, "passes": false}, "pre.women.b1946_1980.paime_p50": {"truth": 752.0, "filled": 771.15, "gap": 0.025146583219783025, "seed_sd": 0.8127277008872491, "tolerance": 0.021471785135136603, "passes": false}, "pre.women.b1946_1980.paime_p75": {"truth": 1583.75, "filled": 1589.05, "gap": 0.0033409007373377264, "seed_sd": 1.2763022245616642, "tolerance": 0.021247962622687265, "passes": true}, "pre.women.b1946_1980.paime_p90": {"truth": 2690.0, "filled": 2691.95, "gap": 0.0007246444449799938, "seed_sd": 1.9049796241485244, "tolerance": 0.019286406429246394, "passes": true}, "pre.women.b1946_1980.ylevel": {"truth": 0.06340849534928693, "filled": 0.07062478439851178, "gap": 0.10778328818546301, "seed_sd": 0.00021066219000565875, "tolerance": 0.015450091379218515, "passes": false}, "pre.women.b1946_1980.yr_cross": {"truth": 0.43636860215681333, "filled": 0.45372177216986065, "gap": 0.017353170013047314, "seed_sd": 0.002315286295492016, "tolerance": 0.015684351176846988, "passes": false}, "pre.women.b1946_1980.yzero": {"truth": 0.4719000529035934, "filled": 0.4926267796826947, "gap": 0.042984637320677144, "seed_sd": 0.0006742279130238006, "tolerance": 0.008556191699814752, "passes": false}, "pre.women.b1956_1965.paime_p10": {"truth": 136.0, "filled": 147.4, "gap": 0.08049509401918353, "seed_sd": 1.2732056517228265, "tolerance": 0.10309020846799896, "passes": true}, "pre.women.b1956_1965.paime_p25": {"truth": 422.0, "filled": 429.75, "gap": 0.01819833022694617, "seed_sd": 1.0699237552766379, "tolerance": 0.05400647590482275, "passes": true}, "pre.women.b1956_1965.paime_p50": {"truth": 996.0, "filled": 1001.6, "gap": 0.00560674276123585, "seed_sd": 1.6982963599783725, "tolerance": 0.03402127173688933, "passes": true}, "pre.women.b1956_1965.paime_p75": {"truth": 1839.0, "filled": 1839.85, "gap": 0.000462100936502452, "seed_sd": 1.8144159564878983, "tolerance": 0.026722598015179132, "passes": true}, "pre.women.b1956_1965.paime_p90": {"truth": 2790.5999999999985, "filled": 2792.7999999999997, "gap": 0.0007880503327202248, "seed_sd": 3.231424649153998, "tolerance": 0.026506809111477153, "passes": true}, "pre.women.b1956_1965.ylevel": {"truth": 0.06347343380872424, "filled": 0.06691935721416442, "gap": 0.05286681764248957, "seed_sd": 0.0003901465800188738, "tolerance": 0.026809792541125404, "passes": false}, "pre.women.b1956_1965.yr_cross": {"truth": 0.3538451043271927, "filled": 0.3495167466131729, "gap": -0.004328357714019793, "seed_sd": 0.003718149045814111, "tolerance": 0.03047360671843986, "passes": true}, "pre.women.b1956_1965.yzero": {"truth": 0.47511142887567126, "filled": 0.4996823706588375, "gap": 0.05042327424751103, "seed_sd": 0.0011326514137311617, "tolerance": 0.017102901274216892, "passes": false}, "pre.women.b1966_1980.paime_p10": {"truth": 73.0, "filled": 87.52000000000002, "gap": 0.18140789752527997, "seed_sd": 0.6786984447804205, "tolerance": 0.0627268236453548, "passes": false}, "pre.women.b1966_1980.paime_p25": {"truth": 205.0, "filled": 224.55, "gap": 0.09108842039533904, "seed_sd": 0.8870412083230169, "tolerance": 0.039676687855380324, "passes": false}, "pre.women.b1966_1980.paime_p50": {"truth": 458.0, "filled": 486.35, "gap": 0.06005934520126388, "seed_sd": 1.4964871146156007, "tolerance": 0.02838346665574009, "passes": false}, "pre.women.b1966_1980.paime_p75": {"truth": 861.0, "filled": 901.2, "gap": 0.0456327041303588, "seed_sd": 1.9695043458751373, "tolerance": 0.023922104075604994, "passes": false}, "pre.women.b1966_1980.paime_p90": {"truth": 1382.5999999999985, "filled": 1421.5699999999997, "gap": 0.027796110124051587, "seed_sd": 2.5250534126710935, "tolerance": 0.02489092426663599, "passes": false}, "pre.women.b1966_1980.ylevel": {"truth": 0.04398552907354234, "filled": 0.06226928105552239, "gap": 0.3476075280866513, "seed_sd": 0.0003490420029782692, "tolerance": 0.017769920520911725, "passes": false}, "pre.women.b1966_1980.yr_cross": {"truth": 0.318767197977468, "filled": 0.3300755287219385, "gap": 0.011308330744470518, "seed_sd": 0.004440189459906143, "tolerance": 0.02158530789673962, "passes": true}, "pre.women.b1966_1980.yzero": {"truth": 0.42453709903219916, "filled": 0.4886847647452675, "gap": 0.14071823213377666, "seed_sd": 0.0010570800354365248, "tolerance": 0.014833297517636257, "passes": false}}, "tier": "not_adopted"}, "primary": {"passes": true, "n_gating": 136, "n_failing": 0, "cells": {"pre.men.b1930_1934.aime_p10": {"truth": 319.29999999999995, "filled": 319.01000000000005, "gap": -0.0009086494648462562, "seed_sd": 3.4908753237999974, "tolerance": 0.3785445178728429, "passes": true}, "pre.men.b1930_1934.aime_p25": {"truth": 961.0, "filled": 955.5125, "gap": -0.0057265632195964145, "seed_sd": 5.253742400472859, "tolerance": 0.1748310883030068, "passes": true}, "pre.men.b1930_1934.aime_p50": {"truth": 1871.0, "filled": 1864.975, "gap": -0.003225399111755678, "seed_sd": 3.4659053651246743, "tolerance": 0.08482500842259981, "passes": true}, "pre.men.b1930_1934.aime_p75": {"truth": 2604.0, "filled": 2607.125, "gap": 0.0011993572883390868, "seed_sd": 2.5860201081971503, "tolerance": 0.0526345782788754, "passes": true}, "pre.men.b1930_1934.aime_p90": {"truth": 3048.7000000000007, "filled": 3048.425, "gap": -9.02064498227162e-05, "seed_sd": 2.036735003303978, "tolerance": 0.032507815796954, "passes": true}, "pre.men.b1930_1934.plevel": {"truth": 0.5292246059138269, "filled": 0.5291854236691471, "gap": -7.403982124609687e-05, "seed_sd": 0.0013655795254600492, "tolerance": 0.05399925993701722, "passes": true}, "pre.men.b1930_1934.pr_cross": {"truth": 0.6005766012618865, "filled": 0.6132148513475164, "gap": 0.01263825008562991, "seed_sd": 0.0042711947827688895, "tolerance": 0.10805017504629746, "passes": true}, "pre.men.b1930_1934.pr_in": {"truth": 0.6408121492387024, "filled": 0.6569132735924044, "gap": 0.016101124353701923, "seed_sd": 0.005878717774556174, "tolerance": 0.08754712042010183, "passes": true}, "pre.men.b1930_1934.pzero": {"truth": 0.23849477516714046, "filled": 0.2343592497524301, "gap": -0.017492209622133048, "seed_sd": 0.0017178638387153479, "tolerance": 0.12848404727056906, "passes": true}, "pre.men.b1930_1945.aime_p10": {"truth": 335.0, "filled": 334.91, "gap": -0.00026869281109842547, "seed_sd": 1.7480515468733664, "tolerance": 0.17154652244393, "passes": true}, "pre.men.b1930_1945.aime_p25": {"truth": 1080.0, "filled": 1080.075, "gap": 6.9442033289846e-05, "seed_sd": 2.3114417920195907, "tolerance": 0.08803344717643322, "passes": true}, "pre.men.b1930_1945.aime_p50": {"truth": 2266.0, "filled": 2269.95, "gap": 0.00174164221319284, "seed_sd": 1.4039418191573851, "tolerance": 0.042293553277777104, "passes": true}, "pre.men.b1930_1945.aime_p75": {"truth": 3412.75, "filled": 3411.5875, "gap": -0.00034069241483081214, "seed_sd": 2.272945862693795, "tolerance": 0.0339810042777025, "passes": true}, "pre.men.b1930_1945.aime_p90": {"truth": 4641.0, "filled": 4645.5, "gap": 0.0009691488401912807, "seed_sd": 3.063365881888381, "tolerance": 0.03097131693794459, "passes": true}, "pre.men.b1930_1945.plevel": {"truth": 0.47071653085260085, "filled": 0.4709284946535627, "gap": 0.00045019895778231067, "seed_sd": 0.0007883479291290621, "tolerance": 0.029173903026433003, "passes": true}, "pre.men.b1930_1945.pr_cross": {"truth": 0.5031824436185776, "filled": 0.5142593494210748, "gap": 0.011076905802497206, "seed_sd": 0.0033007110869924168, "tolerance": 0.04396339753603818, "passes": true}, "pre.men.b1930_1945.pr_in": {"truth": 0.5486408320203577, "filled": 0.5533205928140609, "gap": 0.004679760793703136, "seed_sd": 0.0028791732273558686, "tolerance": 0.04095810136062007, "passes": true}, "pre.men.b1930_1945.pzero": {"truth": 0.22496713863351978, "filled": 0.2234473251045051, "gap": -0.00677863660306266, "seed_sd": 0.0010310919431983968, "tolerance": 0.06285985273036603, "passes": true}, "pre.men.b1935_1939.aime_p10": {"truth": 342.0, "filled": 351.05, "gap": 0.026117926400653246, "seed_sd": 3.9132030224276426, "tolerance": 0.35458282617348863, "passes": true}, "pre.men.b1935_1939.aime_p25": {"truth": 1099.5, "filled": 1097.875, "gap": -0.0014790377575346625, "seed_sd": 3.1366886663285034, "tolerance": 0.17611744524582296, "passes": true}, "pre.men.b1935_1939.aime_p50": {"truth": 2295.0, "filled": 2303.65, "gap": 0.0037619780594635444, "seed_sd": 3.2971279367927098, "tolerance": 0.08805365646562159, "passes": true}, "pre.men.b1935_1939.aime_p75": {"truth": 3360.0, "filled": 3361.725, "gap": 0.0005132611161187128, "seed_sd": 3.0714089000943767, "tolerance": 0.047075663502220186, "passes": true}, "pre.men.b1935_1939.aime_p90": {"truth": 4087.0, "filled": 4089.05, "gap": 0.0005014646541940948, "seed_sd": 4.148239956516752, "tolerance": 0.03677617093870473, "passes": true}, "pre.men.b1935_1939.plevel": {"truth": 0.47885857925887765, "filled": 0.480522922686636, "gap": 0.0034696209881766027, "seed_sd": 0.0014865531284624595, "tolerance": 0.0505548909117149, "passes": true}, "pre.men.b1935_1939.pr_cross": {"truth": 0.5480805317128244, "filled": 0.575558051296075, "gap": 0.02747751958325051, "seed_sd": 0.004077817588025134, "tolerance": 0.08965478073758665, "passes": true}, "pre.men.b1935_1939.pr_in": {"truth": 0.5082293220171369, "filled": 0.5135920240451662, "gap": 0.005362702028029354, "seed_sd": 0.006155107199129598, "tolerance": 0.08171176478999366, "passes": true}, "pre.men.b1935_1939.pzero": {"truth": 0.21315539421109053, "filled": 0.21182816002338956, "gap": -0.006246069945369914, "seed_sd": 0.0015124780729217227, "tolerance": 0.12406583731865686, "passes": true}, "pre.men.b1940_1945.aime_p10": {"truth": 343.0, "filled": 334.09000000000003, "gap": -0.026320029409510504, "seed_sd": 2.4721394952976294, "tolerance": 0.2436998893100363, "passes": true}, "pre.men.b1940_1945.aime_p25": {"truth": 1178.0, "filled": 1176.775, "gap": -0.0010404392016276631, "seed_sd": 3.614572115610279, "tolerance": 0.14802915955392618, "passes": true}, "pre.men.b1940_1945.aime_p50": {"truth": 2795.0, "filled": 2792.95, "gap": -0.000733721701864809, "seed_sd": 3.677456214630749, "tolerance": 0.07497681768693197, "passes": true}, "pre.men.b1940_1945.aime_p75": {"truth": 4361.5, "filled": 4359.025, "gap": -0.0005676263909464296, "seed_sd": 3.290636636531632, "tolerance": 0.04190766386026753, "passes": true}, "pre.men.b1940_1945.aime_p90": {"truth": 5432.800000000003, "filled": 5432.7, "gap": -1.8406884176869198e-05, "seed_sd": 1.2422390650503299, "tolerance": 0.023132300299791037, "passes": true}, "pre.men.b1940_1945.plevel": {"truth": 0.3760217560197671, "filled": 0.37489008956602926, "gap": -0.0030141149514301135, "seed_sd": 0.0009568302876987612, "tolerance": 0.04021166929479513, "passes": true}, "pre.men.b1940_1945.pr_cross": {"truth": 0.376062044153939, "filled": 0.37480361977220633, "gap": -0.0012584243817326812, "seed_sd": 0.0061954412945916595, "tolerance": 0.05919050745324543, "passes": true}, "pre.men.b1940_1945.pr_in": {"truth": 0.36011521796896784, "filled": 0.3591838645046824, "gap": -0.0009313534642854115, "seed_sd": 0.006786464340961721, "tolerance": 0.06500235072966114, "passes": true}, "pre.men.b1940_1945.pzero": {"truth": 0.21918652571223796, "filled": 0.2212452965418384, "gap": 0.009348942200510635, "seed_sd": 0.0012275780068450736, "tolerance": 0.09503409717716321, "passes": true}, "pre.men.b1946_1955.paime_p10": {"truth": 305.0, "filled": 303.13, "gap": -0.006150020206342255, "seed_sd": 1.4542496708829384, "tolerance": 0.11504556883383053, "passes": true}, "pre.men.b1946_1955.paime_p25": {"truth": 1022.0, "filled": 1023.3, "gap": 0.0012712073290597203, "seed_sd": 2.0862961388420795, "tolerance": 0.08011617890879942, "passes": true}, "pre.men.b1946_1955.paime_p50": {"truth": 2610.0, "filled": 2607.7, "gap": -0.0008816145615773152, "seed_sd": 1.8381913307436342, "tolerance": 0.038612530439093365, "passes": true}, "pre.men.b1946_1955.paime_p75": {"truth": 4389.0, "filled": 4387.375, "gap": -0.0003703123484513071, "seed_sd": 1.5293187337745144, "tolerance": 0.022937547887772285, "passes": true}, "pre.men.b1946_1955.paime_p90": {"truth": 5809.0, "filled": 5810.430000000001, "gap": 0.00024613944181695047, "seed_sd": 3.429454460222852, "tolerance": 0.018317056116449435, "passes": true}, "pre.men.b1946_1955.ylevel": {"truth": 0.1479729717947154, "filled": 0.14659159386435963, "gap": -0.009379186894413527, "seed_sd": 0.0002957004696019843, "tolerance": 0.02562210127460557, "passes": true}, "pre.men.b1946_1955.yr_cross": {"truth": 0.4294012571477206, "filled": 0.4399009712087537, "gap": 0.010499714061033116, "seed_sd": 0.0031621441318267795, "tolerance": 0.03213928274663533, "passes": true}, "pre.men.b1946_1955.yzero": {"truth": 0.41447106660067734, "filled": 0.4158327429878276, "gap": 0.0032799502836826644, "seed_sd": 0.0006758004614627337, "tolerance": 0.021706485686521018, "passes": true}, "pre.men.b1946_1980.paime_p10": {"truth": 168.0, "filled": 168.15, "gap": 0.0008924587830199116, "seed_sd": 0.36634754853252327, "tolerance": 0.055691835388134804, "passes": true}, "pre.men.b1946_1980.paime_p25": {"truth": 480.0, "filled": 479.2, "gap": -0.0016680571006970624, "seed_sd": 0.523148363780597, "tolerance": 0.03034406597699225, "passes": true}, "pre.men.b1946_1980.paime_p50": {"truth": 1237.0, "filled": 1234.35, "gap": -0.0021445776726558563, "seed_sd": 0.7451598203705947, "tolerance": 0.02474263079924987, "passes": true}, "pre.men.b1946_1980.paime_p75": {"truth": 2636.0, "filled": 2636.7625, "gap": 0.00028922220764382445, "seed_sd": 1.2016299237636396, "tolerance": 0.02024689542128028, "passes": true}, "pre.men.b1946_1980.paime_p90": {"truth": 4302.0, "filled": 4304.3949999999995, "gap": 0.0005565628958059676, "seed_sd": 1.633361081038826, "tolerance": 0.01712820101125616, "passes": true}, "pre.men.b1946_1980.ylevel": {"truth": 0.09454735400371912, "filled": 0.09367892426336906, "gap": -0.009227573433769454, "seed_sd": 9.960753662665636e-05, "tolerance": 0.01520534060461241, "passes": true}, "pre.men.b1946_1980.yr_cross": {"truth": 0.49823745301361005, "filled": 0.5047654750530509, "gap": 0.006528022039440862, "seed_sd": 0.0008904235198013295, "tolerance": 0.014632893176957621, "passes": true}, "pre.men.b1946_1980.yzero": {"truth": 0.415663002913214, "filled": 0.41931923269525295, "gap": 0.008757678893166476, "seed_sd": 0.0005288771136011216, "tolerance": 0.011384493587046594, "passes": true}, "pre.men.b1956_1965.paime_p10": {"truth": 249.0, "filled": 249.91500000000005, "gap": 0.0036679635844345526, "seed_sd": 0.9702278625473394, "tolerance": 0.11340392915219548, "passes": true}, "pre.men.b1956_1965.paime_p25": {"truth": 760.0, "filled": 760.2125, "gap": 0.0002795661808914218, "seed_sd": 1.6824852388497704, "tolerance": 0.06443128496619711, "passes": true}, "pre.men.b1956_1965.paime_p50": {"truth": 1761.0, "filled": 1759.05, "gap": -0.0011079389210228996, "seed_sd": 1.6928642809405043, "tolerance": 0.03738718984350382, "passes": true}, "pre.men.b1956_1965.paime_p75": {"truth": 2956.0, "filled": 2952.125, "gap": -0.001311753070776689, "seed_sd": 1.6984900414935118, "tolerance": 0.02599848784066106, "passes": true}, "pre.men.b1956_1965.paime_p90": {"truth": 4109.0, "filled": 4109.58, "gap": 0.00014114360411632276, "seed_sd": 2.4312872916908015, "tolerance": 0.022251423931640178, "passes": true}, "pre.men.b1956_1965.ylevel": {"truth": 0.09639046670103465, "filled": 0.09526375922291459, "gap": -0.011757846225209256, "seed_sd": 0.0002008564053301557, "tolerance": 0.027201666962537053, "passes": true}, "pre.men.b1956_1965.yr_cross": {"truth": 0.40595517510382484, "filled": 0.4183289514486261, "gap": 0.012373776344801246, "seed_sd": 0.0029119168385594732, "tolerance": 0.033463015007033206, "passes": true}, "pre.men.b1956_1965.yzero": {"truth": 0.4166095146652036, "filled": 0.42287151658551086, "gap": 0.014919022193501164, "seed_sd": 0.0009039797761472841, "tolerance": 0.022693736955041323, "passes": true}, "pre.men.b1966_1980.paime_p10": {"truth": 104.0, "filled": 104.58000000000001, "gap": 0.005561429618611946, "seed_sd": 0.5908067633151707, "tolerance": 0.07746758178850374, "passes": true}, "pre.men.b1966_1980.paime_p25": {"truth": 297.0, "filled": 295.5, "gap": -0.005063301956546695, "seed_sd": 0.6882472016116853, "tolerance": 0.03917957142016487, "passes": true}, "pre.men.b1966_1980.paime_p50": {"truth": 648.0, "filled": 650.0, "gap": 0.0030816665374082675, "seed_sd": 0.8583950752789521, "tolerance": 0.029131134938009468, "passes": true}, "pre.men.b1966_1980.paime_p75": {"truth": 1218.0, "filled": 1217.95, "gap": -4.105174573165726e-05, "seed_sd": 0.7591546545162483, "tolerance": 0.028246877693415603, "passes": true}, "pre.men.b1966_1980.paime_p90": {"truth": 1883.0, "filled": 1882.02, "gap": -0.0005205815757323151, "seed_sd": 1.0092206477224783, "tolerance": 0.02461284698314315, "passes": true}, "pre.men.b1966_1980.ylevel": {"truth": 0.055118361796504756, "filled": 0.05482193056978226, "gap": -0.005392598813348748, "seed_sd": 0.00010597279122111407, "tolerance": 0.021382204828984532, "passes": true}, "pre.men.b1966_1980.yr_cross": {"truth": 0.4093896749871457, "filled": 0.4115686971693808, "gap": 0.0021790221822351463, "seed_sd": 0.0019117001884655493, "tolerance": 0.02246113667973179, "passes": true}, "pre.men.b1966_1980.yzero": {"truth": 0.41574858870643483, "filled": 0.4189394792286537, "gap": 0.007645745013141858, "seed_sd": 0.0007869981860754034, "tolerance": 0.01675184764087755, "passes": true}, "pre.women.b1930_1934.aime_p10": {"truth": 51.40000000000009, "filled": 55.19000000000001, "gap": 0.07114360499028916, "seed_sd": 1.0104194022390238, "tolerance": 0.3914251373153828, "passes": true}, "pre.women.b1930_1934.aime_p25": {"truth": 192.0, "filled": 189.6, "gap": -0.012578782206859707, "seed_sd": 2.588435821108957, "tolerance": 0.17419751750648213, "passes": true}, "pre.women.b1930_1934.aime_p50": {"truth": 526.0, "filled": 520.85, "gap": -0.009839120307252536, "seed_sd": 3.0482954684803594, "tolerance": 0.10499905152764143, "passes": true}, "pre.women.b1930_1934.aime_p75": {"truth": 1055.0, "filled": 1054.4, "gap": -0.0005688821619243001, "seed_sd": 3.8784153030790676, "tolerance": 0.0942504614966514, "passes": true}, "pre.women.b1930_1934.aime_p90": {"truth": 1663.0, "filled": 1662.6100000000001, "gap": -0.00023454343821960322, "seed_sd": 4.867280663022405, "tolerance": 0.09146081269677721, "passes": true}, "pre.women.b1930_1934.plevel": {"truth": 0.17987838861322633, "filled": 0.177081021329544, "gap": -0.015673628277671492, "seed_sd": 0.0011236798865642981, "tolerance": 0.0866710956881841, "passes": true}, "pre.women.b1930_1934.pr_cross": {"truth": 0.5714979296198577, "filled": 0.6067825253454213, "gap": 0.03528459572556364, "seed_sd": 0.008863825559077805, "tolerance": 0.10849315671129696, "passes": true}, "pre.women.b1930_1934.pr_in": {"truth": 0.5486860492872828, "filled": 0.5817695222067873, "gap": 0.03308347291950453, "seed_sd": 0.013200978751289429, "tolerance": 0.1270201928993476, "passes": true}, "pre.women.b1930_1934.pzero": {"truth": 0.5514059876498446, "filled": 0.5527799005963105, "gap": 0.002488554998091752, "seed_sd": 0.0013713513850248907, "tolerance": 0.04703504786975244, "passes": true}, "pre.women.b1930_1945.aime_p10": {"truth": 79.0, "filled": 78.0, "gap": -0.012739025777429802, "seed_sd": 0.7254762501100116, "tolerance": 0.16621909484652816, "passes": true}, "pre.women.b1930_1945.aime_p25": {"truth": 280.0, "filled": 279.0, "gap": -0.0035778213478838694, "seed_sd": 1.3377121081198773, "tolerance": 0.09520232121757725, "passes": true}, "pre.women.b1930_1945.aime_p50": {"truth": 771.0, "filled": 761.95, "gap": -0.01180743682745966, "seed_sd": 1.669383750149485, "tolerance": 0.06218129686271377, "passes": true}, "pre.women.b1930_1945.aime_p75": {"truth": 1574.0, "filled": 1574.6, "gap": 0.0003811217730183003, "seed_sd": 2.891002375793231, "tolerance": 0.0540210895745143, "passes": true}, "pre.women.b1930_1945.aime_p90": {"truth": 2560.0, "filled": 2559.55, "gap": -0.00017579670133471836, "seed_sd": 3.677456214630749, "tolerance": 0.04855891958559449, "passes": true}, "pre.women.b1930_1945.plevel": {"truth": 0.18433938803699407, "filled": 0.17995218343257352, "gap": -0.024087390805605846, "seed_sd": 0.0007199009581150091, "tolerance": 0.050000040503342155, "passes": true}, "pre.women.b1930_1945.pr_cross": {"truth": 0.4395941263612976, "filled": 0.47329830691293084, "gap": 0.03370418055163327, "seed_sd": 0.006410383721397189, "tolerance": 0.0677006808491353, "passes": true}, "pre.women.b1930_1945.pr_in": {"truth": 0.372715069734525, "filled": 0.40355800362085176, "gap": 0.03084293388632675, "seed_sd": 0.005884207173703561, "tolerance": 0.07412154624977427, "passes": true}, "pre.women.b1930_1945.pzero": {"truth": 0.5120706676834471, "filled": 0.517344820360381, "gap": 0.01024697779893513, "seed_sd": 0.0010493346222991984, "tolerance": 0.02913422225552515, "passes": true}, "pre.women.b1935_1939.aime_p10": {"truth": 75.0, "filled": 73.26000000000002, "gap": -0.023473356185641947, "seed_sd": 1.8590886278808187, "tolerance": 0.3512121935176347, "passes": true}, "pre.women.b1935_1939.aime_p25": {"truth": 267.0, "filled": 267.725, "gap": 0.002711675886689413, "seed_sd": 3.2372218433647535, "tolerance": 0.180762984617868, "passes": true}, "pre.women.b1935_1939.aime_p50": {"truth": 726.0, "filled": 716.075, "gap": -0.01376510474655035, "seed_sd": 3.5919317482672923, "tolerance": 0.11655697880988479, "passes": true}, "pre.women.b1935_1939.aime_p75": {"truth": 1448.0, "filled": 1437.2875, "gap": -0.007425637288465126, "seed_sd": 3.934726466785341, "tolerance": 0.08830721415636139, "passes": true}, "pre.women.b1935_1939.aime_p90": {"truth": 2293.8999999999996, "filled": 2280.29, "gap": -0.005950797917480877, "seed_sd": 5.162914313692055, "tolerance": 0.08570712841284106, "passes": true}, "pre.women.b1935_1939.plevel": {"truth": 0.18459645468504016, "filled": 0.17846485916570784, "gap": -0.03378040207475497, "seed_sd": 0.0010822452970810686, "tolerance": 0.09407530912273358, "passes": true}, "pre.women.b1935_1939.pr_cross": {"truth": 0.5177687037806673, "filled": 0.5511541429766346, "gap": 0.03338543919596726, "seed_sd": 0.010162657342750367, "tolerance": 0.1216792156910919, "passes": true}, "pre.women.b1935_1939.pr_in": {"truth": 0.48588537175161917, "filled": 0.5109930990815804, "gap": 0.02510772732996125, "seed_sd": 0.011878561501692732, "tolerance": 0.13268326765003252, "passes": true}, "pre.women.b1935_1939.pzero": {"truth": 0.5152445437812253, "filled": 0.5244790297133843, "gap": 0.017763815300941732, "seed_sd": 0.0019054062085465567, "tolerance": 0.0510656392812699, "passes": true}, "pre.women.b1940_1945.aime_p10": {"truth": 110.0, "filled": 107.55, "gap": -0.02252451007867773, "seed_sd": 1.5719582155957412, "tolerance": 0.26939110963753704, "passes": true}, "pre.women.b1940_1945.aime_p25": {"truth": 386.0, "filled": 386.2125, "gap": 0.0005503666551991415, "seed_sd": 1.9504975748443485, "tolerance": 0.14001979176307694, "passes": true}, "pre.women.b1940_1945.aime_p50": {"truth": 1041.0, "filled": 1037.5, "gap": -0.003367816510116306, "seed_sd": 2.417044737516849, "tolerance": 0.08949151188066314, "passes": true}, "pre.women.b1940_1945.aime_p75": {"truth": 2058.0, "filled": 2049.5125, "gap": -0.004132677419653064, "seed_sd": 3.882717272871233, "tolerance": 0.06627004271644896, "passes": true}, "pre.women.b1940_1945.aime_p90": {"truth": 3179.7000000000007, "filled": 3176.3300000000004, "gap": -0.0010604104498526112, "seed_sd": 3.293869904438844, "tolerance": 0.06421030653828377, "passes": true}, "pre.women.b1940_1945.plevel": {"truth": 0.19012308454067162, "filled": 0.1855864479487595, "gap": -0.024150875623204504, "seed_sd": 0.0008659010649369567, "tolerance": 0.06996499056922599, "passes": true}, "pre.women.b1940_1945.pr_cross": {"truth": 0.3145537439002571, "filled": 0.34696255632716466, "gap": 0.032408812426907574, "seed_sd": 0.008928812485620915, "tolerance": 0.10514700123361043, "passes": true}, "pre.women.b1940_1945.pr_in": {"truth": 0.19634726483554435, "filled": 0.23404289705626513, "gap": 0.03769563222072078, "seed_sd": 0.007665655829315212, "tolerance": 0.10970200704526734, "passes": true}, "pre.women.b1940_1945.pzero": {"truth": 0.45477910425062734, "filled": 0.46078891339061673, "gap": 0.013128233708663006, "seed_sd": 0.0015635862218243877, "tolerance": 0.04701699812231318, "passes": true}, "pre.women.b1946_1955.paime_p10": {"truth": 156.0, "filled": 155.08499999999998, "gap": -0.005882653542770733, "seed_sd": 0.7727428931716918, "tolerance": 0.1092839592303832, "passes": true}, "pre.women.b1946_1955.paime_p25": {"truth": 516.0, "filled": 515.0875, "gap": -0.0017699763370702115, "seed_sd": 1.2441309585832447, "tolerance": 0.06230795442282464, "passes": true}, "pre.women.b1946_1955.paime_p50": {"truth": 1311.0, "filled": 1311.425, "gap": 0.00032412748026811045, "seed_sd": 1.2276957449243124, "tolerance": 0.0404504662154056, "passes": true}, "pre.women.b1946_1955.paime_p75": {"truth": 2484.0, "filled": 2480.5, "gap": -0.001410011312265702, "seed_sd": 1.4001879573076872, "tolerance": 0.0286716465849327, "passes": true}, "pre.women.b1946_1955.paime_p90": {"truth": 3780.0, "filled": 3776.390000000001, "gap": -0.0009554827833504476, "seed_sd": 1.7873532328411794, "tolerance": 0.03102321311868942, "passes": true}, "pre.women.b1946_1955.ylevel": {"truth": 0.09059452101103667, "filled": 0.09040111213730556, "gap": -0.002137167002129292, "seed_sd": 0.00024301583562582729, "tolerance": 0.029569090149685853, "passes": true}, "pre.women.b1946_1955.yr_cross": {"truth": 0.34158903474392094, "filled": 0.35976623713880124, "gap": 0.0181772023948803, "seed_sd": 0.005037967725535398, "tolerance": 0.03669253843323441, "passes": true}, "pre.women.b1946_1955.yzero": {"truth": 0.5347643649882455, "filled": 0.5381167650757204, "gap": 0.006249361469893522, "seed_sd": 0.000702873211849729, "tolerance": 0.015053483015288157, "passes": true}, "pre.women.b1946_1980.paime_p10": {"truth": 103.0, "filled": 102.9, "gap": -0.0009713453896322832, "seed_sd": 0.30779350562554625, "tolerance": 0.05217708504283977, "passes": true}, "pre.women.b1946_1980.paime_p25": {"truth": 304.0, "filled": 303.35, "gap": -0.002140447017914937, "seed_sd": 0.4893604849295929, "tolerance": 0.031237246580028546, "passes": true}, "pre.women.b1946_1980.paime_p50": {"truth": 752.0, "filled": 752.55, "gap": 0.0007311156485316772, "seed_sd": 0.6048053188292994, "tolerance": 0.021471785135136603, "passes": true}, "pre.women.b1946_1980.paime_p75": {"truth": 1583.75, "filled": 1580.6375, "gap": -0.0019672059782536166, "seed_sd": 0.871609974339682, "tolerance": 0.021247962622687265, "passes": true}, "pre.women.b1946_1980.paime_p90": {"truth": 2690.0, "filled": 2690.230000000002, "gap": 8.549820366088312e-05, "seed_sd": 1.5444637972601931, "tolerance": 0.019286406429246394, "passes": true}, "pre.women.b1946_1980.ylevel": {"truth": 0.06340849534928693, "filled": 0.06327556972848844, "gap": -0.0020985381152560656, "seed_sd": 9.26175686774783e-05, "tolerance": 0.015450091379218515, "passes": true}, "pre.women.b1946_1980.yr_cross": {"truth": 0.43636860215681333, "filled": 0.4451774460795388, "gap": 0.008808843922725462, "seed_sd": 0.0018089186135343722, "tolerance": 0.015684351176846988, "passes": true}, "pre.women.b1946_1980.yzero": {"truth": 0.4719000529035934, "filled": 0.47480921762755485, "gap": 0.0061458654129843415, "seed_sd": 0.00043829093555835673, "tolerance": 0.008556191699814752, "passes": true}, "pre.women.b1956_1965.paime_p10": {"truth": 136.0, "filled": 134.12000000000003, "gap": -0.013919964138024099, "seed_sd": 0.6436736669102601, "tolerance": 0.10309020846799896, "passes": true}, "pre.women.b1956_1965.paime_p25": {"truth": 422.0, "filled": 419.85, "gap": -0.005107809406439401, "seed_sd": 1.0894228312566054, "tolerance": 0.05400647590482275, "passes": true}, "pre.women.b1956_1965.paime_p50": {"truth": 996.0, "filled": 996.6, "gap": 0.0006022282627062836, "seed_sd": 0.7539370349250519, "tolerance": 0.03402127173688933, "passes": true}, "pre.women.b1956_1965.paime_p75": {"truth": 1839.0, "filled": 1838.425, "gap": -0.0003127188207425746, "seed_sd": 1.2594380534692113, "tolerance": 0.026722598015179132, "passes": true}, "pre.women.b1956_1965.paime_p90": {"truth": 2790.5999999999985, "filled": 2790.0499999999997, "gap": -0.0001971096563231356, "seed_sd": 2.580799549710988, "tolerance": 0.026506809111477153, "passes": true}, "pre.women.b1956_1965.ylevel": {"truth": 0.06347343380872424, "filled": 0.06311257468924407, "gap": -0.005701421527938955, "seed_sd": 0.00015504667222653533, "tolerance": 0.026809792541125404, "passes": true}, "pre.women.b1956_1965.yr_cross": {"truth": 0.3538451043271927, "filled": 0.3758782599407146, "gap": 0.022033155613521926, "seed_sd": 0.00382221620303007, "tolerance": 0.03047360671843986, "passes": true}, "pre.women.b1956_1965.yzero": {"truth": 0.47511142887567126, "filled": 0.47814290277925675, "gap": 0.006360283971058256, "seed_sd": 0.0008973301455629317, "tolerance": 0.017102901274216892, "passes": true}, "pre.women.b1966_1980.paime_p10": {"truth": 73.0, "filled": 74.35, "gap": 0.018324231757772758, "seed_sd": 0.4893604849295929, "tolerance": 0.0627268236453548, "passes": true}, "pre.women.b1966_1980.paime_p25": {"truth": 205.0, "filled": 204.3, "gap": -0.003420477314832304, "seed_sd": 0.5712405705774795, "tolerance": 0.039676687855380324, "passes": true}, "pre.women.b1966_1980.paime_p50": {"truth": 458.0, "filled": 456.9, "gap": -0.002404635544959177, "seed_sd": 0.6407232755171874, "tolerance": 0.02838346665574009, "passes": true}, "pre.women.b1966_1980.paime_p75": {"truth": 861.0, "filled": 862.275, "gap": 0.0014797408801827672, "seed_sd": 1.0447235846964749, "tolerance": 0.023922104075604994, "passes": true}, "pre.women.b1966_1980.paime_p90": {"truth": 1382.5999999999985, "filled": 1381.23, "gap": -0.0009913779879404672, "seed_sd": 1.0488590292113351, "tolerance": 0.02489092426663599, "passes": true}, "pre.women.b1966_1980.ylevel": {"truth": 0.04398552907354234, "filled": 0.04407810496564042, "gap": 0.002102477991286822, "seed_sd": 8.952444715162168e-05, "tolerance": 0.017769920520911725, "passes": true}, "pre.women.b1966_1980.yr_cross": {"truth": 0.318767197977468, "filled": 0.3245024764728558, "gap": 0.005735278495387797, "seed_sd": 0.003385430928445152, "tolerance": 0.02158530789673962, "passes": true}, "pre.women.b1966_1980.yzero": {"truth": 0.42453709903219916, "filled": 0.427032564367886, "gap": 0.005860876886955135, "seed_sd": 0.0006091590835569396, "tolerance": 0.014833297517636257, "passes": true}}, "tier": "certified"}, "adopted": "primary"}}} \ No newline at end of file diff --git a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl index 13e1e11d..af53bb85 100644 --- a/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl +++ b/docs/amendments/gate_epuf_fill_dev_scores_after_round_2.jsonl @@ -3,3 +3,16 @@ {"utc": "2026-10-04T07:31:20+00:00", "candidate": "pre_chain2", "family": "pre", "artifact_sha256": "37c2d72e6a7700cf7ac9dfe705434152f71e708ed92adf0fbad78611328b5761", "code_sha256": "ff279cd69d1b94ec1447d9d14db0d893546142cd34b54fa8b23e22b3e76a0584", "seeds": [7100, 7101, 7102, 7103], "n_failing": 55, "n_gating": 136, "tier": "not_adopted", "worst": [["pre.women.b1966_1980.ylevel", 0.45917079627409274, 0.017769920520911725], ["pre.men.b1966_1980.ylevel", 0.4811824306056769, 0.021382204828984532], ["pre.women.b1966_1980.yzero", 0.2583487735491933, 0.014833297517636257], ["pre.women.b1946_1980.yzero", 0.1489263401252473, 0.008556191699814752], ["pre.women.b1946_1980.ylevel", 0.20714311009839204, 0.015450091379218515], ["pre.men.b1946_1980.yzero", 0.15148656243554004, 0.011384493587046594], ["pre.men.b1966_1980.yzero", 0.19620307014155636, 0.01675184764087755], ["pre.men.b1946_1980.ylevel", 0.1430531564860713, 0.01520534060461241], ["pre.women.b1956_1965.yzero", 0.15643826543566397, 0.017102901274216892], ["pre.men.b1956_1965.yzero", 0.16956100424184672, 0.022693736955041323]]} {"utc": "2026-10-04T07:42:02+00:00", "candidate": "odd_qrf_sex2", "family": "odd", "artifact_sha256": "be03f0f3c702e391008fb72eec16d799385be93fbd5de01ce78a79e1fe45330b", "code_sha256": "9b2c5dd94d3d51e5f53994b7aa707ce7955b99eeb1e6e8bc481e753d1260761a", "seeds": [7100, 7101, 7102, 7103], "n_failing": 23, "n_gating": 183, "tier": "not_adopted", "worst": [["odd.women.a22_29.r2", 0.02714856253605591, 0.013111971249341917], ["odd.men.a60_74.zint", 0.33247368988113246, 0.17758159695027936], ["odd.women.a60_74.r2", -0.03422723027556407, 0.022174026993056747], ["odd.men.a22_74.r2", -0.007062508310260118, 0.004576681103770273], ["odd.men.a22_29.r1", -0.010609081809470733, 0.007273490161268873], ["odd.women.a30_44.r4", -0.014271254170137415, 0.010187042430903364], ["odd.men.a30_44.zexit", -0.04086629032766431, 0.031953074094816486], ["odd.men.a60_74.r2", -0.02629022691682592, 0.020757725353477027], ["odd.men.a30_44.r4", -0.011478929773406477, 0.009218001294372115], ["odd.women.a45_59.r2", -0.009149018075044535, 0.0074184353060767995], ["odd.men.a22_74.zexit", -0.022985129792897574, 0.01916127382099423], ["odd.women.a60_74.r1", -0.011531037538756617, 0.009751654703991025], ["odd.men.a30_44.r2", -0.0075100613637079094, 0.006500650570758907], ["odd.women.a22_29.level", 0.020540551604554036, 0.018517427407013797], ["odd.men.a22_74.r4", -0.007762715111083396, 0.007000982741779954], ["odd.women.a60_74.r4", -0.03868148727219123, 0.036314307016843086], ["odd.women.a22_29.q50", 0.0232742152048111, 0.021849929026002995], ["odd.men.a45_59.r2", -0.007692352520425327, 0.0074262368998509335], ["odd.women.a45_59.r4", -0.012890705852779516, 0.012512985825002505], ["odd.women.a22_74.r4", -0.007885437242168614, 0.007666262733486471], ["odd.men.a22_29.r3", -0.01343644400734656, 0.013280307041936976], ["odd.men.a22_29.q50", 0.023239013149086052, 0.022980444394274636], ["odd.women.a30_44.r2", -0.007087772863819786, 0.007078708407960211], ["odd.men.a30_44.q10", -0.04751745643811356, 0.047669064141889934], ["odd.women.a22_74.r3", -0.005150669696892374, 0.005171487109756781], ["odd.men.a30_44.zint", -0.08884255086813031, 0.08989768311292742], ["odd.women.a22_29.r4", 0.01925238657360928, 0.01982762523559322], ["odd.men.a60_74.r1", -0.009265651157384092, 0.009750532797445479], ["odd.men.a22_29.level", 0.020253468381523643, 0.02171146931539837], ["odd.men.a22_74.r3", -0.004148209781093093, 0.004722018684826323]]} {"utc": "2026-10-04T07:44Z (approximate)", "candidate": "pre_chain3 (diagnostic check, not scored against tolerances)", "family": "pre", "description": "the chained alternative refitted with single-year ages 15-24 and unit years 1951-2005; DEV persons born 1966-1980 (men, 30,000) filled with seed 7100 and compared with the truth by age", "printed": {"15": {"truth_mean": 0.0028, "truth_zero": 0.857, "fill_mean": 0.0081, "fill_zero": 0.863}, "16": {"truth_mean": 0.0094, "truth_zero": 0.661, "fill_mean": 0.0189, "fill_zero": 0.677}, "17": {"truth_mean": 0.0222, "truth_zero": 0.488, "fill_mean": 0.0406, "fill_zero": 0.508}, "18": {"truth_mean": 0.0409, "truth_zero": 0.369, "fill_mean": 0.0737, "fill_zero": 0.387}, "19": {"truth_mean": 0.0697, "truth_zero": 0.304, "fill_mean": 0.1113, "fill_zero": 0.331}, "20": {"truth_mean": 0.0934, "truth_zero": 0.287, "fill_mean": 0.1285, "fill_zero": 0.311}, "21": {"truth_mean": 0.1116, "truth_zero": 0.276, "fill_mean": 0.134, "fill_zero": 0.294}}, "earlier_check": "the same comparison with pre_chain2 (five-year age bands) printed fill means 0.0236-0.1416 against truth 0.0028-0.1116 at ages 15-21", "code_sha256": "9b2c5dd94d3d51e5f53994b7aa707ce7955b99eeb1e6e8bc481e753d1260761a"} +{"utc": "2026-10-04T07:54:03+00:00", "candidate": "odd_qrf3", "family": "odd", "artifact_sha256": "28d52765c4f540909e9d5c1deb13724e7e222d6feac27c9aed797cb0516f0993", "code_sha256": "36e045bdc94681f5cc54fed30b0fd447fc4f87bacd4239c84ca4c1b6c03a027b", "seeds": [7100, 7101, 7102, 7103], "n_failing": 12, "n_gating": 183, "tier": "improves", "worst": [["odd.women.a22_29.r2", 0.027344747016985638, 0.013111971249341917], ["odd.men.a60_74.zint", 0.3415042291139394, 0.17758159695027936], ["odd.men.a30_44.zexit", -0.0511542734002437, 0.031953074094816486], ["odd.men.a22_29.r1", -0.011611591760373075, 0.007273490161268873], ["odd.men.a22_74.zexit", -0.027937271991644974, 0.01916127382099423], ["odd.women.a30_44.r4", -0.013133885403330492, 0.010187042430903364], ["odd.women.a60_74.r1", -0.011707903966676314, 0.009751654703991025], ["odd.women.a22_74.r3", -0.005771964915830763, 0.005171487109756781], ["odd.men.a22_29.q50", 0.024899453547185146, 0.022980444394274636], ["odd.men.a22_29.level", 0.02192345936691109, 0.02171146931539837], ["odd.women.a22_29.level", 0.018596125147572584, 0.018517427407013797], ["odd.men.a22_29.r3", -0.01329969151609589, 0.013280307041936976], ["odd.men.a45_59.zexit", -0.0351852359581335, 0.036314919811016276], ["odd.men.a30_44.zint", -0.08623838272974282, 0.08989768311292742], ["odd.men.a30_44.q10", -0.04558952377760228, 0.047669064141889934], ["odd.women.a22_29.q50", 0.020520979391902117, 0.021849929026002995], ["odd.women.a30_44.zint", -0.06508790701647582, 0.07029190354044747], ["odd.men.a22_74.r3", -0.004260068220721669, 0.004722018684826323], ["odd.men.a22_29.q10", 0.06445677303612518, 0.07329266358707544], ["odd.men.a60_74.r1", -0.008430272781665082, 0.009750532797445479], ["odd.men.a22_74.r1", -0.002051107586787837, 0.00258049589094515], ["odd.men.a22_29.q90", 0.01862070347052136, 0.02353059621893112], ["odd.women.a22_29.r4", 0.015341851710927168, 0.01982762523559322], ["odd.women.a22_74.r4", -0.0058897478528845415, 0.007666262733486471], ["odd.women.a22_74.zexit", -0.011878230212011065, 0.016039678883419197]]} +{"utc": "2026-10-04T07:59:25+00:00", "candidate": "odd_qrf4", "family": "odd", "artifact_sha256": "9b92fd98cd4021171c838bf4a8852ec25c2a396049eecbc1bca4ef9e8fcdbd2a", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101, 7102, 7103], "n_failing": 8, "n_gating": 183, "tier": "improves", "worst": [["odd.men.a22_29.r1", -0.01998280153218135, 0.007273490161268873], ["odd.women.a22_29.r1", -0.012394194276425408, 0.006445547562698701], ["odd.women.a22_29.zint", 0.14907864826487227, 0.09342121518593281], ["odd.men.a22_29.zint", 0.16973294481418666, 0.12140160693630424], ["odd.women.a22_74.zint", 0.055533732527217605, 0.04476234327525859], ["odd.women.a60_74.r1", -0.010935700430824924, 0.009751654703991025], ["odd.women.a22_74.r1", -0.0025564811431657564, 0.002475922564645967], ["odd.men.a22_74.r1", -0.002637012673456507, 0.00258049589094515], ["odd.men.a22_29.r3", -0.012301908709043796, 0.013280307041936976], ["odd.women.a22_74.r3", -0.004555637079359798, 0.005171487109756781], ["odd.men.a22_74.r3", -0.003944498885934511, 0.004722018684826323], ["odd.men.a60_74.r1", -0.008071695817202462, 0.009750532797445479], ["odd.women.a22_29.wint", 0.12118226962992029, 0.15037337545812676], ["odd.men.a22_74.zint", 0.043294991137900585, 0.05481211657295162], ["odd.women.a22_29.r2", 0.009530501163799499, 0.013111971249341917], ["odd.men.a30_44.r4", -0.006666088981186702, 0.009218001294372115], ["odd.women.a30_44.r4", -0.007245568446286876, 0.010187042430903364], ["odd.men.a22_74.r4", -0.0049113140287062595, 0.007000982741779954], ["odd.women.a22_29.r3", -0.00828787522771579, 0.012054115899361516], ["odd.men.a30_44.r3", -0.004138657644559562, 0.006087370619188444], ["odd.men.a60_74.zint", 0.11754876132006542, 0.17758159695027936], ["odd.women.a22_74.r2", 0.003035764997925239, 0.004772836686607259]]} +{"utc": "2026-10-04T08:01:26+00:00", "candidate": "pre_donor_k25", "family": "pre", "artifact_sha256": "eafa0ae5611817c69b800b42fa0798e71bc3fe34b2a946359eb89042b8879443", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 6, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1946_1980.yr_cross", 0.029133392188555096, 0.015684351176846988], ["pre.women.b1946_1980.yzero", 0.015405372793597327, 0.008556191699814752], ["pre.women.b1956_1965.yr_cross", 0.05360057865508633, 0.03047360671843986], ["pre.women.b1946_1955.yr_cross", 0.04836863108053191, 0.03669253843323441], ["pre.women.b1966_1980.yzero", 0.0182391249424394, 0.014833297517636257], ["pre.men.b1946_1980.yzero", 0.013244286631349356, 0.011384493587046594], ["pre.women.b1946_1955.yzero", 0.015014321048421264, 0.015053483015288157], ["pre.women.b1966_1980.yr_cross", 0.021044902116193698, 0.02158530789673962]]} +{"utc": "2026-10-04T08:02:45+00:00", "candidate": "pre_donor_k50", "family": "pre", "artifact_sha256": "d53cded2b59ec6452f52f1ca4b30a2dc97f1aa1ec9a989e03b32f201edf3e910", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 13, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1946_1980.yzero", 0.020559950615929745, 0.008556191699814752], ["pre.women.b1946_1980.yr_cross", 0.034508472442058846, 0.015684351176846988], ["pre.women.b1956_1965.yr_cross", 0.05776594239087035, 0.03047360671843986], ["pre.men.b1946_1980.yzero", 0.018480756025274547, 0.011384493587046594], ["pre.women.b1946_1955.yzero", 0.023269971969184455, 0.015053483015288157], ["pre.women.b1966_1980.yzero", 0.0200676390110347, 0.014833297517636257], ["pre.women.b1966_1980.yr_cross", 0.029076933330211885, 0.02158530789673962], ["pre.men.b1946_1980.yr_cross", 0.018382403592013652, 0.014632893176957621]]} +{"utc": "2026-10-04T08:04:25+00:00", "candidate": "pre_donor_k5", "family": "pre", "artifact_sha256": "6c739ba6f7437a1fc104a268ceb1f90d184fc1f60f3be811d4d35217e245d21e", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 4, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1956_1965.yr_cross", 0.04463879596543152, 0.03047360671843986], ["pre.women.b1946_1980.yr_cross", 0.02293633800904532, 0.015684351176846988], ["pre.women.b1946_1980.yzero", 0.011766480578584093, 0.008556191699814752], ["pre.women.b1946_1955.yr_cross", 0.03914897033531145, 0.03669253843323441], ["pre.men.b1946_1980.yzero", 0.010066340298164667, 0.011384493587046594], ["pre.women.b1966_1980.yzero", 0.012846678222160568, 0.014833297517636257], ["pre.women.b1966_1980.yr_cross", 0.01800530689095292, 0.02158530789673962], ["pre.men.b1946_1980.yr_cross", 0.012084512584237206, 0.014632893176957621]]} +{"utc": "2026-10-04T08:05:54+00:00", "candidate": "pre_donor_k3", "family": "pre", "artifact_sha256": "9902743dfd0f0b17d7ec063421d3a25da2a9de9987942b7508cbc0c4694eedfc", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 3, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1946_1980.yzero", 0.011163616040424484, 0.008556191699814752], ["pre.women.b1946_1980.yr_cross", 0.01978006406808891, 0.015684351176846988], ["pre.women.b1956_1965.yr_cross", 0.03759069775273621, 0.03047360671843986], ["pre.men.b1946_1980.yzero", 0.009549309004077244, 0.011384493587046594], ["pre.women.b1966_1980.yr_cross", 0.017873523200933716, 0.02158530789673962], ["pre.women.b1966_1980.yzero", 0.01194888116310322, 0.014833297517636257]]} +{"utc": "2026-10-04T08:07:00+00:00", "candidate": "pre_donor_k1", "family": "pre", "artifact_sha256": "bc7a7d28a503cfdfdba022ef71387d1420f7ad9da1471c2ba2dba58b775f5c6b", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 4, "n_gating": 136, "tier": "improves", "worst": [["pre.women.b1946_1980.yzero", 0.00991715298416973, 0.008556191699814752], ["pre.women.b1956_1965.yr_cross", 0.03522344626992119, 0.03047360671843986], ["pre.women.b1946_1980.yr_cross", 0.016289968599504656, 0.015684351176846988], ["pre.women.b1966_1980.yzero", 0.015207687342106424, 0.014833297517636257], ["pre.men.b1946_1980.yzero", 0.011275693354747762, 0.011384493587046594], ["pre.men.b1966_1980.yzero", 0.013394282698992344, 0.01675184764087755]]} +{"utc": "2026-10-04T08:10:45+00:00", "candidate": "pre_donor_b5k3", "family": "pre", "artifact_sha256": "47304b01a2c8a3ebdc8d9799efc7f86c0e501c925ba232191a420412980ea950", "code_sha256": "565e3ba2717b7e69cd1e9765a05ef3c8c753d3fbc18c97bc7c9fc079ea76f02b", "seeds": [7100, 7101], "n_failing": 0, "n_gating": 136, "tier": "certified", "worst": [["pre.women.b1956_1965.yr_cross", 0.03042477794600501, 0.03047360671843986], ["pre.men.b1946_1980.yzero", 0.010177331108892296, 0.011384493587046594], ["pre.women.b1946_1980.yzero", 0.007455221837990744, 0.008556191699814752], ["pre.men.b1946_1980.ylevel", -0.011326381911699102, 0.01520534060461241], ["pre.women.b1946_1980.yr_cross", 0.011465522706747888, 0.015684351176846988], ["pre.men.b1946_1980.yr_cross", 0.010614564663255555, 0.014632893176957621]]} +{"utc": "2026-10-04T08:23:28+00:00", "candidate": "pre_donor_ball_k3", "family": "pre", "artifact_sha256": "c472c84325711c8856dc8067d02ef3c9c27cb8bce63befae22df2d18cab50062", "code_sha256": "4be7556f2263ef7890929ef27a92975d93938415e22153b38bf0afd01fef5983", "seeds": [7100, 7101, 7102, 7103], "n_failing": 0, "n_gating": 136, "tier": "certified", "worst": [["pre.men.b1946_1980.yzero", 0.00809038443776322, 0.011384493587046594], ["pre.women.b1946_1980.yzero", 0.005732645830089589, 0.008556191699814752], ["pre.men.b1956_1965.yzero", 0.015012948537747373, 0.022693736955041323], ["pre.women.b1956_1965.yr_cross", 0.019137592768893985, 0.03047360671843986], ["pre.women.b1946_1980.yr_cross", 0.00910100469507158, 0.015684351176846988], ["pre.men.b1946_1980.ylevel", -0.008408704694680136, 0.01520534060461241], ["pre.women.b1930_1945.pr_cross", 0.036302636511399144, 0.0677006808491353], ["pre.women.b1930_1945.pr_in", 0.03831971636456638, 0.07412154624977427]]} +{"utc": "2026-10-04T08:27:20+00:00", "candidate": "odd_knn2", "family": "odd", "artifact_sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 47, "n_gating": 183, "tier": "not_adopted", "worst": [["odd.men.a22_74.wint", -0.7579019412918284, 0.07337302886151653], ["odd.women.a22_74.wint", -0.6178668454511387, 0.07133033012317215], ["odd.men.a45_59.wint", -1.0410721850282765, 0.13344110142481172], ["odd.women.a45_59.wint", -0.9735852305534904, 0.1271041647107106], ["odd.men.a30_44.wint", -0.7720545590267562, 0.12001713872675526], ["odd.women.a30_44.wint", -0.5937411523100029, 0.10093703169083498]]} +{"utc": "2026-10-04T08:27:45+00:00", "candidate": "pre_chain3", "family": "pre", "artifact_sha256": "74c62142e1de4bfed543de3b1d8c88c0d70c674a8a2fa44129bfe5943d4ff0ea", "code_sha256": "c5e4a37417ec9c7dae765dcaed464e44a32711d4c0806ae17fdb24511c0b279d", "seeds": [7100, 7101, 7102, 7103], "n_failing": 50, "n_gating": 136, "tier": "not_adopted", "worst": [["pre.women.b1966_1980.ylevel", 0.35125489825189504, 0.017769920520911725], ["pre.men.b1966_1980.ylevel", 0.35211188653405623, 0.021382204828984532], ["pre.women.b1966_1980.yzero", 0.13965247274046733, 0.014833297517636257], ["pre.women.b1946_1980.ylevel", 0.1097696357515745, 0.015450091379218515], ["pre.men.b1930_1945.plevel", -0.18375090914832892, 0.029173903026433003], ["pre.women.b1930_1945.pzero", -0.18200182661313102, 0.02913422225552515]]} +{"utc": "2026-10-04T08:52:00+00:00", "candidate": "registered set (manifest runs/epuf_fill_candidates_v1.json)", "family": "odd+pre", "description": "DEV dry run of the registered TEST procedure: epuf_fill_scoring.score_registered with the DEV matrix, 20 draw seeds, candidates loaded by SHA-256, both readings of the current odd rule", "manifest_sha256": "8d421536351d884184af441889333db2361fb0f50326e3a4481e6fa4f07aeecf", "summary": {"odd": {"adopted": "primary", "dropped": [], "primary": {"n_failing": 7, "n_gating": 183, "tier": "improves", "worst": [[2.690259040347596, "odd.men.a22_29.r1"], [2.0188009070760624, "odd.women.a22_29.r1"], [1.5849202031321585, "odd.women.a22_29.zint"], [1.3654563419043049, "odd.men.a22_29.zint"], [1.2322030452483468, "odd.women.a22_74.zint"]]}, "alternative": {"n_failing": 46, "n_gating": 183, "tier": "not_adopted", "worst": [[10.235181548470665, "odd.men.a22_74.wint"], [8.632139516178636, "odd.women.a22_74.wint"], [7.7476496021426895, "odd.women.a45_59.wint"], [7.571934298028397, "odd.men.a45_59.wint"], [6.518179638279406, "odd.men.a30_44.wint"]]}, "current": {"fallback": {"n_failing": 100}, "two_sided": {"n_failing": 99}}}, "pre": {"adopted": "primary", "dropped": [], "primary": {"n_failing": 0, "n_gating": 136, "tier": "certified", "worst": [[0.769263808372738, "pre.men.b1946_1980.yzero"], [0.7230242162372418, "pre.women.b1956_1965.yr_cross"], [0.7182944969684824, "pre.women.b1946_1980.yzero"], [0.6574070292194412, "pre.men.b1956_1965.yzero"], [0.6068639745544633, "pre.men.b1946_1980.ylevel"]]}, "alternative": {"n_failing": 50, "n_gating": 136, "tier": "not_adopted", "worst": [[19.56156909523513, "pre.women.b1966_1980.ylevel"], [16.437839846619262, "pre.men.b1966_1980.ylevel"], [9.486645296938711, "pre.women.b1966_1980.yzero"], [6.976223346512972, "pre.women.b1946_1980.ylevel"], [6.268232573425137, "pre.men.b1930_1945.plevel"]]}, "current": {"fallback": {"n_failing": 131}}}}} +{"utc": "2026-10-04T19:10:00+00:00", "candidate": "none: K sweep of the registered dry run", "family": "odd+pre", "description": "DEV-only, post hoc sweep of K (tolerance multiplier) over the registered TEST-procedure dry run on DEV, re-scored with the repository's own score/adoption_tier/adopt/combined_current; shown to Max before he ruled on d927 (ratify K = 1). K = 1 was registered at 14045be4, before any DEV score. Script, input and output: docs/amendments/gate_epuf_fill_dev_k_sweep.py, gate_epuf_fill_dev_registered_dryrun.json, gate_epuf_fill_dev_k_sweep.txt.", "dryrun_sha256": "086378aa9c6cd4277edfcd88fbff4f1b3907d6619d82f554d64520db88ef49a6", "summary": {"k1_reproduces_record": true, "odd": {"primary_improves_from_K": 0.8968, "primary_certified_from_K": 2.6903, "alternative_improves_from_K": 3.4117, "adopted_primary_for_K": [0.8968, 4.0]}, "pre": {"primary_improves_from_K": 0.2564, "primary_certified_from_K": 0.7693, "adopted_primary_for_K": [0.2564, 4.0]}, "failing_at_K": {"odd_primary": {"1.0": 7, "1.5": 3, "2.0": 2, "3.0": 0}, "pre_primary": {"1.0": 0}}, "thresholds": "exact: each is a cell's |gap|/sigma or |gap|/(3 sigma); the script's 0.005 grid finds no other transition"}} diff --git a/reviews/gate_epuf_fill_pr516_code_review_20261004.md b/reviews/gate_epuf_fill_pr516_code_review_20261004.md new file mode 100644 index 00000000..48e0af5c --- /dev/null +++ b/reviews/gate_epuf_fill_pr516_code_review_20261004.md @@ -0,0 +1,90 @@ +**Verdict: REQUEST CHANGES** + +The registration itself holds up. The candidates are fitted on TRAIN only, nothing reads a masked cell, the files are pinned by SHA-256, and the DEV numbers in the document match the log. But the opt-in PSID path has a real correctness bug (finding 1), and two robustness hazards (findings 2 and 3) should be fixed before lock. + +Most fixes touch `epuf_fill.py`. That file is pinned by `test_the_fill_code_is_unchanged_since_the_fit`, so changing it means re-fitting at a new commit and updating `code_commit`. None of the fixes below changes the bytes `to_bytes` writes (finding 4 is the one exception, flagged there). + +**I could not run git or pytest in this session**: only read and search tools were available. So I have not checked `git diff 70553e70 940ad6f1`, whether `career.py`, `cohorts/psid2010.py` and the registered runs are untouched, the test results, or the manifest's SHA-256. Everything below comes from reading the files at the PR head. + +## Findings + +**1. Medium — learned odd fills break or diverge on real PSID gap years** (`psid2010_epuf_fill.py:120-128`, `epuf_fill.py:983-1003`, `epuf_fill.py:652-698`) +- **What goes wrong:** `fill_careers` builds the shares the fills see from `observed` rows only. But `career._impute_gap` (`career.py:964-978`) also fills a gap from other neighbours: a pre-career observed year, or `boundary_2014` for the 2013 seam (`career.py:1059-1062`). So a `gap_imputed` year can have no neighbour the fill can see. Two examples: + - the 2013 seam when 2012 is unknown; + - a person whose career starts in 1997, whose 1996 is pre-career, and whose 1998 is missing. +- **What each fill does with such a year:** + - `OddKnnFill` leaves it NaN, because `take` requires a finite `left`. `drawn()` then raises "a learned fill returned an invalid share", so the registered alternative crashes on real data. + - `OddForestFill` silently writes a zero where the assembler wrote the neighbour's value. +- **Out-of-range years:** the module docstring (`:11-12`) says *every* `gap_imputed` year gets a draw. That includes 2007-2013, which lie outside the fills' training years (1991-2005) and outside anything the gate scores (`epuf_fill_gate.py:141-143`). +- **Tests miss it:** neither test's data has the seam, `boundary_2014`, or a gap with no visible neighbour. +- **Fix (in `psid2010_epuf_fill.py`, which is not pinned):** + - leave gap rows with no visible neighbour as `gap_imputed`, or route them through an explicit fallback; + - then either limit the learned odd fill to the scored years (≤2005) or state the extrapolation in the docstring and the registration; + - add a seam case to the test. + +**2. Medium — the nearest-donor cache is keyed by `id(self)`, not by content** (`epuf_fill.py:1212-1220`) +- **What goes wrong:** the cache key includes `str((id(self), self.k))`. CPython reuses addresses, so if one `PreDonorFill` is garbage-collected and another with a different bank but the same `k` lands at the same address, the same inputs hit the cache. That returns the old bank's donor indices for the new bank: wrong blocks, or an index out of range. +- **Registered path:** not affected, because the fills are loaded once and stay alive. +- **Fix:** key on a digest of `bank_match`, `bank_sex`, `bank_birth_year` and `k`, computed once. Add a test with two fills fitted on different banks. + +**3. Medium-low — the fit script overwrites the registered files before it refuses** (`scripts/fit_epuf_fills.py:178-179, 207`) +- **What goes wrong:** `path.write_bytes(blob)` replaces `~/…/fills/{name}_v1.npz` unconditionally. `write_new` only refuses an existing manifest at the very end. The manifest's note invites a refit to reproduce the bytes; doing that at a different commit or in a different environment would destroy the staged registered files, and only then fail. +- **Fix:** check that the manifest does not exist before fitting. Write each file with `open(path, "xb")`, or write to a temporary path and compare against the existing file. + +**4. Low — tree thresholds are stored as float32, so traversal can differ from scikit-learn's** (`epuf_fill.py:531, 544, 822, 840`) +- **What goes wrong:** scikit-learn's threshold is a float64 midpoint of two float32 values, compared as `float64(x) <= thr`. Casting that midpoint to float32 rounds half-to-even when the two values are one ulp apart, which can give `Xf[p]`. Then `x == Xf[p]` goes left here and right in scikit-learn. +- **Effect:** the stored model is self-consistent, since the fit and the draws use the same traversal. But leaf membership can differ from `estimator.apply`, and an empty leaf makes `_value` (`:636`) read `leaf_values[start-1]`, the neighbouring leaf's value. +- **Tests:** no test compares the traversal with `apply`. +- **Fix:** add a test asserting every leaf has a count of at least 1 and that `_tree_leaves` matches `apply`. Keeping the thresholds in float64 would change the artifacts' bytes, so do that only if you re-register. + +**5. Low — persons of uncoded sex get no copula** (`epuf_fill.py:678`, `:1601-1604`) +- **What goes wrong:** `BySexFill` sends sex 3 to the men's part. That part looks up `rho[3, band]`, which is 0.0 because calibration only sets the row of its own sex (manifest: the men's part has `rho` rows 0, 2 and 3 all zero). So these persons' masked years are drawn independently. +- **Fix:** in `BySexFill.fill`, pass `sex = value` for the rows it routes there. + +**6. Low — "byte-reproducible" depends on the environment** (`epuf_fill.py:184-206`, `fit_epuf_fills.py:197-201`) +- **What goes wrong:** the deflate output depends on the zlib build (for example zlib against zlib-ng), and `ZipInfo.create_system` depends on the platform. The manifest records the numpy, scipy and scikit-learn versions, but not the Python version, the zlib runtime version or the platform. +- **Fix:** record `zlib.ZLIB_RUNTIME_VERSION`, the Python version and the platform, or use `ZIP_STORED`. Qualify the reproducibility claim in the registration document. + +**7. Low — provenance gaps in what the scripts record** (`scripts/score_epuf_fill_test.py:59-81`, `fit_epuf_fills.py:43-46`) +- **Score script:** + - It refuses before lock (through `test_part`, `epuf_fill_gate.py:1450-1453`) and refuses an existing output — both fine. + - But it accepts any `--manifest` and only records that manifest's SHA. Pin the registered `8d42153…` and refuse others. + - It does not record whether the working tree was clean. Record that, as the fit script does. + - It writes an absolute local path (including the home directory) into a JSON file that will be published. + - If the run crashes after the TEST read, nothing is written. Consider writing a "started" record first. +- **Fit script:** `CODE_FILES` leaves out `harness/epuf_fill_gate.py` and `harness/epuf_operator.py`, which the fit depends on (TRAIN split, `YEARS`, `wage_base`). So the "code unchanged" test does not cover them. + +**8. Nits** +- `PreDonorFill`'s class docstring is stale (`epuf_fill.py:1124-1136`). It says 2,000 donors, `k` of 10, five match years and "scaled by five". The code matches on `MATCH_DIMS = 7`, and the registered fill uses a bank of 100,000 and `k` of 3. +- `start_year` in `fill_careers` is really the *last* year (`psid2010_epuf_fill.py:87, 99`). If it is set below 2013, later gap rows silently keep the assembler's value. +- `bank_block` is cast to `uint16` without clipping (`:1200`). A share above 1 would wrap. EPUF is capped so it is safe today, but the forest clips and this should too. +- `ok` in `PreChainFill.fill` is always True (`:1497`). +- For PSID recipients, `match_vector`'s career summaries (`:1080`) cover years past 2006, while the bank's stop at 2006. Consider limiting them to ≤2006. +- `test_the_forest_copula_is_calibrated_by_band` only checks shape and range. Assert that the coded-sex row of each part is nonzero. +- The `n_units` registered as 3,000,000 applies per sex: the diagnostics show 3M for each part. Say so in the registration document. + +## What I verified (by reading) + +- **Leakage:** + - `fit_epuf_fills` reads only `epuf_matrix(TRAIN)`. + - The copula's calibration persons are TRAIN persons held out of the forests. + - Every fill reads only cells it is allowed to read: `known = isfinite & ~mask`, or the masked cells set to NaN. + - The scoring path hides the union mask (`epuf_fill_gate.py:1151-1153`). + - The forest's training contexts exclude pre-career years. +- **`hash_uniform`:** keyed by stream, seed, person and year; values strictly inside (0, 1); broadcasting is correct. +- **The draws:** + - The forest: zero below `p0`, otherwise the rescaled quantile; one `eta` per person; one `epsilon` per unit; a separate uniform picks the tree. + - kNN and the chain: draws do not depend on the persons' order. + - The donor draw: each group's donors are stored contiguously, so the fallback random donor is correct. +- **Loading:** `load_fill` checks the SHA-256 before parsing, and nested `BySexFill` files round-trip. +- **`fill_careers`:** + - observed rows are untouched; + - gap rows are relabelled `gap_epuf_drawn`; + - pre-career rows are only added (`pre_mask & ~in_career`) as `pre_career_epuf_donor`; + - shares are converted to dollars at each year's base; + - wage bases past 2006 come from the captured step function (`ss/params.py:151-160`). +- **Birth-evidence reducer:** both new modules are excluded in `scripts/first_estimates_birth_evidence.py:358-361`, with the reachability assertions in `tests/estimates/test_birth_evidence_artifact.py:217-220, 476-482`. +- **Claims:** + - The registration document's SHAs, sizes, library versions and `code_commit` match `runs/epuf_fill_candidates_v1.json`. + - The manifest SHA `8d42153…` matches the one in the DEV dry-run log (line 17). + - The dry run's tier counts match the document's tables: odd primary 7 of 183 failing ("improves"), odd alternative 46, current odd rule 100 and 99; pre primary 0 of 136 ("certified"), pre alternative 50, current pre rule 131. \ No newline at end of file diff --git a/runs/epuf_fill_candidates_v1.json b/runs/epuf_fill_candidates_v1.json new file mode 100644 index 00000000..1504efd3 --- /dev/null +++ b/runs/epuf_fill_candidates_v1.json @@ -0,0 +1,1404 @@ +{ + "schema": "populace_dynamics.epuf_fill_candidates.v1", + "registration_id": "2026-10-03-epuf-career-fill", + "code_commit": "b722382e2427cf84253daab96c206f875a8c0544", + "code_files_clean": true, + "built_at_utc": "2026-10-04T09:21:00+00:00", + "part": "train", + "n_persons": 2629944, + "versions": { + "numpy": "2.5.1", + "scipy": "1.18.0", + "scikit_learn": "1.9.0", + "python": "3.14.4", + "zlib_runtime": "1.2.12", + "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O" + }, + "staging": "files live outside the repository, like EPUF; a refit with this script at code_commit, on the same library, zlib and platform versions, reproduces their bytes", + "fills": { + "odd_forest": { + "family": "odd", + "role": "primary", + "params": { + "unit_years": [ + 1991, + 2005 + ], + "n_units": 3000000, + "n_trees": 10, + "min_leaf": 15, + "max_features": 0.8, + "seed": 0 + }, + "file": "odd_forest_v1.npz", + "sha256": "37a9ea76c9cac3692efb4e6b29b184a1a480f4b3caa8b15462133f5af659ebfa", + "bytes": 44836853, + "fit_seconds": 71.5, + "diagnostics": { + "1": { + "n_units": 3000000, + "n_positive_units": 1267291, + "n_trees": 10, + "min_leaf": 15, + "n_leaves": 250386, + "n_nodes": 500762, + "calibration_persons": 136391, + "rho": [ + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.1, + 0.1, + 0.15, + 0.15, + 0.45, + 0.45 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + ], + "calibration": { + "1.1.rho_0.0": [ + -0.00937054964764883, + -0.006568577649590401 + ], + "1.2.rho_0.0": [ + -0.009185985434689847, + -0.017087922487923013 + ], + "1.3.rho_0.0": [ + -0.008760608986963514, + -0.005892891185897087 + ], + "1.4.rho_0.0": [ + -0.014822290100570679, + -0.0029128598220162782 + ], + "2.1.rho_0.0": [ + "nan", + "nan" + ], + "2.2.rho_0.0": [ + "nan", + "nan" + ], + "2.3.rho_0.0": [ + "nan", + "nan" + ], + "2.4.rho_0.0": [ + "nan", + "nan" + ], + "1.1.rho_0.05": [ + -0.0025221662657621824, + -0.0025405849711389594 + ], + "1.2.rho_0.05": [ + -0.005747487224877057, + -0.014480367835756236 + ], + "1.3.rho_0.05": [ + -0.006653418258712018, + -0.003659386315042923 + ], + "1.4.rho_0.05": [ + -0.013918461926527237, + -0.0014846818917918503 + ], + "2.1.rho_0.05": [ + "nan", + "nan" + ], + "2.2.rho_0.05": [ + "nan", + "nan" + ], + "2.3.rho_0.05": [ + "nan", + "nan" + ], + "2.4.rho_0.05": [ + "nan", + "nan" + ], + "1.1.rho_0.1": [ + 0.0022622655578613537, + 0.0017571621431731188 + ], + "1.2.rho_0.1": [ + -0.003463604263099551, + -0.013817366044770019 + ], + "1.3.rho_0.1": [ + -0.0043256050642532795, + -0.0020772149123101658 + ], + "1.4.rho_0.1": [ + -0.011090372488350764, + 0.0003620037980666124 + ], + "2.1.rho_0.1": [ + "nan", + "nan" + ], + "2.2.rho_0.1": [ + "nan", + "nan" + ], + "2.3.rho_0.1": [ + "nan", + "nan" + ], + "2.4.rho_0.1": [ + "nan", + "nan" + ], + "1.1.rho_0.15": [ + 0.00735065280082603, + 0.00547377504308344 + ], + "1.2.rho_0.15": [ + -0.00043591174742074745, + -0.011407436015852035 + ], + "1.3.rho_0.15": [ + -0.001469747919156772, + -6.807326989843876e-05 + ], + "1.4.rho_0.15": [ + -0.007842751587615049, + 0.0035570177498509548 + ], + "2.1.rho_0.15": [ + "nan", + "nan" + ], + "2.2.rho_0.15": [ + "nan", + "nan" + ], + "2.3.rho_0.15": [ + "nan", + "nan" + ], + "2.4.rho_0.15": [ + "nan", + "nan" + ], + "1.1.rho_0.2": [ + 0.01235436992080352, + 0.007278338434604126 + ], + "1.2.rho_0.2": [ + 0.0018633014396086667, + -0.00951255602154033 + ], + "1.3.rho_0.2": [ + 0.0014639774440903253, + 0.0017977312065096118 + ], + "1.4.rho_0.2": [ + -0.007672387545435644, + 0.005532053798539716 + ], + "2.1.rho_0.2": [ + "nan", + "nan" + ], + "2.2.rho_0.2": [ + "nan", + "nan" + ], + "2.3.rho_0.2": [ + "nan", + "nan" + ], + "2.4.rho_0.2": [ + "nan", + "nan" + ], + "1.1.rho_0.25": [ + 0.017532837584111616, + 0.012910343871726515 + ], + "1.2.rho_0.25": [ + 0.005432755081799634, + -0.0058967638868855365 + ], + "1.3.rho_0.25": [ + 0.004461578182951564, + 0.003245424028756938 + ], + "1.4.rho_0.25": [ + -0.006896856131266005, + 0.007402313522806625 + ], + "2.1.rho_0.25": [ + "nan", + "nan" + ], + "2.2.rho_0.25": [ + "nan", + "nan" + ], + "2.3.rho_0.25": [ + "nan", + "nan" + ], + "2.4.rho_0.25": [ + "nan", + "nan" + ], + "1.1.rho_0.3": [ + 0.02283322303498425, + 0.01782293559705972 + ], + "1.2.rho_0.3": [ + 0.007917659257822063, + -0.0042736292458098735 + ], + "1.3.rho_0.3": [ + 0.005912180793823385, + 0.004708972924462929 + ], + "1.4.rho_0.3": [ + -0.0037362954092261536, + 0.012169812222417309 + ], + "2.1.rho_0.3": [ + "nan", + "nan" + ], + "2.2.rho_0.3": [ + "nan", + "nan" + ], + "2.3.rho_0.3": [ + "nan", + "nan" + ], + "2.4.rho_0.3": [ + "nan", + "nan" + ], + "1.1.rho_0.35": [ + 0.02936744538086289, + 0.02269930584128066 + ], + "1.2.rho_0.35": [ + 0.010991221409968999, + -0.0020151785137914047 + ], + "1.3.rho_0.35": [ + 0.008154499506314195, + 0.006864335349464179 + ], + "1.4.rho_0.35": [ + -0.0033667339378640193, + 0.010716886425056416 + ], + "2.1.rho_0.35": [ + "nan", + "nan" + ], + "2.2.rho_0.35": [ + "nan", + "nan" + ], + "2.3.rho_0.35": [ + "nan", + "nan" + ], + "2.4.rho_0.35": [ + "nan", + "nan" + ], + "1.1.rho_0.4": [ + 0.033991025478408377, + 0.025266554928974005 + ], + "1.2.rho_0.4": [ + 0.01395682475791915, + 0.00013629350615307345 + ], + "1.3.rho_0.4": [ + 0.009812482470721196, + 0.008655649864332982 + ], + "1.4.rho_0.4": [ + -0.0018532086431369832, + 0.010892834474684587 + ], + "2.1.rho_0.4": [ + "nan", + "nan" + ], + "2.2.rho_0.4": [ + "nan", + "nan" + ], + "2.3.rho_0.4": [ + "nan", + "nan" + ], + "2.4.rho_0.4": [ + "nan", + "nan" + ], + "1.1.rho_0.45": [ + 0.04078840124406191, + 0.03125399303118248 + ], + "1.2.rho_0.45": [ + 0.016853239128019615, + 0.0023949314237127206 + ], + "1.3.rho_0.45": [ + 0.012218818061125014, + 0.0103161648513439 + ], + "1.4.rho_0.45": [ + 0.0007757976805343736, + 0.012061337583499476 + ], + "2.1.rho_0.45": [ + "nan", + "nan" + ], + "2.2.rho_0.45": [ + "nan", + "nan" + ], + "2.3.rho_0.45": [ + "nan", + "nan" + ], + "2.4.rho_0.45": [ + "nan", + "nan" + ], + "1.1.rho_0.5": [ + 0.04627958132301435, + 0.035137264874086305 + ], + "1.2.rho_0.5": [ + 0.020285078982263505, + 0.004963715627156029 + ], + "1.3.rho_0.5": [ + 0.014599779251665335, + 0.012138390969464785 + ], + "1.4.rho_0.5": [ + 0.004685831714231314, + 0.015445592554809928 + ], + "2.1.rho_0.5": [ + "nan", + "nan" + ], + "2.2.rho_0.5": [ + "nan", + "nan" + ], + "2.3.rho_0.5": [ + "nan", + "nan" + ], + "2.4.rho_0.5": [ + "nan", + "nan" + ], + "1.1.rho_0.55": [ + 0.0531305404913871, + 0.03958250535156038 + ], + "1.2.rho_0.55": [ + 0.023484969938003974, + 0.007224393712798816 + ], + "1.3.rho_0.55": [ + 0.016946622715777737, + 0.013757911329974615 + ], + "1.4.rho_0.55": [ + 0.005386466630163178, + 0.01558864988803843 + ], + "2.1.rho_0.55": [ + "nan", + "nan" + ], + "2.2.rho_0.55": [ + "nan", + "nan" + ], + "2.3.rho_0.55": [ + "nan", + "nan" + ], + "2.4.rho_0.55": [ + "nan", + "nan" + ], + "1.1.rho_0.6": [ + 0.0594398122408063, + 0.045308024035041305 + ], + "1.2.rho_0.6": [ + 0.02665635658052512, + 0.01009428430629955 + ], + "1.3.rho_0.6": [ + 0.020127026484403232, + 0.017144987639204246 + ], + "1.4.rho_0.6": [ + 0.0098509448536922, + 0.01789310382229714 + ], + "2.1.rho_0.6": [ + "nan", + "nan" + ], + "2.2.rho_0.6": [ + "nan", + "nan" + ], + "2.3.rho_0.6": [ + "nan", + "nan" + ], + "2.4.rho_0.6": [ + "nan", + "nan" + ], + "1.1.rho_0.65": [ + 0.06636452154965067, + 0.048690911284104854 + ], + "1.2.rho_0.65": [ + 0.02988279156747209, + 0.012593270355836461 + ], + "1.3.rho_0.65": [ + 0.023092885480960224, + 0.018772455583715764 + ], + "1.4.rho_0.65": [ + 0.01241442885124, + 0.02076244248410586 + ], + "2.1.rho_0.65": [ + "nan", + "nan" + ], + "2.2.rho_0.65": [ + "nan", + "nan" + ], + "2.3.rho_0.65": [ + "nan", + "nan" + ], + "2.4.rho_0.65": [ + "nan", + "nan" + ], + "1.1.rho_0.7": [ + 0.07254218457867379, + 0.0526910466051157 + ], + "1.2.rho_0.7": [ + 0.03329166905191294, + 0.015175105205589068 + ], + "1.3.rho_0.7": [ + 0.0258155301357077, + 0.020502762171670685 + ], + "1.4.rho_0.7": [ + 0.014453411163469543, + 0.020846671921446847 + ], + "2.1.rho_0.7": [ + "nan", + "nan" + ], + "2.2.rho_0.7": [ + "nan", + "nan" + ], + "2.3.rho_0.7": [ + "nan", + "nan" + ], + "2.4.rho_0.7": [ + "nan", + "nan" + ], + "1.1.rho_0.75": [ + 0.07916776188641705, + 0.057886291657737066 + ], + "1.2.rho_0.75": [ + 0.03725049246712253, + 0.018772837363772443 + ], + "1.3.rho_0.75": [ + 0.028609776086117367, + 0.02307257411267194 + ], + "1.4.rho_0.75": [ + 0.016869649042968726, + 0.02362923713906029 + ], + "2.1.rho_0.75": [ + "nan", + "nan" + ], + "2.2.rho_0.75": [ + "nan", + "nan" + ], + "2.3.rho_0.75": [ + "nan", + "nan" + ], + "2.4.rho_0.75": [ + "nan", + "nan" + ], + "1.1.rho_0.8": [ + 0.08680076817335436, + 0.06389736793273171 + ], + "1.2.rho_0.8": [ + 0.040963145105353815, + 0.021929454410539284 + ], + "1.3.rho_0.8": [ + 0.03147254961686874, + 0.02464394115084445 + ], + "1.4.rho_0.8": [ + 0.017611836623551924, + 0.02768136631318152 + ], + "2.1.rho_0.8": [ + "nan", + "nan" + ], + "2.2.rho_0.8": [ + "nan", + "nan" + ], + "2.3.rho_0.8": [ + "nan", + "nan" + ], + "2.4.rho_0.8": [ + "nan", + "nan" + ], + "1.1.rho_0.85": [ + 0.09307651453825694, + 0.06816742141440546 + ], + "1.2.rho_0.85": [ + 0.04423261009456436, + 0.024040513696466315 + ], + "1.3.rho_0.85": [ + 0.03373239871230438, + 0.02743872467269859 + ], + "1.4.rho_0.85": [ + 0.02157870721620181, + 0.02748553411253518 + ], + "2.1.rho_0.85": [ + "nan", + "nan" + ], + "2.2.rho_0.85": [ + "nan", + "nan" + ], + "2.3.rho_0.85": [ + "nan", + "nan" + ], + "2.4.rho_0.85": [ + "nan", + "nan" + ], + "1.1.rho_0.9": [ + 0.10107252485664509, + 0.07345061064036906 + ], + "1.2.rho_0.9": [ + 0.048178209491760327, + 0.026937093526634648 + ], + "1.3.rho_0.9": [ + 0.036002361151492135, + 0.02854935592244301 + ], + "1.4.rho_0.9": [ + 0.024308070051008657, + 0.02733002869966339 + ], + "2.1.rho_0.9": [ + "nan", + "nan" + ], + "2.2.rho_0.9": [ + "nan", + "nan" + ], + "2.3.rho_0.9": [ + "nan", + "nan" + ], + "2.4.rho_0.9": [ + "nan", + "nan" + ] + } + }, + "2": { + "n_units": 3000000, + "n_positive_units": 1215539, + "n_trees": 10, + "min_leaf": 15, + "n_leaves": 246831, + "n_nodes": 493652, + "calibration_persons": 126094, + "rho": [ + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.05, + 0.05, + 0.15, + 0.25, + 0.9, + 0.9 + ], + [ + 0.0, + 0.0, + 0.0, + 0.0, + 0.0, + 0.0 + ] + ], + "calibration": { + "1.1.rho_0.0": [ + "nan", + "nan" + ], + "1.2.rho_0.0": [ + "nan", + "nan" + ], + "1.3.rho_0.0": [ + "nan", + "nan" + ], + "1.4.rho_0.0": [ + "nan", + "nan" + ], + "2.1.rho_0.0": [ + -0.0007703488441346273, + -0.0010508879964993278 + ], + "2.2.rho_0.0": [ + -0.007493032484792828, + -0.013193098454176932 + ], + "2.3.rho_0.0": [ + -0.009429974516101503, + -0.01195790961172627 + ], + "2.4.rho_0.0": [ + -0.03692428228749378, + -0.046884412622328786 + ], + "1.1.rho_0.05": [ + "nan", + "nan" + ], + "1.2.rho_0.05": [ + "nan", + "nan" + ], + "1.3.rho_0.05": [ + "nan", + "nan" + ], + "1.4.rho_0.05": [ + "nan", + "nan" + ], + "2.1.rho_0.05": [ + 0.0011501760114371873, + 0.0002049050210158887 + ], + "2.2.rho_0.05": [ + -0.005131023108985944, + -0.01216928087935032 + ], + "2.3.rho_0.05": [ + -0.008520996492143773, + -0.010608354960477073 + ], + "2.4.rho_0.05": [ + -0.03473496584300617, + -0.05402554667776105 + ], + "1.1.rho_0.1": [ + "nan", + "nan" + ], + "1.2.rho_0.1": [ + "nan", + "nan" + ], + "1.3.rho_0.1": [ + "nan", + "nan" + ], + "1.4.rho_0.1": [ + "nan", + "nan" + ], + "2.1.rho_0.1": [ + 0.006927157675038598, + 0.004290292190204048 + ], + "2.2.rho_0.1": [ + -0.001703132719541478, + -0.010456836329726604 + ], + "2.3.rho_0.1": [ + -0.007321013146343036, + -0.009035871667865014 + ], + "2.4.rho_0.1": [ + -0.030882723546502233, + -0.048033202775634276 + ], + "1.1.rho_0.15": [ + "nan", + "nan" + ], + "1.2.rho_0.15": [ + "nan", + "nan" + ], + "1.3.rho_0.15": [ + "nan", + "nan" + ], + "1.4.rho_0.15": [ + "nan", + "nan" + ], + "2.1.rho_0.15": [ + 0.011751352091050937, + 0.008076268003208154 + ], + "2.2.rho_0.15": [ + 0.0020210219373677507, + -0.0070992294258820365 + ], + "2.3.rho_0.15": [ + -0.005528714504464016, + -0.0069653177500850205 + ], + "2.4.rho_0.15": [ + -0.029923398294732673, + -0.05002522943078225 + ], + "1.1.rho_0.2": [ + "nan", + "nan" + ], + "1.2.rho_0.2": [ + "nan", + "nan" + ], + "1.3.rho_0.2": [ + "nan", + "nan" + ], + "1.4.rho_0.2": [ + "nan", + "nan" + ], + "2.1.rho_0.2": [ + 0.016343948418381715, + 0.011187555046907938 + ], + "2.2.rho_0.2": [ + 0.0033473023560638415, + -0.0067290354973356115 + ], + "2.3.rho_0.2": [ + -0.002877760745891411, + -0.0033953050183980205 + ], + "2.4.rho_0.2": [ + -0.025096610145121545, + -0.05020372098724413 + ], + "1.1.rho_0.25": [ + "nan", + "nan" + ], + "1.2.rho_0.25": [ + "nan", + "nan" + ], + "1.3.rho_0.25": [ + "nan", + "nan" + ], + "1.4.rho_0.25": [ + "nan", + "nan" + ], + "2.1.rho_0.25": [ + 0.021425408842097315, + 0.01429463213074944 + ], + "2.2.rho_0.25": [ + 0.0065957040671158484, + -0.004705541238731792 + ], + "2.3.rho_0.25": [ + -1.7567476131130633e-05, + -0.0010814093051690898 + ], + "2.4.rho_0.25": [ + -0.0212615966241938, + -0.04842399617411386 + ], + "1.1.rho_0.3": [ + "nan", + "nan" + ], + "1.2.rho_0.3": [ + "nan", + "nan" + ], + "1.3.rho_0.3": [ + "nan", + "nan" + ], + "1.4.rho_0.3": [ + "nan", + "nan" + ], + "2.1.rho_0.3": [ + 0.02693682357503624, + 0.016009169813349877 + ], + "2.2.rho_0.3": [ + 0.009021593962986185, + -0.001825312400977941 + ], + "2.3.rho_0.3": [ + 0.002480943042680095, + -0.0003852668318783392 + ], + "2.4.rho_0.3": [ + -0.020630295377933372, + -0.05211604831280381 + ], + "1.1.rho_0.35": [ + "nan", + "nan" + ], + "1.2.rho_0.35": [ + "nan", + "nan" + ], + "1.3.rho_0.35": [ + "nan", + "nan" + ], + "1.4.rho_0.35": [ + "nan", + "nan" + ], + "2.1.rho_0.35": [ + 0.03296557024474411, + 0.021046225240840877 + ], + "2.2.rho_0.35": [ + 0.012335421323450668, + 0.0016580900689775468 + ], + "2.3.rho_0.35": [ + 0.004414993633125364, + 0.0012509622619388816 + ], + "2.4.rho_0.35": [ + -0.017200316022581097, + -0.04689517410403199 + ], + "1.1.rho_0.4": [ + "nan", + "nan" + ], + "1.2.rho_0.4": [ + "nan", + "nan" + ], + "1.3.rho_0.4": [ + "nan", + "nan" + ], + "1.4.rho_0.4": [ + "nan", + "nan" + ], + "2.1.rho_0.4": [ + 0.03881518696695174, + 0.02617586875204103 + ], + "2.2.rho_0.4": [ + 0.014916453205874203, + 0.0031898299704220534 + ], + "2.3.rho_0.4": [ + 0.006149846624828648, + 0.0020327028751055964 + ], + "2.4.rho_0.4": [ + -0.014400133887248034, + -0.04119061699894566 + ], + "1.1.rho_0.45": [ + "nan", + "nan" + ], + "1.2.rho_0.45": [ + "nan", + "nan" + ], + "1.3.rho_0.45": [ + "nan", + "nan" + ], + "1.4.rho_0.45": [ + "nan", + "nan" + ], + "2.1.rho_0.45": [ + 0.04408209328648782, + 0.03100375174528891 + ], + "2.2.rho_0.45": [ + 0.018276462063509857, + 0.0054696188221817765 + ], + "2.3.rho_0.45": [ + 0.0076902880826571485, + 0.00218766241834345 + ], + "2.4.rho_0.45": [ + -0.014908359815873351, + -0.0393019026587349 + ], + "1.1.rho_0.5": [ + "nan", + "nan" + ], + "1.2.rho_0.5": [ + "nan", + "nan" + ], + "1.3.rho_0.5": [ + "nan", + "nan" + ], + "1.4.rho_0.5": [ + "nan", + "nan" + ], + "2.1.rho_0.5": [ + 0.04970088094513325, + 0.034854865497334075 + ], + "2.2.rho_0.5": [ + 0.021724257795143087, + 0.008193822209299872 + ], + "2.3.rho_0.5": [ + 0.009745593525695817, + 0.003562541404541153 + ], + "2.4.rho_0.5": [ + -0.014383130743268246, + -0.04271338044982742 + ], + "1.1.rho_0.55": [ + "nan", + "nan" + ], + "1.2.rho_0.55": [ + "nan", + "nan" + ], + "1.3.rho_0.55": [ + "nan", + "nan" + ], + "1.4.rho_0.55": [ + "nan", + "nan" + ], + "2.1.rho_0.55": [ + 0.05502109102831798, + 0.03890967682059354 + ], + "2.2.rho_0.55": [ + 0.024839414066947896, + 0.010572170562098249 + ], + "2.3.rho_0.55": [ + 0.011772980929248611, + 0.003908378086523001 + ], + "2.4.rho_0.55": [ + -0.013131700928571188, + -0.04285851871646029 + ], + "1.1.rho_0.6": [ + "nan", + "nan" + ], + "1.2.rho_0.6": [ + "nan", + "nan" + ], + "1.3.rho_0.6": [ + "nan", + "nan" + ], + "1.4.rho_0.6": [ + "nan", + "nan" + ], + "2.1.rho_0.6": [ + 0.06088619981198029, + 0.0420376792891467 + ], + "2.2.rho_0.6": [ + 0.028468055929460223, + 0.012861481376605255 + ], + "2.3.rho_0.6": [ + 0.014264953161897465, + 0.006716731791151176 + ], + "2.4.rho_0.6": [ + -0.01411854107127919, + -0.04322951820182941 + ], + "1.1.rho_0.65": [ + "nan", + "nan" + ], + "1.2.rho_0.65": [ + "nan", + "nan" + ], + "1.3.rho_0.65": [ + "nan", + "nan" + ], + "1.4.rho_0.65": [ + "nan", + "nan" + ], + "2.1.rho_0.65": [ + 0.06682408268645768, + 0.04478633762350159 + ], + "2.2.rho_0.65": [ + 0.03150037859622068, + 0.014635098136048796 + ], + "2.3.rho_0.65": [ + 0.016061091785444348, + 0.008363861184702004 + ], + "2.4.rho_0.65": [ + -0.013183126314957772, + -0.04726560797289203 + ], + "1.1.rho_0.7": [ + "nan", + "nan" + ], + "1.2.rho_0.7": [ + "nan", + "nan" + ], + "1.3.rho_0.7": [ + "nan", + "nan" + ], + "1.4.rho_0.7": [ + "nan", + "nan" + ], + "2.1.rho_0.7": [ + 0.07410396740658753, + 0.05061327862615472 + ], + "2.2.rho_0.7": [ + 0.03457230831577063, + 0.01752794757556897 + ], + "2.3.rho_0.7": [ + 0.018780887727013362, + 0.009609081527960028 + ], + "2.4.rho_0.7": [ + -0.011927889063067187, + -0.04575175956846467 + ], + "1.1.rho_0.75": [ + "nan", + "nan" + ], + "1.2.rho_0.75": [ + "nan", + "nan" + ], + "1.3.rho_0.75": [ + "nan", + "nan" + ], + "1.4.rho_0.75": [ + "nan", + "nan" + ], + "2.1.rho_0.75": [ + 0.08092707240167418, + 0.05483242561629709 + ], + "2.2.rho_0.75": [ + 0.03658199157962372, + 0.018699837196579194 + ], + "2.3.rho_0.75": [ + 0.021946755597646472, + 0.011296140023272394 + ], + "2.4.rho_0.75": [ + -0.009788245012555041, + -0.045540955638491365 + ], + "1.1.rho_0.8": [ + "nan", + "nan" + ], + "1.2.rho_0.8": [ + "nan", + "nan" + ], + "1.3.rho_0.8": [ + "nan", + "nan" + ], + "1.4.rho_0.8": [ + "nan", + "nan" + ], + "2.1.rho_0.8": [ + 0.08743759706500098, + 0.060096830084552244 + ], + "2.2.rho_0.8": [ + 0.038972747910327676, + 0.01962918025075 + ], + "2.3.rho_0.8": [ + 0.02443160357398244, + 0.011703961899924842 + ], + "2.4.rho_0.8": [ + -0.008951498727275298, + -0.042132069352851076 + ], + "1.1.rho_0.85": [ + "nan", + "nan" + ], + "1.2.rho_0.85": [ + "nan", + "nan" + ], + "1.3.rho_0.85": [ + "nan", + "nan" + ], + "1.4.rho_0.85": [ + "nan", + "nan" + ], + "2.1.rho_0.85": [ + 0.09255511715216769, + 0.06287928620893612 + ], + "2.2.rho_0.85": [ + 0.04200109430794119, + 0.021762940153467025 + ], + "2.3.rho_0.85": [ + 0.027419910814330595, + 0.013140726589435658 + ], + "2.4.rho_0.85": [ + -0.0057698938150033685, + -0.03462993739531095 + ], + "1.1.rho_0.9": [ + "nan", + "nan" + ], + "1.2.rho_0.9": [ + "nan", + "nan" + ], + "1.3.rho_0.9": [ + "nan", + "nan" + ], + "1.4.rho_0.9": [ + "nan", + "nan" + ], + "2.1.rho_0.9": [ + 0.09982884654946289, + 0.06787532650906625 + ], + "2.2.rho_0.9": [ + 0.04549333428061564, + 0.02470270553781595 + ], + "2.3.rho_0.9": [ + 0.03002262611212725, + 0.014889240292517925 + ], + "2.4.rho_0.9": [ + -0.0024130894891942756, + -0.034578411611535076 + ] + } + } + } + }, + "odd_knn": { + "family": "odd", + "role": "alternative", + "params": { + "unit_years": [ + 1991, + 2005 + ], + "k": 10, + "seed": 0 + }, + "file": "odd_knn_v1.npz", + "sha256": "8c7d323ded317dc336189feb7c0779cef149ce15a67dc7467008a7fad7175291", + "bytes": 4076580, + "fit_seconds": 5.8, + "diagnostics": { + "n_units": 27840831, + "bank": 1147199 + } + }, + "pre_chain": { + "family": "pre", + "role": "alternative", + "params": { + "unit_years": [ + 1951, + 2005 + ] + }, + "file": "pre_chain_v1.npz", + "sha256": "8bb48b022d9d0d27cb9f6d0517384c3b7839f636469106b30253d9b495b13724", + "bytes": 236450, + "fit_seconds": 73.6, + "diagnostics": { + "n_units": 144646920 + } + }, + "pre_donor": { + "family": "pre", + "role": "primary", + "params": { + "k": 3, + "bank_size": 100000, + "birth_years": [ + 1905, + 1985 + ] + }, + "file": "pre_donor_v1.npz", + "sha256": "3c31fbd3e93484470210d451eaca62c8fb99cf13d051fba7e648f931bd3218f7", + "bytes": 31768107, + "fit_seconds": 2.9, + "diagnostics": { + "bank": 1518845 + } + } + }, + "elapsed_seconds": 182.1 +} diff --git a/runs/epuf_fill_candidates_v1.json.env.json b/runs/epuf_fill_candidates_v1.json.env.json new file mode 100644 index 00000000..a3f2cfbe --- /dev/null +++ b/runs/epuf_fill_candidates_v1.json.env.json @@ -0,0 +1,19 @@ +{ + "environment": { + "python": "3.14.4", + "numpy": "2.5.1", + "pandas": "3.0.3", + "sklearn": "1.9.0", + "scipy": "1.18.0", + "platform": "macOS-26.6.2-arm64-arm-64bit-Mach-O", + "fitting_stack": { + "populace_fit": "absent", + "populace_frame": "absent" + } + }, + "contract": { + "blob_sha": "b0c39af1e13a705f90b85d3e6b9a91e1d3c5485c", + "head_sha": "b722382e2427cf84253daab96c206f875a8c0544", + "path": "gates.yaml" + } +} diff --git a/scripts/first_estimates_birth_evidence.py b/scripts/first_estimates_birth_evidence.py index 18b864e4..9108c560 100644 --- a/scripts/first_estimates_birth_evidence.py +++ b/scripts/first_estimates_birth_evidence.py @@ -355,6 +355,10 @@ # EPUF careers after the fact; nothing historical imports them. Path("src/populace_dynamics/harness/epuf_fill_gate.py"), Path("src/populace_dynamics/harness/epuf_fill_scoring.py"), + # The opt-in learned EPUF career fills and their application to a + # built PSID-2010 cohort; nothing historical imports them. + Path("src/populace_dynamics/estimates/epuf_fill.py"), + Path("src/populace_dynamics/cohorts/psid2010_epuf_fill.py"), ) POST_REVIEW_SHARED_SOURCE_BLOBS = { Path( diff --git a/scripts/fit_epuf_fills.py b/scripts/fit_epuf_fills.py new file mode 100644 index 00000000..982fa3da --- /dev/null +++ b/scripts/fit_epuf_fills.py @@ -0,0 +1,246 @@ +"""Fit gate_epuf_fill's four registered candidate fills on EPUF TRAIN. + +Reads only the TRAIN part of the pinned EPUF +(``epuf_fill_gate.epuf_matrix(TRAIN)``) and fits, with the registered +parameters below: + +- ``odd_forest`` (odd years, primary): :class:`BySexFill` of + :class:`OddForestFill`; +- ``odd_knn`` (odd years, alternative): :class:`OddKnnFill`; +- ``pre_donor`` (pre-career years, primary): :class:`PreDonorFill`; +- ``pre_chain`` (pre-career years, alternative): :class:`PreChainFill`. + +Each fill is written as a byte-reproducible ``.npz`` outside the repository +(``~/PolicyEngine/epuf-data/fills``, or ``POPULACE_DYNAMICS_EPUF_FILLS_DIR``), +as EPUF itself is. The manifest records each file's SHA-256, size and +parameters, the code commit, whether the fill code was clean, and the +library versions; ``epuf_fill_scoring`` loads the candidates by that +SHA-256. Usage:: + + python scripts/fit_epuf_fills.py --manifest runs/epuf_fill_candidates_v1.json +""" + +from __future__ import annotations + +import argparse +import datetime as dt +import hashlib +import os +import platform +import subprocess +import time +import zlib +from pathlib import Path + +import numpy as np + +from populace_dynamics.artifacts import write_new +from populace_dynamics.estimates import epuf_fill as F +from populace_dynamics.harness import epuf_fill_gate as g +from populace_dynamics.harness.epuf_operator import wage_base + +ROOT = Path(__file__).resolve().parents[1] +SCHEMA = "populace_dynamics.epuf_fill_candidates.v1" +DEFAULT_DIR = Path("~/PolicyEngine/epuf-data/fills").expanduser() +CODE_FILES = ( + "src/populace_dynamics/estimates/epuf_fill.py", + "src/populace_dynamics/harness/epuf_fill_gate.py", + "src/populace_dynamics/harness/epuf_operator.py", + "scripts/fit_epuf_fills.py", +) +ODD_UNIT_YEARS = tuple(range(1991, 2006)) +PRE_UNIT_YEARS = tuple(range(1951, 2006)) +#: The registered parameters of each candidate. +REGISTERED = { + "odd_forest": { + "family": "odd", + "role": "primary", + "params": { + "unit_years": [ODD_UNIT_YEARS[0], ODD_UNIT_YEARS[-1]], + "n_units": 3_000_000, + "n_trees": 10, + "min_leaf": 15, + "max_features": 0.8, + "seed": 0, + }, + }, + "odd_knn": { + "family": "odd", + "role": "alternative", + "params": { + "unit_years": [ODD_UNIT_YEARS[0], ODD_UNIT_YEARS[-1]], + "k": 10, + "seed": 0, + }, + }, + "pre_donor": { + "family": "pre", + "role": "primary", + "params": {"k": 3, "bank_size": 100_000, "birth_years": [1905, 1985]}, + }, + "pre_chain": { + "family": "pre", + "role": "alternative", + "params": {"unit_years": [PRE_UNIT_YEARS[0], PRE_UNIT_YEARS[-1]]}, + }, +} + + +def _git(*args: str) -> str: + return subprocess.run( + ["git", "-C", str(ROOT), *args], + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +def _clean() -> bool: + return ( + subprocess.run( + ["git", "-C", str(ROOT), "diff", "--quiet", "HEAD", "--"] + + list(CODE_FILES), + check=False, + ).returncode + == 0 + ) + + +def fit(name: str, shares, birth, sex, person_id): + params = REGISTERED[name]["params"] + if name == "odd_forest": + return F.BySexFill.fit( + F.OddForestFill, + shares, + g.YEARS, + birth, + sex, + ODD_UNIT_YEARS, + person_id, + n_units=params["n_units"], + n_trees=params["n_trees"], + min_leaf=params["min_leaf"], + max_features=params["max_features"], + seed=params["seed"], + ) + if name == "odd_knn": + return F.OddKnnFill.fit( + shares, + np.asarray(g.YEARS), + birth, + sex, + ODD_UNIT_YEARS, + k=params["k"], + seed=params["seed"], + ) + if name == "pre_donor": + return F.PreDonorFill.fit( + shares, + np.asarray(g.YEARS), + birth, + sex, + person_id, + k=params["k"], + birth_years=tuple(params["birth_years"]), + bank_size=params["bank_size"], + ) + if name == "pre_chain": + return F.PreChainFill.fit( + shares, np.asarray(g.YEARS), birth, sex, PRE_UNIT_YEARS + ) + raise ValueError(name) + + +def main() -> None: + import scipy + import sklearn + + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--manifest", type=Path, required=True) + parser.add_argument( + "--out-dir", + type=Path, + default=Path( + os.environ.get("POPULACE_DYNAMICS_EPUF_FILLS_DIR", DEFAULT_DIR) + ), + ) + parser.add_argument("--only", nargs="*", default=sorted(REGISTERED)) + args = parser.parse_args() + if args.manifest.exists(): + raise FileExistsError( + f"{args.manifest} exists; a registered manifest is never refitted " + "in place" + ) + started = time.time() + built_at = dt.datetime.now(dt.UTC).isoformat(timespec="seconds") + matrix = g.epuf_matrix(g.TRAIN) + caps = np.array([float(wage_base(year)) for year in g.YEARS]) + shares = matrix.earnings / caps[None, :] + args.out_dir.mkdir(parents=True, exist_ok=True) + fills = {} + for name in args.only: + t = time.time() + fill, diagnostics = fit( + name, shares, matrix.birth_year, matrix.sex, matrix.person_id + ) + blob = fill.to_bytes() + path = args.out_dir / f"{name}_v1.npz" + if path.exists(): + # A staged file is never replaced: a refit must reproduce it. + if path.read_bytes() != blob: + raise FileExistsError( + f"{path} exists with other bytes; refusing to replace it" + ) + else: + with path.open("xb") as handle: + handle.write(blob) + fills[name] = { + **REGISTERED[name], + "file": path.name, + "sha256": hashlib.sha256(blob).hexdigest(), + "bytes": len(blob), + "fit_seconds": round(time.time() - t, 1), + "diagnostics": _jsonable(diagnostics), + } + print(f"{name}: {fills[name]['sha256']} {len(blob)} bytes", flush=True) + document = { + "schema": SCHEMA, + "registration_id": g.REGISTRATION_ID, + "code_commit": _git("rev-parse", "HEAD"), + "code_files_clean": _clean(), + "built_at_utc": built_at, + "part": "train", + "n_persons": int(len(matrix.birth_year)), + "versions": { + "numpy": np.__version__, + "scipy": scipy.__version__, + "scikit_learn": sklearn.__version__, + "python": platform.python_version(), + "zlib_runtime": zlib.ZLIB_RUNTIME_VERSION, + "platform": platform.platform(), + }, + "staging": "files live outside the repository, like EPUF; a refit " + "with this script at code_commit, on the same library, zlib and " + "platform versions, reproduces their bytes", + "fills": fills, + "elapsed_seconds": round(time.time() - started, 1), + } + write_new(args.manifest, document, sidecar=True) + + +def _jsonable(value): + if isinstance(value, dict): + return {str(k): _jsonable(v) for k, v in value.items()} + if isinstance(value, list | tuple): + return [_jsonable(v) for v in value] + if isinstance(value, np.ndarray): + return _jsonable(value.tolist()) + if isinstance(value, float | np.floating): + return float(value) if np.isfinite(value) else str(float(value)) + if isinstance(value, np.integer): + return int(value) + return value + + +if __name__ == "__main__": + main() diff --git a/scripts/score_epuf_fill_test.py b/scripts/score_epuf_fill_test.py new file mode 100644 index 00000000..f61434a2 --- /dev/null +++ b/scripts/score_epuf_fill_test.py @@ -0,0 +1,145 @@ +"""gate_epuf_fill's one TEST scoring of the registered candidates. + +Runs only after ``gates.yaml`` locks the gate: it reads TEST through +``epuf_fill_gate.test_part`` (via ``epuf_fill_scoring.score_registered``), +which refuses otherwise. It loads the registered candidates named in the +manifest (``runs/epuf_fill_candidates_v1.json``) by their SHA-256 from the +staged fills folder, scores each family's current rule, primary and +alternative over the registered draw seeds against the registered v3 +floors, and writes the result whether the candidates pass or fail. It +refuses to overwrite an existing result. Usage:: + + python scripts/score_epuf_fill_test.py \ + --manifest runs/epuf_fill_candidates_v1.json \ + --output runs/epuf_fill_gate_test_v1.json +""" + +from __future__ import annotations + +import argparse +import datetime as dt +import hashlib +import json +import os +import subprocess +import time +from pathlib import Path + +from populace_dynamics.artifacts import write_new +from populace_dynamics.harness import epuf_fill_gate as g +from populace_dynamics.harness import epuf_fill_scoring as scoring + +ROOT = Path(__file__).resolve().parents[1] +DEFAULT_DIR = Path("~/PolicyEngine/epuf-data/fills").expanduser() +#: The registered manifest; any other manifest is refused. +REGISTERED_MANIFEST = "runs/epuf_fill_candidates_v1.json" +REGISTERED_MANIFEST_SHA256 = ( + "83d17a14f960033c7c0ed0d602ae395ea4b8d66ff5facab85115f493f3b96e2c" +) +#: Files whose state the record reports. +CODE_FILES = ( + "src/populace_dynamics/harness/epuf_fill_gate.py", + "src/populace_dynamics/harness/epuf_fill_scoring.py", + "src/populace_dynamics/harness/epuf_cells.py", + "src/populace_dynamics/harness/epuf_operator.py", + "src/populace_dynamics/estimates/epuf_fill.py", + "scripts/score_epuf_fill_test.py", +) + + +def candidates_from(manifest: dict, fills_dir: Path) -> dict: + """``{family: {role: (path, sha256)}}`` from a candidate manifest.""" + + spec: dict[str, dict[str, tuple[Path, str]]] = {} + for record in manifest["fills"].values(): + spec.setdefault(record["family"], {})[record["role"]] = ( + fills_dir / record["file"], + record["sha256"], + ) + return spec + + +def main(argv: list[str] | None = None) -> None: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--manifest", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument( + "--fills-dir", + type=Path, + default=Path( + os.environ.get("POPULACE_DYNAMICS_EPUF_FILLS_DIR", DEFAULT_DIR) + ), + ) + args = parser.parse_args(argv) + if args.output.exists(): + raise FileExistsError(f"{args.output} exists; TEST is scored once") + manifest_bytes = args.manifest.read_bytes() + manifest_sha256 = hashlib.sha256(manifest_bytes).hexdigest() + if manifest_sha256 != REGISTERED_MANIFEST_SHA256: + raise ValueError( + f"{args.manifest} has SHA-256 {manifest_sha256}, not the " + f"registered {REGISTERED_MANIFEST_SHA256}" + ) + started = time.time() + clean = ( + subprocess.run( + ["git", "-C", str(ROOT), "diff", "--quiet", "HEAD", "--"] + + list(CODE_FILES), + check=False, + ).returncode + == 0 + ) + # Refuse before any marker unless the gate is locked (test_part checks + # again before it reads TEST). + status = g._gate_lock_status(ROOT / "gates.yaml") + if not status["locked"] or status["registration_id"] != g.REGISTRATION_ID: + raise g.TestPartLocked( + "gate_epuf_fill is not locked; TEST stays unread" + ) + # The staged candidates must be the registered bytes before any marker. + manifest = json.loads(manifest_bytes) + spec = candidates_from(manifest, args.fills_dir) + for family, roles in spec.items(): + for role, (path, sha256) in roles.items(): + if hashlib.sha256(path.read_bytes()).hexdigest() != sha256: + raise ValueError( + f"{family} {role}: {path.name} is not the registered bytes" + ) + # A started marker, so a run that fails after reading TEST leaves a + # trace of the read. + marker = Path(f"{args.output}.started.json") + with marker.open("x") as handle: + json.dump( + { + "started_at_utc": dt.datetime.now(dt.UTC).isoformat( + timespec="seconds" + ), + "manifest_sha256": manifest_sha256, + }, + handle, + ) + record = scoring.score_registered(spec) + # Published paths are the registered file names, not local folders. + for family in record["candidates"].values(): + for role in family.values(): + role["path"] = Path(role["path"]).name + document = { + "schema": "populace_dynamics.epuf_fill_gate_test.v1", + "code_commit": subprocess.run( + ["git", "-C", str(ROOT), "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + ).stdout.strip(), + "scored_at_utc": dt.datetime.now(dt.UTC).isoformat(timespec="seconds"), + "code_files_clean": clean, + "manifest": REGISTERED_MANIFEST, + "manifest_sha256": manifest_sha256, + **record, + "elapsed_seconds": round(time.time() - started, 1), + } + write_new(args.output, document, sidecar=True) + + +if __name__ == "__main__": + main() diff --git a/src/populace_dynamics/cohorts/psid2010_epuf_fill.py b/src/populace_dynamics/cohorts/psid2010_epuf_fill.py new file mode 100644 index 00000000..35726600 --- /dev/null +++ b/src/populace_dynamics/cohorts/psid2010_epuf_fill.py @@ -0,0 +1,221 @@ +"""Learned EPUF career fills applied to a built PSID-2010 cohort (opt-in). + +The PSID-2010 cohort's careers (:func:`populace_dynamics.cohorts.psid2010. +build_psid2010_cohort`) come from ``career.build_career``, which fills each +odd income year from 1997 with its neighbours' mean (provenance +``gap_imputed``) and counts nothing before ``max(1968, birth_year + 22)``. +:func:`fill_careers` replaces either rule with a fill learned from SSA's +Earnings Public-Use File and registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``): + +- every ``gap_imputed`` year in ``odd_years`` with a neighbour the fill + can see becomes the odd fill's draw, with provenance + :attr:`EPUFFillProvenance.GAP_EPUF_DRAWN`; + - by default ``odd_years`` is the years the gate scored (1997-2005); + - the PSID's later gap years (2007-2011 and the 2013 seam) keep the + assembler's value unless the caller passes ``odd_years=None``, which + fills every gap year and is an extrapolation the gate does not certify; + - a gap year with no visible neighbour keeps the assembler's value, which + used a neighbour the fill does not see (an observed pre-career year, or + the 2014 boundary year); +- every year from 1951 before the career start that has no career row + becomes the pre-career fill's draw, with provenance + :attr:`EPUFFillProvenance.PRE_CAREER_EPUF_DONOR` (the assembler's careers + start at the career start, so in practice every such year). + +The fills see what the gate's scoring path gives them: +- capped shares of the wage base for the career years the PSID recorded; +- every other year unknown, including pre-career years the PSID happened to + record. + +A drawn share becomes capped earnings at that year's wage base. Every other +career row is left exactly as built. Nothing here changes +``career.build_career``, ``cohorts.psid2010`` or any registered run; using +learned fills in a registered comparison needs its own registration. +""" + +from __future__ import annotations + +import hashlib +from dataclasses import dataclass +from enum import Enum +from typing import Any + +import numpy as np +import pandas as pd + +__all__ = [ + "EPUFFillProvenance", + "FilledCareers", + "fill_careers", +] + +FIRST_YEAR = 1951 +_SEX_CODE = {"male": 1, "female": 2} + + +class EPUFFillProvenance(str, Enum): + """Provenance of a career year filled by a learned EPUF fill.""" + + GAP_EPUF_DRAWN = "gap_epuf_drawn" + PRE_CAREER_EPUF_DONOR = "pre_career_epuf_donor" + + +@dataclass(frozen=True) +class FilledCareers: + """Careers after the learned fills, with what produced them.""" + + careers: pd.DataFrame + fills: dict[str, str] + seed: int + content_sha256: str + + +def _wage_bases(years: np.ndarray) -> np.ndarray: + from populace_dynamics.cola_track_a.statutory import ( + captured_ssa_parameters, + ) + + params = captured_ssa_parameters() + return np.array([float(params.wage_base_for(int(y))) for y in years]) + + +def _content_sha256(careers: pd.DataFrame) -> str: + ordered = careers.sort_values(["person_id", "year"], kind="stable") + return hashlib.sha256( + ordered.to_csv(index=False, float_format="%.6f").encode() + ).hexdigest() + + +#: The gap years the gate scored (``epuf_fill_gate.MASKED_ODD_YEARS``). +SCORED_ODD_YEARS: tuple[int, ...] = (1997, 1999, 2001, 2003, 2005) + + +def fill_careers( + cohort: Any, + *, + odd_fill: Any | None = None, + pre_fill: Any | None = None, + seed: int, + last_year: int | None = None, + odd_years: tuple[int, ...] | None = SCORED_ODD_YEARS, +) -> FilledCareers: + """Replace the assembler's fill rules with learned fills. + + ``cohort`` needs ``persons`` (``person_id``, ``birth_year``, ``sex`` as + ``"male"`` / ``"female"``) and ``careers`` (``person_id``, ``year``, + ``earnings``, ``provenance``). Either fill may be None, which keeps that + rule. ``last_year`` is the last career year read (default: the latest + in ``careers``). ``odd_years`` limits the odd fill to those gap years; + None fills every gap year. + """ + + persons = cohort.persons[["person_id", "birth_year", "sex"]].copy() + careers = cohort.careers.copy() + last = int(careers["year"].max()) if last_year is None else last_year + years = np.arange(FIRST_YEAR, last + 1) + caps = _wage_bases(years) + person_ids = persons["person_id"].to_numpy(dtype=np.int64) + row_of = {int(pid): i for i, pid in enumerate(person_ids)} + birth = persons["birth_year"].to_numpy(dtype=np.int64) + sex = persons["sex"].map(_SEX_CODE).fillna(3).to_numpy(dtype=np.int64) + n = len(person_ids) + + shares = np.full((n, len(years)), np.nan) + odd_mask = np.zeros((n, len(years)), dtype=bool) + in_career = np.zeros((n, len(years)), dtype=bool) + rows = careers["person_id"].map(row_of).to_numpy() + if np.isnan(rows.astype(float)).any(): + raise ValueError("a career row's person is not in the cohort") + columns = careers["year"].to_numpy(dtype=np.int64) - FIRST_YEAR + inside = (columns >= 0) & (columns < len(years)) + rows, columns = rows[inside].astype(np.int64), columns[inside] + provenance = careers["provenance"].astype(str).to_numpy()[inside] + earnings = careers["earnings"].to_numpy(dtype=np.float64)[inside] + in_career[rows, columns] = True + observed = provenance == "observed" + shares[rows[observed], columns[observed]] = np.minimum( + np.maximum(earnings[observed], 0.0) / caps[columns[observed]], 1.0 + ) + gap = provenance == "gap_imputed" + if odd_years is not None: + gap &= np.isin(years[columns], np.asarray(odd_years)) + start = np.maximum(1968, birth + 22) + pre_mask = years[None, :] < start[:, None] + # Every gap year is unknown to the fills, filled or not. + all_gaps = np.zeros_like(odd_mask) + is_gap = provenance == "gap_imputed" + all_gaps[rows[is_gap], columns[is_gap]] = True + given = np.where(all_gaps | pre_mask, np.nan, shares) + # A gap year is filled only if a neighbour is visible to the fill. + left = np.full(len(rows), np.nan) + right = np.full(len(rows), np.nan) + inner = columns > 0 + left[inner] = given[rows[inner], columns[inner] - 1] + outer = columns + 1 < len(years) + right[outer] = given[rows[outer], columns[outer] + 1] + gap &= np.isfinite(left) | np.isfinite(right) + odd_mask[rows[gap], columns[gap]] = True + + def drawn(fill, mask): + out = np.asarray( + fill.fill( + given.copy(), years, birth, sex, person_ids, mask.copy(), seed + ), + dtype=np.float64, + ) + values = out[mask] + if ( + not np.isfinite(values).all() + or ((values < 0) | (values > 1)).any() + ): + raise ValueError("a learned fill returned an invalid share") + return out + + result = careers.copy() + names: dict[str, str] = {} + if odd_fill is not None: + out = drawn(odd_fill, odd_mask) + cells = out[rows[gap], columns[gap]] + values = np.where( + cells >= 1.0, caps[columns[gap]], cells * caps[columns[gap]] + ) + index = careers.index[inside][gap] + result.loc[index, "earnings"] = values + result.loc[index, "provenance"] = ( + EPUFFillProvenance.GAP_EPUF_DRAWN.value + ) + names["odd"] = getattr(odd_fill, "name", type(odd_fill).__name__) + if pre_fill is not None: + out = drawn(pre_fill, pre_mask) + pre_rows, pre_columns = np.nonzero(pre_mask & ~in_career) + cells = out[pre_rows, pre_columns] + added = pd.DataFrame( + { + "person_id": person_ids[pre_rows], + "year": years[pre_columns], + "earnings": np.where( + cells >= 1.0, + caps[pre_columns], + cells * caps[pre_columns], + ), + "provenance": EPUFFillProvenance.PRE_CAREER_EPUF_DONOR.value, + } + ) + result = pd.concat([added, result], ignore_index=True) + names["pre"] = getattr(pre_fill, "name", type(pre_fill).__name__) + result = result.sort_values(["person_id", "year"], kind="stable") + result = result.reset_index(drop=True).astype( + { + "person_id": "int64", + "year": "int64", + "earnings": "float64", + "provenance": "string", + } + ) + return FilledCareers( + careers=result, + fills=names, + seed=int(seed), + content_sha256=_content_sha256(result), + ) diff --git a/src/populace_dynamics/estimates/epuf_fill.py b/src/populace_dynamics/estimates/epuf_fill.py new file mode 100644 index 00000000..807932df --- /dev/null +++ b/src/populace_dynamics/estimates/epuf_fill.py @@ -0,0 +1,1703 @@ +"""Career fills learned from SSA's Earnings Public-Use File (EPUF). + +The career assembler (:func:`populace_dynamics.estimates.career.build_career`) +fills the years the PSID did not record with two fixed rules: each odd +income year from 1997 is the mean of its neighbours, and nothing counts +before ``max(1968, birth_year + 22)``. This module holds the learned +replacements registered by ``gate_epuf_fill`` +(``docs/amendments/gate_epuf_fill_registration_proposal.md``, section 7), +fitted on the gate's TRAIN persons only: + +- :class:`OddForestFill` (odd years, primary; fitted per sex through + :class:`BySexFill`): a two-part draw from random forests. A probability + forest gives the chance of a zero year, and a quantile regression forest + gives the positive share. Both condition on the recorded shares around + ``t``, sex, age and year. A person-level Gaussian copula, calibrated on + held-out TRAIN persons, carries the persistence across a person's masked + years. +- :class:`OddKnnFill` (odd years, alternative): the share at ``t`` copied + from one of the ``k`` nearest TRAIN person-years in the shares at ``t-1`` + and ``t+1``, by sex and age. +- :class:`PreDonorFill` (pre-career years, primary): rank-kNN donor + careers. The whole masked block is copied from one of the ``k`` TRAIN + donors of the same sex and birth year nearest in percentile rank over + the first five recorded years and two career-wide summaries. +- :class:`PreChainFill` (pre-career years, alternative): a chained + one-sided draw of year ``y`` given year ``y+1``, sex and age, backward + from the career start. + +Every fill works on **shares**: capped earnings over the year's wage base, +in [0, 1], NaN where a year is unknown. It fills only the cells of +``fill_mask`` and leaves every other cell as given. Draws come from +counter-based uniforms keyed by the fill, the draw seed, the person key and +the year (:func:`hash_uniform`), so a person's draw never depends on which +other persons are filled or in what order. +""" + +from __future__ import annotations + +import hashlib +import io +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +import numpy as np +from scipy.special import ndtr, ndtri + +__all__ = [ + "BySexFill", + "FILL_CLASSES", + "OddForestFill", + "OddKnnFill", + "PreChainFill", + "PreDonorFill", + "block_first_year", + "career_start", + "hash_uniform", + "load_fill", + "odd_context", +] + +CAREER_FIRST_YEAR = 1968 +CAREER_START_AGE = 22 +#: EPUF has no earnings below this age for cohorts born after 1937. +FIRST_EARNING_AGE = 15 +QUANTILE_POINTS = 65 +MIN_CELL = 200 + +_MASK64 = np.uint64(0xFFFFFFFFFFFFFFFF) +_GOLDEN = np.uint64(0x9E3779B97F4A7C15) +_MIX1 = np.uint64(0xBF58476D1CE4E5B9) +_MIX2 = np.uint64(0x94D049BB133111EB) + + +def _splitmix64(values: np.ndarray) -> np.ndarray: + with np.errstate(over="ignore"): + z = values.astype(np.uint64) + _GOLDEN + z = (z ^ (z >> np.uint64(30))) * _MIX1 + z = (z ^ (z >> np.uint64(27))) * _MIX2 + return z ^ (z >> np.uint64(31)) + + +def _tag(name: str) -> np.uint64: + digest = hashlib.sha256(name.encode()).digest()[:8] + return np.uint64(int.from_bytes(digest, "big")) + + +def hash_uniform( + stream: str, seed: int, person_key: np.ndarray, year: np.ndarray +) -> np.ndarray: + """Uniforms in (0, 1) keyed by stream, seed, person and year. + + A splitmix64 chain over ``(stream tag XOR seed, person key, year)``; + broadcasting ``person_key`` against ``year`` gives one uniform per + person-year. + """ + + person_key = np.asarray(person_key, dtype=np.int64).astype(np.uint64) + year = np.asarray(year, dtype=np.int64).astype(np.uint64) + base = _splitmix64(np.asarray(_tag(stream) ^ np.uint64(seed))) + with np.errstate(over="ignore"): + state = _splitmix64(base ^ person_key) + state = _splitmix64(state ^ (year * _GOLDEN)) + return ((state >> np.uint64(11)).astype(np.float64) + 0.5) / 2.0**53 + + +def career_start(birth_year: np.ndarray) -> np.ndarray: + """The assembler's first career year, ``max(1968, birth_year + 22)``.""" + + return np.maximum( + CAREER_FIRST_YEAR, np.asarray(birth_year, dtype=np.int64) + 22 + ) + + +def _age_band(age: np.ndarray) -> np.ndarray: + """0 below 15; 1 for 15-19 through 14 for 80-84; 15 from 85.""" + + age = np.asarray(age, dtype=np.int64) + return np.where(age < 15, 0, np.minimum((age - 15) // 5 + 1, 15)) + + +def _column_of(years: np.ndarray, target: np.ndarray) -> np.ndarray: + """Column of each target year, -1 outside the matrix's years.""" + + years = np.asarray(years, dtype=np.int64) + target = np.asarray(target, dtype=np.int64) + column = target - years[0] + return np.where((column >= 0) & (column < len(years)), column, -1) + + +def _take(shares: np.ndarray, rows: np.ndarray, column: np.ndarray): + """Shares at (row, column), NaN where the column is -1.""" + + safe = np.maximum(column, 0) + out = shares[rows, safe] + return np.where(column >= 0, out, np.nan) + + +def _quantile_table( + keys: np.ndarray, values: np.ndarray, points: int = QUANTILE_POINTS +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Per key: sorted unique keys, counts, and ``points`` quantiles. + + The quantiles are at levels ``j / (points - 1)`` with linear + interpolation, so they include each key's minimum and maximum. + """ + + order = np.lexsort((values, keys)) + keys = keys[order] + values = values[order] + unique, start, count = np.unique( + keys, return_index=True, return_counts=True + ) + levels = np.linspace(0.0, 1.0, points) + position = levels[None, :] * (count[:, None] - 1) + low = np.floor(position).astype(np.int64) + high = np.minimum(low + 1, count[:, None] - 1) + weight = position - low + base = start[:, None] + table = (1.0 - weight) * values[base + low] + weight * values[base + high] + return unique, count, table.astype(np.float32) + + +def _lookup(table_keys: np.ndarray, keys: np.ndarray) -> np.ndarray: + """Index of each key in sorted ``table_keys``, -1 where absent.""" + + if len(table_keys) == 0: + return np.full(len(keys), -1, dtype=np.int64) + position = np.searchsorted(table_keys, keys) + position = np.minimum(position, len(table_keys) - 1) + return np.where(table_keys[position] == keys, position, -1) + + +def _interpolate(table: np.ndarray, rows: np.ndarray, level: np.ndarray): + """Row-wise linear interpolation of quantile tables at levels in [0, 1].""" + + points = table.shape[1] + position = np.clip(level, 0.0, 1.0) * (points - 1) + low = np.minimum(np.floor(position).astype(np.int64), points - 2) + weight = position - low + return (1.0 - weight) * table[rows, low] + weight * table[rows, low + 1] + + +def _to_npz(arrays: Mapping[str, np.ndarray]) -> bytes: + """A compressed ``.npz`` whose bytes depend only on the arrays. + + ``numpy.savez_compressed`` stamps each member with the time of writing, + so two writes of the same fill differ. This writer fixes every member's + timestamp and order, so a fill's SHA-256 can be registered and refit. + """ + + import zipfile + + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive: + for name in sorted(arrays): + member = io.BytesIO() + np.lib.format.write_array( + member, np.asanyarray(arrays[name]), allow_pickle=False + ) + info = zipfile.ZipInfo( + f"{name}.npy", date_time=(1980, 1, 1, 0, 0, 0) + ) + info.compress_type = zipfile.ZIP_DEFLATED + archive.writestr(info, member.getvalue()) + return buffer.getvalue() + + +# -------------------------------------------------------------------------- +# Odd years: the context of a masked unit +# -------------------------------------------------------------------------- +#: Offsets whose positivity forms the wider context ``W``. +_WIDE_OFFSETS = (-9, -7, -5, 5, 7, 9) + + +@dataclass(frozen=True) +class OddContext: + """The recorded neighbourhood of masked units (one row per unit).""" + + left: np.ndarray + right: np.ndarray + left3: np.ndarray + right3: np.ndarray + wide: np.ndarray + sex: np.ndarray + age: np.ndarray + wide_mean: np.ndarray + wide_positive: np.ndarray + wide_known: np.ndarray + year: np.ndarray + + +def odd_context( + shares: np.ndarray, + years: np.ndarray, + rows: np.ndarray, + unit_year: np.ndarray, + birth_year: np.ndarray, + sex: np.ndarray, + known: np.ndarray, +) -> OddContext: + """Neighbour shares of units ``(rows, unit_year)``; NaN where unknown. + + ``known`` (persons by years) flags the cells a fill may read: recorded + and not masked. ``wide`` is 1 if any known share at offsets 5, 7 or 9 + on either side is positive. + """ + + readable = np.where(known, shares, np.nan) + + def at(offset: int) -> np.ndarray: + return _take(readable, rows, _column_of(years, unit_year + offset)) + + wide = np.zeros(len(rows), dtype=np.int64) + total = np.zeros(len(rows)) + positive = np.zeros(len(rows)) + count = np.zeros(len(rows)) + for offset in _WIDE_OFFSETS: + value = at(offset) + known_value = np.isfinite(value) + is_positive = np.nan_to_num(value, nan=0.0) > 0 + wide |= is_positive.astype(np.int64) + count += known_value + positive += is_positive + total += np.where(is_positive, value, 0.0) + return OddContext( + left=at(-1), + right=at(1), + left3=at(-3), + right3=at(3), + wide=wide, + sex=np.asarray(sex)[rows].astype(np.int64), + age=unit_year - np.asarray(birth_year)[rows], + wide_mean=np.where( + positive > 0, total / np.maximum(positive, 1), -1.0 + ), + wide_positive=np.where( + count > 0, positive / np.maximum(count, 1), -1.0 + ), + wide_known=count, + year=np.asarray(unit_year, dtype=np.int64), + ) + + +def odd_features(context: OddContext) -> np.ndarray: + """Forest features of masked units; -1 marks an unknown share. + + Sex, age, the shares at ``t-1`` and ``t+1`` (a missing one takes the + other's value, and a flag records it), at ``t-3`` and ``t+3``; the + mean, the geometric mean of the positive ones, and the number positive + of the known shares among those four; the mean positive share and the + share of positive years among the known shares at offsets 5, 7 and 9 + on both sides, and the number of those known. + """ + + left = context.left + right = context.right + missing = np.isnan(left) | np.isnan(right) + left = np.where(np.isnan(left), right, left) + right = np.where(np.isnan(right), context.left, right) + near = np.column_stack([left, right, context.left3, context.right3]) + known = np.isfinite(near) + values = np.where(known, near, 0.0) + count = known.sum(axis=1) + positive = (values > 0) & known + n_positive = positive.sum(axis=1) + mean = np.where(count > 0, values.sum(axis=1) / np.maximum(count, 1), -1) + log_positive = np.where(positive, np.log(np.where(positive, values, 1)), 0) + geometric = np.where( + n_positive > 0, + np.exp(log_positive.sum(axis=1) / np.maximum(n_positive, 1)), + -1.0, + ) + return np.column_stack( + [ + context.sex.astype(np.float64), + context.age.astype(np.float64), + np.nan_to_num(left, nan=-1.0), + np.nan_to_num(right, nan=-1.0), + missing.astype(np.float64), + np.nan_to_num(context.left3, nan=-1.0), + np.nan_to_num(context.right3, nan=-1.0), + mean, + geometric, + n_positive.astype(np.float64), + context.wide_mean, + context.wide_positive, + context.wide_known, + context.year.astype(np.float64), + ] + ).astype(np.float32) + + +def _tree_leaves( + left: np.ndarray, + right: np.ndarray, + feature: np.ndarray, + threshold: np.ndarray, + x: np.ndarray, +) -> np.ndarray: + """Leaf node of each row, following ``x[feature] <= threshold`` left.""" + + node = np.zeros(len(x), dtype=np.int64) + while True: + internal = left[node] >= 0 + if not internal.any(): + return node + rows = np.flatnonzero(internal) + current = node[rows] + go_left = x[rows, feature[current]] <= threshold[current] + node[rows] = np.where(go_left, left[current], right[current]) + + +_SHARE_SCALE = 65_535 + + +#: Age bands of the person-level copula (the gate's odd-year bands). +_COPULA_BAND_EDGES = (22, 30, 45, 60, 75) +_RHO_GRID = tuple(np.round(np.arange(0.0, 0.91, 0.05), 2)) +#: TRAIN persons held out of the forest to calibrate the copula. +_CALIBRATION_SHARE = 0.1 +_CALIBRATION_YEARS = (1997, 1999, 2001, 2003, 2005) + + +def _copula_band(age: np.ndarray) -> np.ndarray: + """0 under 22, 1 for 22-29, 2 for 30-44, 3 for 45-59, 4 for 60-74, 5 on.""" + + return np.digitize(np.asarray(age), _COPULA_BAND_EDGES) + + +@dataclass(frozen=True) +class OddForestFill: + """A quantile regression forest draw (Meinshausen 2006), with a copula. + + Two parts, both random forests (scikit-learn) on :func:`odd_features` + of TRAIN units inside the career, whose contexts see the career only: + + 1. a probability forest for a zero year: ``p0`` is the mean over trees + of the zero share of the unit's leaves; + 2. a quantile regression forest on positive shares (split target + ``log share``): every positive TRAIN unit used in the fit is passed + down every tree, and each leaf keeps the sorted true shares that + reach it (the cap included, stored as shares times 65,535). + + A draw maps the copula uniform ``u`` to zero below ``p0``; otherwise a + second seeded uniform picks a tree, and the share is that tree's leaf + value at the quantile ``(u - p0) / (1 - p0)``. + + The copula is person-level: a unit's normal score is ``sqrt(rho) * eta + + sqrt(1 - rho) * eps``, with ``eta`` one draw per person and ``eps`` + one per unit, and ``rho`` by sex and age band at the unit. It carries + the persistence across a person's masked years that the conditioning + leaves. ``rho`` is calibrated on TRAIN persons held out of the forest + (one in ten, by hash): their odd years 1997-2005 are masked as the + gate masks them, and each band's ``rho`` is the grid value whose fills + best match their true two- and four-year rank persistence between + masked years. + """ + + tree_offsets: np.ndarray + node_left: np.ndarray + node_right: np.ndarray + node_feature: np.ndarray + node_threshold: np.ndarray + node_leaf: np.ndarray + leaf_offsets: np.ndarray + leaf_values: np.ndarray + rho: np.ndarray + zero_tree_offsets: np.ndarray + zero_node_left: np.ndarray + zero_node_right: np.ndarray + zero_node_feature: np.ndarray + zero_node_threshold: np.ndarray + zero_node_leaf: np.ndarray + zero_leaf_p: np.ndarray + stream: str = "epuf_fill.odd_forest.v4" + name: str = "odd_forest" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + unit_years, + person_key=None, + *, + n_units=3_000_000, + n_trees=10, + min_leaf=15, + max_features=0.8, + seed=0, + n_jobs=10, + ): + from sklearn.ensemble import RandomForestRegressor + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex) + n = len(shares) + key = np.arange(n) if person_key is None else np.asarray(person_key) + calibration = ( + hash_uniform(cls.stream + ".calibration", seed, key, 0) + < _CALIBRATION_SHARE + ) + fitting = np.flatnonzero(~calibration) + rows = np.concatenate([fitting for _ in unit_years]) + unit_year = np.concatenate( + [np.full(len(fitting), y) for y in unit_years] + ) + # Units lie inside the career, and their contexts see the career + # only, as a fill's do (pre-career years are unknown to it). + inside = unit_year >= career_start(birth_year[rows]) + rows, unit_year = rows[inside], unit_year[inside] + pre_career = years[None, :] < career_start(birth_year)[:, None] + known = np.isfinite(shares) & ~pre_career + target = _take(shares, rows, _column_of(years, unit_year)) + rng = np.random.default_rng(seed) + chosen = np.sort( + rng.choice(len(rows), size=min(n_units, len(rows)), replace=False) + ) + context = odd_context( + shares, + years, + rows[chosen], + unit_year[chosen], + birth_year, + sex, + known, + ) + x_all = odd_features(context) + y_all = target[chosen] + # Part one: the probability of a zero year, a probability forest. + from sklearn.ensemble import RandomForestClassifier + + zero_forest = RandomForestClassifier( + n_estimators=n_trees, + min_samples_leaf=4 * min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + zero_forest.fit(x_all, (y_all <= 0).astype(np.int8)) + zero_arrays = _forest_arrays( + zero_forest, x_all, (y_all <= 0).astype(np.float64) + ) + # Part two: the positive share, a quantile regression forest. + positive = y_all > 0 + x = x_all[positive] + y = y_all[positive] + forest = RandomForestRegressor( + n_estimators=n_trees, + min_samples_leaf=min_leaf, + max_features=max_features, + bootstrap=True, + max_samples=0.5, + random_state=seed, + n_jobs=n_jobs, + ) + forest.fit(x, np.log(y)) + stored = np.round(np.clip(y, 0.0, 1.0) * _SHARE_SCALE).astype( + np.uint16 + ) + tree_offsets = [0] + leaf_offsets = [0] + lefts, rights, features, thresholds, leaf_index, values = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + index = np.full(tree.node_count, -1, dtype=np.int64) + n_leaves = int(is_leaf.sum()) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + order = np.lexsort((stored, local)) + counts = np.bincount(local, minlength=n_leaves) + if (counts == 0).any(): + raise ValueError( + "a forest leaf holds no TRAIN unit under the stored " + "traversal" + ) + leaf_offsets.extend( + (leaf_offsets[-1] + np.cumsum(counts)).tolist() + ) + values.append(stored[order]) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + tree_offsets.append(tree_offsets[-1] + tree.node_count) + provisional = cls( + tree_offsets=np.asarray(tree_offsets, dtype=np.int64), + node_left=np.concatenate(lefts).astype(np.int32), + node_right=np.concatenate(rights).astype(np.int32), + node_feature=np.concatenate(features), + node_threshold=np.concatenate(thresholds), + node_leaf=np.concatenate(leaf_index).astype(np.int32), + leaf_offsets=np.asarray(leaf_offsets, dtype=np.int64), + leaf_values=np.concatenate(values), + rho=np.zeros((4, 6)), + **zero_arrays, + ) + rho, calibration_record = provisional._calibrate( + shares[calibration], + years, + birth_year[calibration], + sex[calibration], + key[calibration], + seed, + ) + fill = cls(**{**provisional.__dict__, "rho": rho}) + return fill, { + "n_units": int(len(y_all)), + "n_positive_units": int(len(y)), + "n_trees": n_trees, + "min_leaf": min_leaf, + "n_leaves": int(leaf_count), + "n_nodes": int(tree_offsets[-1]), + "calibration_persons": int(calibration.sum()), + "rho": rho.tolist(), + "calibration": calibration_record, + } + + # -- the conditional law ---------------------------------------------------- + @property + def n_trees(self) -> int: + return len(self.tree_offsets) - 1 + + def _leaves(self, x: np.ndarray, tree: int) -> np.ndarray: + start, stop = self.tree_offsets[tree], self.tree_offsets[tree + 1] + node = _tree_leaves( + self.node_left[start:stop].astype(np.int64), + self.node_right[start:stop].astype(np.int64), + self.node_feature[start:stop].astype(np.int64), + self.node_threshold[start:stop], + np.asarray(x, dtype=np.float32), + ) + return self.node_leaf[start:stop][node].astype(np.int64) + + def _p_zero(self, x) -> np.ndarray: + """The probability forest's zero-year probability (mean over trees).""" + + x = np.asarray(x, dtype=np.float32) + n_trees = len(self.zero_tree_offsets) - 1 + total = np.zeros(len(x)) + for tree in range(n_trees): + start = self.zero_tree_offsets[tree] + stop = self.zero_tree_offsets[tree + 1] + node = _tree_leaves( + self.zero_node_left[start:stop].astype(np.int64), + self.zero_node_right[start:stop].astype(np.int64), + self.zero_node_feature[start:stop].astype(np.int64), + self.zero_node_threshold[start:stop], + x, + ) + total += self.zero_leaf_p[ + self.zero_node_leaf[start:stop][node].astype(np.int64) + ] + return total / n_trees + + def _chosen_leaves(self, x, tree_u) -> np.ndarray: + """Each unit's leaf in the tree its uniform picks.""" + + tree = np.minimum( + (tree_u * self.n_trees).astype(np.int64), self.n_trees - 1 + ) + leaves = np.empty(len(tree_u), dtype=np.int64) + for t in range(self.n_trees): + rows = np.flatnonzero(tree == t) + if len(rows): + leaves[rows] = self._leaves(x[rows], t) + return leaves + + def _value(self, leaves, u) -> np.ndarray: + """The leaf's stored share at quantile ``u``.""" + + start = self.leaf_offsets[leaves] + count = self.leaf_offsets[leaves + 1] - start + pick = start + np.minimum((u * count).astype(np.int64), count - 1) + return self.leaf_values[pick] / _SHARE_SCALE + + def _units(self, shares, years, birth_year, sex, person_key, mask, seed): + """Per masked unit: row, year, leaf, epsilon, eta, sex and band.""" + + known = np.isfinite(shares) & ~mask + eta = ndtri(hash_uniform(self.stream + ".person", seed, person_key, 0)) + out = [] + for column in np.flatnonzero(mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + valid = ~(np.isnan(context.left) & np.isnan(context.right)) + tree_u = hash_uniform( + self.stream + ".tree", seed, person_key[rows], unit_year + ) + leaves = np.full(len(rows), -1, dtype=np.int64) + p_zero = np.ones(len(rows)) + if valid.any(): + features = odd_features( + OddContext( + **{k: v[valid] for k, v in context.__dict__.items()} + ) + ) + leaves[valid] = self._chosen_leaves(features, tree_u[valid]) + p_zero[valid] = self._p_zero(features) + out.append( + { + "column": column, + "rows": rows, + "leaves": leaves, + "epsilon": ndtri( + hash_uniform( + self.stream, seed, person_key[rows], unit_year + ) + ), + "eta": eta[rows], + "p_zero": p_zero, + "sex": np.clip(context.sex, 0, 3), + "band": _copula_band(context.age), + } + ) + return out + + def _apply(self, units, shares, mask, rho): + out = np.where(mask, np.nan, shares) + for unit in units: + r = rho[unit["sex"], unit["band"]] + z = np.sqrt(r) * unit["eta"] + np.sqrt(1.0 - r) * unit["epsilon"] + drawn = np.zeros(len(unit["rows"])) + valid = unit["leaves"] >= 0 + u = ndtr(z) + p0 = unit["p_zero"] + valid = valid & (u >= p0) + v = (u[valid] - p0[valid]) / np.maximum(1.0 - p0[valid], 1e-12) + drawn[valid] = self._value(unit["leaves"][valid], v) + # A unit with no known neighbour is filled with zero, the + # assembler's treatment of a year it cannot fill. + out[unit["rows"], unit["column"]] = drawn + return out + + def _calibrate(self, shares, years, birth_year, sex, key, seed): + """Choose rho by sex and band to match masked-year persistence.""" + + from scipy.stats import spearmanr + + mask = np.zeros(shares.shape, dtype=bool) + columns = _column_of(years, np.asarray(_CALIBRATION_YEARS)) + mask[:, columns[columns >= 0]] = True + start = career_start(birth_year) + pre_career = years[None, :] < start[:, None] + mask &= ~pre_career + given = np.where(mask | pre_career, np.nan, shares) + units = self._units(given, years, birth_year, sex, key, mask, seed) + age = years[None, :] - birth_year[:, None] + band = _copula_band(age) + + def persistence(matrix): + out = {} + for s in (1, 2): + for b in range(1, 5): + values = [] + for lag in (2, 4): + pairs = [] + for year in _CALIBRATION_YEARS: + if year + lag not in _CALIBRATION_YEARS: + continue + c0 = year - years[0] + c1 = year + lag - years[0] + take = ( + (sex == s) + & (band[:, c0] == b) + & mask[:, c0] + & mask[:, c1] + ) + a = matrix[take, c0] + d = matrix[take, c1] + ok = (a > 0) & (d > 0) + if ok.sum() > 50: + pairs.append(spearmanr(a[ok], d[ok])[0]) + values.append(np.mean(pairs) if pairs else np.nan) + out[(s, b)] = values + return out + + truth = persistence(shares) + record = {} + rho = np.zeros((4, 6)) + best = {key_: (np.inf, 0.0) for key_ in truth} + for value in _RHO_GRID: + trial = np.full((4, 6), value) + filled = self._apply(units, given, mask, trial) + scores = persistence(filled) + for key_, (r2, r4) in scores.items(): + t2, t4 = truth[key_] + loss = abs(r2 - t2) + 0.5 * abs(r4 - t4) + if np.isfinite(loss) and loss < best[key_][0]: + best[key_] = (loss, value) + record[f"{key_[0]}.{key_[1]}.rho_{value}"] = [ + float(r2 - t2), + float(r4 - t4), + ] + for (s, b), (_, value) in best.items(): + rho[s, b] = value + # Bands outside the gate's take their neighbour's value. + rho[:, 0] = rho[:, 1] + rho[:, 5] = rho[:, 4] + return rho, record + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + units = self._units( + shares, + years, + birth_year, + sex, + np.asarray(person_key), + fill_mask, + seed, + ) + return self._apply(units, shares, fill_mask, self.rho) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{name: getattr(self, name) for name in _FOREST_ARRAYS}, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddForestFill: + return cls(**{name: arrays[name] for name in _FOREST_ARRAYS}) + + +def _forest_arrays(forest, x, y) -> dict[str, np.ndarray]: + """A fitted probability forest as arrays: nodes, and each leaf's mean y.""" + + offsets = [0] + lefts, rights, features, thresholds, leaf_index, means = ( + [], + [], + [], + [], + [], + [], + ) + leaf_count = 0 + for estimator in forest.estimators_: + tree = estimator.tree_ + left = tree.children_left.astype(np.int64) + right = tree.children_right.astype(np.int64) + is_leaf = left < 0 + n_leaves = int(is_leaf.sum()) + index = np.full(tree.node_count, -1, dtype=np.int64) + index[is_leaf] = leaf_count + np.arange(n_leaves) + nodes = _tree_leaves( + left, + right, + tree.feature.astype(np.int64), + tree.threshold.astype(np.float32), + x, + ) + local = index[nodes] - leaf_count + total = np.bincount(local, minlength=n_leaves) + hits = np.bincount(local, weights=y, minlength=n_leaves) + means.append(hits / np.maximum(total, 1)) + lefts.append(left) + rights.append(right) + features.append(tree.feature.astype(np.int16)) + thresholds.append(tree.threshold.astype(np.float32)) + leaf_index.append(index) + leaf_count += n_leaves + offsets.append(offsets[-1] + tree.node_count) + return { + "zero_tree_offsets": np.asarray(offsets, dtype=np.int64), + "zero_node_left": np.concatenate(lefts).astype(np.int32), + "zero_node_right": np.concatenate(rights).astype(np.int32), + "zero_node_feature": np.concatenate(features), + "zero_node_threshold": np.concatenate(thresholds), + "zero_node_leaf": np.concatenate(leaf_index).astype(np.int32), + "zero_leaf_p": np.concatenate(means).astype(np.float32), + } + + +_FOREST_ARRAYS = ( + "zero_tree_offsets", + "zero_node_left", + "zero_node_right", + "zero_node_feature", + "zero_node_threshold", + "zero_node_leaf", + "zero_leaf_p", + "tree_offsets", + "node_left", + "node_right", + "node_feature", + "node_threshold", + "node_leaf", + "leaf_offsets", + "leaf_values", + "rho", +) + + +# -------------------------------------------------------------------------- +# Odd years, alternative: kNN triples +# -------------------------------------------------------------------------- +_KNN_BANK = 40_000 +_JITTER = 1e-4 + + +@dataclass(frozen=True) +class OddKnnFill: + """The share at ``t`` copied from one of ``k`` nearest TRAIN units. + + Per sex and age band, a bank of up to 40,000 TRAIN person-years holds + the shares at ``t-1``, ``t``, ``t+1``. A masked unit's ``k`` nearest + bank units in (``t-1``, ``t+1``) are found after a deterministic jitter + of 1e-4 on both sides (so ties are broken at random), and one is chosen + by the seeded uniform. A missing neighbour takes the other's value. + """ + + bank_stratum: np.ndarray + bank_left: np.ndarray + bank_right: np.ndarray + bank_centre: np.ndarray + k: int = 10 + stream: str = "epuf_fill.odd_knn.v1" + name: str = "odd_knn" + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years, k=10, seed=0): + shares = np.asarray(shares, dtype=np.float64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + # Units inside the career; contexts see the career only. + start = career_start(np.asarray(birth_year)) + inside = unit_year >= start[rows] + rows, unit_year = rows[inside], unit_year[inside] + known = np.isfinite(shares) & ~( + np.asarray(years)[None, :] < start[:, None] + ) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + centre = _take(shares, rows, _column_of(years, unit_year)) + # A missing neighbour takes the other's value, as in the draw. + left = np.where(np.isnan(context.left), context.right, context.left) + right = np.where(np.isnan(context.right), context.left, context.right) + usable = np.isfinite(left) & np.isfinite(right) + stratum = context.sex * 16 + _age_band(context.age) + rng = np.random.default_rng(seed) + keep = [] + for value in np.unique(stratum[usable]): + members = np.flatnonzero((stratum == value) & usable) + if len(members) > _KNN_BANK: + members = rng.choice(members, _KNN_BANK, replace=False) + keep.append(np.sort(members)) + keep = np.concatenate(keep) + fill = cls( + bank_stratum=stratum[keep].astype(np.int64), + bank_left=left[keep].astype(np.float32), + bank_right=right[keep].astype(np.float32), + bank_centre=centre[keep].astype(np.float32), + k=k, + ) + return fill, {"n_units": int(len(centre)), "bank": int(len(keep))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + from scipy.spatial import cKDTree + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + known = np.isfinite(shares) & ~fill_mask + out = np.where(fill_mask, np.nan, shares) + bank_index = np.arange(len(self.bank_stratum)) + jitter_bank = ( + hash_uniform(self.stream + ".bank", 0, bank_index, 0) - 0.5, + hash_uniform(self.stream + ".bank", 1, bank_index, 0) - 0.5, + ) + trees = {} + for value in np.unique(self.bank_stratum): + members = np.flatnonzero(self.bank_stratum == value) + points = np.column_stack( + [ + self.bank_left[members] + + _JITTER * jitter_bank[0][members], + self.bank_right[members] + + _JITTER * jitter_bank[1][members], + ] + ) + trees[int(value)] = (cKDTree(points), members) + for column in np.flatnonzero(fill_mask.any(axis=0)): + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + unit_year = np.full(len(rows), year) + context = odd_context( + shares, years, rows, unit_year, birth_year, sex, known + ) + left = np.where( + np.isnan(context.left), context.right, context.left + ) + right = np.where( + np.isnan(context.right), context.left, context.right + ) + stratum = context.sex * 16 + _age_band(context.age) + u = hash_uniform(self.stream, seed, person_key[rows], unit_year) + jitter = ( + hash_uniform( + self.stream + ".q0", seed, person_key[rows], unit_year + ) + - 0.5, + hash_uniform( + self.stream + ".q1", seed, person_key[rows], unit_year + ) + - 0.5, + ) + drawn = np.full(len(rows), np.nan) + for value in np.unique(stratum): + take = (stratum == value) & np.isfinite(left) + if not take.any(): + continue + if int(value) not in trees: + trees[int(value)] = trees[self._nearest(int(value))] + tree, members = trees[int(value)] + query = np.column_stack( + [ + left[take] + _JITTER * jitter[0][take], + right[take] + _JITTER * jitter[1][take], + ] + ) + k = min(self.k, len(members)) + _, neighbours = tree.query(query, k=k) + neighbours = np.asarray(neighbours).reshape(len(query), k) + pick = np.minimum((u[take] * k).astype(np.int64), k - 1) + chosen = members[neighbours[np.arange(len(query)), pick]] + drawn[take] = self.bank_centre[chosen] + out[rows, column] = drawn + return out + + def _nearest(self, value: int) -> int: + strata = np.unique(self.bank_stratum) + same_sex = strata[strata // 16 == value // 16] + if len(same_sex) == 0: + same_sex = strata[strata // 16 == 1] + value = 16 + value % 16 + return int(same_sex[np.argmin(np.abs(same_sex - value))]) + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_stratum": self.bank_stratum, + "bank_left": self.bank_left, + "bank_right": self.bank_right, + "bank_centre": self.bank_centre, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> OddKnnFill: + return cls( + bank_stratum=arrays["bank_stratum"], + bank_left=arrays["bank_left"], + bank_right=arrays["bank_right"], + bank_centre=arrays["bank_centre"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, primary: rank-kNN donor careers +# -------------------------------------------------------------------------- +MATCH_YEARS = 5 +_DONOR_BANK = 2_000 + + +def _first_recorded( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Shares in the first MATCH_YEARS years from the career start.""" + + start = career_start(birth_year) + columns = _column_of( + years, start[:, None] + np.arange(MATCH_YEARS)[None, :] + ) + rows = np.repeat(np.arange(len(shares)), MATCH_YEARS).reshape( + len(shares), MATCH_YEARS + ) + return _take(shares, rows.ravel(), columns.ravel()).reshape( + len(shares), MATCH_YEARS + ) + + +#: Nearest-donor lists by input and bank content, reused across draw seeds. +_NEAREST_CACHE: dict = {} +#: Bank digests by fill (the fill is held so its id cannot be reused). +_BANK_DIGESTS: dict = {} +#: The match vector: the first MATCH_YEARS shares from the career start, +#: then the mean share and the share of positive years over every known +#: career year. +MATCH_DIMS = MATCH_YEARS + 2 +#: Odd years the PSID never records (1997 on); hidden when a bank's match +#: vectors are built, so they are built as a recipient's are. +_UNRECORDED_ODD_FROM = 1997 +#: EPUF's last year, the last year a donor bank records. +_BANK_LAST_YEAR = 2006 + + +def match_vector( + shares: np.ndarray, years: np.ndarray, birth_year: np.ndarray +) -> np.ndarray: + """Persons by MATCH_DIMS: the donor-match features; NaN where unknown.""" + + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + first = _first_recorded(shares, years, birth_year) + # The career summaries cover the bank's years (through 2006) only, so a + # PSID recipient's later years do not enter them. + career = (years[None, :] >= career_start(birth_year)[:, None]) & ( + years[None, :] <= _BANK_LAST_YEAR + ) + known = career & np.isfinite(shares) + count = known.sum(axis=1) + values = np.where(known, shares, 0.0) + mean = np.where( + count > 0, values.sum(axis=1) / np.maximum(count, 1), np.nan + ) + positive = np.where( + count > 0, + ((values > 0) & known).sum(axis=1) / np.maximum(count, 1), + np.nan, + ) + return np.column_stack([first, mean, positive]) + + +def _midrank(reference: np.ndarray, values: np.ndarray) -> np.ndarray: + """Percentile mid-rank of each value in a sorted reference sample.""" + + if len(reference) == 0: + return np.full(len(values), np.nan) + below = np.searchsorted(reference, values, side="left") + above = np.searchsorted(reference, values, side="right") + return (below + 0.5 * (above - below)) / len(reference) + + +def block_first_year(birth_year: np.ndarray) -> np.ndarray: + """First year a pre-career block can be positive in EPUF. + + 1951 for cohorts born by 1937; the year of age 15 for later cohorts, + whose earnings at 14 and under SSA zeroed. + """ + + birth_year = np.asarray(birth_year, dtype=np.int64) + return np.where(birth_year <= 1937, 1951, birth_year + FIRST_EARNING_AGE) + + +#: The widest block: 1951-1967. +BLOCK_WIDTH = CAREER_FIRST_YEAR - 1951 + + +@dataclass(frozen=True) +class PreDonorFill: + """Whole pre-career blocks copied from rank-matched TRAIN donors. + + Per sex and birth year, a bank of up to ``bank_size`` TRAIN donors + (those with a positive share from their career start through 2006, + chosen by the lowest hash of their person id; the registered fill keeps + them all) holds each donor's shares in the years from + :func:`block_first_year` to the year before the career start (at most + the 17 years 1951-1967, stored as shares times 65,535, rounded) and the + donor's match vector (:func:`match_vector`). The match vector has seven + features: the shares in the first five years from the career start, + and the mean share and share of positive years over the known career + years through 2006. + + A recipient's features are its percentile mid-ranks within the bank in + each feature it has. Distance is Euclidean over the features both have, + scaled by seven over their number. One of the ``k`` nearest donors is + chosen by the seeded uniform and its block copied; masked years before + :func:`block_first_year` are zero. A recipient with no feature takes a + donor chosen at random from its group. + """ + + bank_sex: np.ndarray + bank_birth_year: np.ndarray + bank_match: np.ndarray + bank_block: np.ndarray + k: int = 10 + stream: str = "epuf_fill.pre_donor.v1" + name: str = "pre_donor" + + @classmethod + def fit( + cls, + shares, + years, + birth_year, + sex, + person_key, + k=10, + birth_years=(1905, 1985), + bank_size=_DONOR_BANK, + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + start = career_start(birth_year) + recorded = years[None, :] >= start[:, None] + universe = ((shares > 0) & recorded).any(axis=1) & np.isin(sex, (1, 2)) + universe &= (birth_year >= birth_years[0]) & ( + birth_year <= birth_years[1] + ) + order_key = hash_uniform(cls.stream + ".bank", 0, person_key, 0) + chosen = [] + for s in (1, 2): + for b in np.unique(birth_year[universe & (sex == s)]): + members = np.flatnonzero( + universe & (sex == s) & (birth_year == b) + ) + members = members[np.argsort(order_key[members])][:bank_size] + chosen.append(np.sort(members)) + chosen = np.concatenate(chosen) + first = block_first_year(birth_year[chosen]) + offsets = np.arange(BLOCK_WIDTH) + block_years = first[:, None] + offsets[None, :] + inside = block_years < start[chosen][:, None] + columns = _column_of(years, block_years) + values = _take( + shares, + np.repeat(chosen, BLOCK_WIDTH), + columns.ravel(), + ).reshape(len(chosen), BLOCK_WIDTH) + values = np.where(inside, np.nan_to_num(values), 0.0) + hidden = (years[None, :] >= _UNRECORDED_ODD_FROM) & ( + years[None, :] % 2 == 1 + ) + fill = cls( + bank_sex=sex[chosen], + bank_birth_year=birth_year[chosen], + bank_match=match_vector( + np.where(hidden, np.nan, shares[chosen]), + years, + birth_year[chosen], + ).astype(np.float32), + bank_block=np.round( + np.clip(values, 0.0, 1.0) * _SHARE_SCALE + ).astype(np.uint16), + k=k, + ) + return fill, {"bank": int(len(chosen))} + + @property + def bank_digest(self) -> str: + """SHA-256 of the bank and k, the cache's key for this fill.""" + + cached = _BANK_DIGESTS.get(id(self)) + if cached is not None and cached[0] is self: + return cached[1] + digest = hashlib.sha256( + np.ascontiguousarray(self.bank_sex).tobytes() + + np.ascontiguousarray(self.bank_birth_year).tobytes() + + np.ascontiguousarray(self.bank_match).tobytes() + + np.ascontiguousarray(self.bank_block).tobytes() + + str(self.k).encode() + ).hexdigest() + _BANK_DIGESTS[id(self)] = (self, digest) + return digest + + def _nearest(self, match, birth_year, sex, targets): + """Each target's ``k`` nearest bank rows, and its group's bank rows. + + Seed-free, so it is computed once for a matrix and reused across + draw seeds (cached by the content of its inputs). + """ + + digest = hashlib.sha256( + np.ascontiguousarray(match[targets]).tobytes() + + np.ascontiguousarray(birth_year[targets]).tobytes() + + np.ascontiguousarray(sex[targets]).tobytes() + + np.ascontiguousarray(targets).tobytes() + + self.bank_digest.encode() + ).hexdigest() + if digest in _NEAREST_CACHE: + return _NEAREST_CACHE[digest] + nearest = np.full((len(targets), self.k), -1, dtype=np.int64) + group_first = np.full(len(targets), -1, dtype=np.int64) + group_size = np.zeros(len(targets), dtype=np.int64) + no_match = np.zeros(len(targets), dtype=bool) + for s, b in sorted( + set( + zip( + sex[targets].tolist(), + birth_year[targets].tolist(), + strict=True, + ) + ) + ): + local = np.flatnonzero( + (sex[targets] == s) & (birth_year[targets] == b) + ) + recipients = targets[local] + donors = np.flatnonzero( + (self.bank_sex == s) & (self.bank_birth_year == b) + ) + if len(donors) == 0: + continue + group_first[local] = donors[0] + group_size[local] = len(donors) + donor_match = self.bank_match[donors].astype(np.float64) + ranks_donor = np.empty_like(donor_match) + ranks_recipient = np.full((len(recipients), MATCH_DIMS), np.nan) + for j in range(MATCH_DIMS): + finite = np.isfinite(donor_match[:, j]) + column = np.sort(donor_match[finite, j]) + ranks_donor[:, j] = np.where( + finite, + _midrank(column, np.nan_to_num(donor_match[:, j])), + np.nan, + ) + values = match[recipients, j] + ok = np.isfinite(values) + ranks_recipient[ok, j] = _midrank(column, values[ok]) + k = min(self.k, len(donors)) + for start in range(0, len(recipients), 1_000): + block = slice(start, start + 1_000) + diff = ( + ranks_recipient[block][:, None, :] + - ranks_donor[None, :, :] + ) + available = np.isfinite(diff) + count = available.sum(axis=2) + distance = np.where(available, diff**2, 0.0).sum(axis=2) + distance = distance * MATCH_DIMS / np.maximum(count, 1) + order = np.argpartition(distance, k - 1, axis=1)[:, :k] + near_distance = np.take_along_axis(distance, order, 1) + ranked = np.lexsort((order, near_distance), axis=1) + order = np.take_along_axis(order, ranked, 1) + rows = local[block] + nearest[rows, :k] = donors[order] + no_match[rows] = count.max(axis=1) == 0 + result = (nearest, group_first, group_size, no_match) + if len(_NEAREST_CACHE) >= 4: + _NEAREST_CACHE.pop(next(iter(_NEAREST_CACHE))) + _NEAREST_CACHE[digest] = result + return result + + def donors( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + """The bank row each recipient (a row with a masked cell) copies. + + One of the ``k`` nearest bank donors of the recipient's sex and + birth year, chosen by the seeded uniform; a recipient with no + recorded match feature takes a random donor of its group. -1 for + rows with no masked cell or no bank donor of their group. + """ + + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + readable = np.where(fill_mask, np.nan, shares) + match = match_vector(readable, years, birth_year) + u = hash_uniform(self.stream, seed, person_key, 0) + targets = np.flatnonzero(fill_mask.any(axis=1)) + nearest, group_first, group_size, no_match = self._nearest( + match, birth_year, sex, targets + ) + out = np.full(len(shares), -1, dtype=np.int64) + has_group = group_size > 0 + k_available = (nearest >= 0).sum(axis=1) + pick = np.minimum( + (u[targets] * np.maximum(k_available, 1)).astype(np.int64), + np.maximum(k_available - 1, 0), + ) + chosen = nearest[np.arange(len(targets)), pick] + random_donor = group_first + np.minimum( + (u[targets] * np.maximum(group_size, 1)).astype(np.int64), + np.maximum(group_size - 1, 0), + ) + chosen = np.where(no_match, random_donor, chosen) + out[targets[has_group]] = chosen[has_group] + return out + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + donor = self.donors( + shares, years, birth_year, sex, person_key, fill_mask, seed + ) + rows = np.flatnonzero(donor >= 0) + first = block_first_year(birth_year[rows]) + block = self.bank_block[donor[rows]].astype(np.float64) / _SHARE_SCALE + for offset in range(BLOCK_WIDTH): + columns = _column_of(years, first + offset) + ok = columns >= 0 + target_rows = rows[ok] + target_columns = columns[ok] + masked = fill_mask[target_rows, target_columns] + out[target_rows[masked], target_columns[masked]] = block[ok][ + masked, offset + ] + # Masked years outside a donor block are zero, and so are those of a + # recipient with no bank of its sex and birth year (the current + # rule; the bank covers coded sex and births 1905-1985). + before = fill_mask & ( + years[None, :] < block_first_year(birth_year)[:, None] + ) + out[before] = 0.0 + out[fill_mask & (donor < 0)[:, None]] = 0.0 + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + "bank_sex": self.bank_sex, + "bank_birth_year": self.bank_birth_year, + "bank_match": self.bank_match, + "bank_block": self.bank_block, + "k": np.array(self.k), + } + ) + + @classmethod + def from_arrays(cls, arrays) -> PreDonorFill: + return cls( + bank_sex=arrays["bank_sex"], + bank_birth_year=arrays["bank_birth_year"], + bank_match=arrays["bank_match"], + bank_block=arrays["bank_block"], + k=int(arrays["k"]), + ) + + +# -------------------------------------------------------------------------- +# Pre-career, alternative: the chained one-sided draw +# -------------------------------------------------------------------------- +def _chain_age(age: np.ndarray) -> np.ndarray: + """0 below 15; single years 15-24 as 1-10; then five-year bands.""" + + age = np.asarray(age, dtype=np.int64) + return np.where( + age < 15, + 0, + np.where(age <= 24, age - 14, np.minimum((age - 25) // 5 + 11, 22)), + ) + + +@dataclass(frozen=True) +class PreChainFill: + """Year ``y`` drawn from year ``y+1``, sex and age, backward to 1951. + + Cells are the finest of (sex, age (single years 15-24, then five-year + bands), bin of the next known share), + (sex, bin), (bin) with at least ``MIN_CELL`` TRAIN units; in a cell, + ``p0`` and 65 quantiles of ``log(x_y / x_{y+1})`` (of ``log x_y`` when + ``x_{y+1}`` is zero). Each year's uniform is independent. + """ + + level_edges: np.ndarray + level_keys: tuple[np.ndarray, ...] + level_p0: tuple[np.ndarray, ...] + level_quantiles: tuple[np.ndarray, ...] + stream: str = "epuf_fill.pre_chain.v1" + name: str = "pre_chain" + + @staticmethod + def _keys(sex, age, following, edges): + bins = np.where( + following <= 0, + 0, + np.where( + following >= 1.0, + len(edges) + 2, + np.searchsorted(edges, following, side="right") + 1, + ), + ) + band = _chain_age(age) + + def key(s, a, b): + return (s * 40 + a) * 32 + b + + return [ + key(sex, band, bins), + key(sex, 39, bins), + key(0 * sex, 39, bins), + ] + + @classmethod + def fit(cls, shares, years, birth_year, sex, unit_years): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + n = len(shares) + rows = np.concatenate([np.arange(n) for _ in unit_years]) + unit_year = np.concatenate([np.full(n, y) for y in unit_years]) + target = _take(shares, rows, _column_of(years, unit_year)) + following = _take(shares, rows, _column_of(years, unit_year + 1)) + sex_u = np.asarray(sex)[rows].astype(np.int64) + age = unit_year - np.asarray(birth_year)[rows] + inside = following[(following > 0) & (following < 1.0)] + edges = np.quantile(inside, np.linspace(0, 1, 21)[1:-1]) + keys = cls._keys(sex_u, age, following, edges) + positive = target > 0 + base = np.where(following > 0, following, 1.0) + residual = np.log(np.where(positive, target, 1.0)) - np.log(base) + level_keys, level_p0, level_quantiles = [], [], [] + for key in keys: + unique, count = np.unique(key, return_counts=True) + populated = unique[count >= MIN_CELL] + in_cells = np.isin(key, populated) + zu, zc = np.unique(key[in_cells & ~positive], return_counts=True) + p0 = np.zeros(len(populated)) + p0[np.searchsorted(populated, zu)] = zc + p0 = p0 / count[count >= MIN_CELL] + q_keys, _, table = _quantile_table( + key[in_cells & positive], residual[in_cells & positive] + ) + full = np.zeros((len(populated), QUANTILE_POINTS), np.float32) + full[np.searchsorted(populated, q_keys)] = table + level_keys.append(populated.astype(np.int64)) + level_p0.append(p0) + level_quantiles.append(full) + fill = cls( + level_edges=edges, + level_keys=tuple(level_keys), + level_p0=tuple(level_p0), + level_quantiles=tuple(level_quantiles), + ) + return fill, {"n_units": int(len(target))} + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + years = np.asarray(years, dtype=np.int64) + birth_year = np.asarray(birth_year, dtype=np.int64) + sex = np.asarray(sex, dtype=np.int64) + out = np.where(fill_mask, np.nan, shares) + for column in np.flatnonzero(fill_mask.any(axis=0))[::-1]: + year = int(years[column]) + rows = np.flatnonzero(fill_mask[:, column]) + # The next known (or already drawn) later year's share. + later = out[rows, column + 1 :] + if later.shape[1]: + finite = np.isfinite(later) + first = np.argmax(finite, axis=1) + following = np.where( + finite.any(axis=1), + later[np.arange(len(rows)), first], + np.nan, + ) + else: + following = np.full(len(rows), np.nan) + # With no known later year (a career starting after the file's + # last year), the chain starts from a zero year. + following = np.nan_to_num(following, nan=0.0) + age = year - birth_year[rows] + keys = self._keys( + sex[rows], age, np.nan_to_num(following), self.level_edges + ) + u = hash_uniform(self.stream, seed, person_key[rows], year) + level = np.full(len(rows), -1) + row = np.full(len(rows), -1) + for index, key in enumerate(keys): + found = _lookup(self.level_keys[index], key) + take = (level < 0) & (found >= 0) + level[take] = index + row[take] = found[take] + drawn = np.full(len(rows), np.nan) + for index in np.unique(level[level >= 0]): + take = level == index + p0 = self.level_p0[index][row[take]] + positive = u[take] >= p0 + v = np.where( + positive, (u[take] - p0) / np.maximum(1.0 - p0, 1e-12), 0.0 + ) + residual = _interpolate( + self.level_quantiles[index], row[take], v + ) + base = np.where(following[take] > 0, following[take], 1.0) + drawn[take] = np.where( + positive, np.minimum(base * np.exp(residual), 1.0), 0.0 + ) + # EPUF has no earnings below age 15. + drawn = np.where(age < FIRST_EARNING_AGE, 0.0, drawn) + out[rows, column] = drawn + return out + + def to_bytes(self) -> bytes: + arrays = {"kind": np.array(self.name), "level_edges": self.level_edges} + for index, (k, p, q) in enumerate( + zip( + self.level_keys, + self.level_p0, + self.level_quantiles, + strict=True, + ) + ): + arrays[f"keys_{index}"] = k + arrays[f"p0_{index}"] = p + arrays[f"quantiles_{index}"] = q + return _to_npz(arrays) + + @classmethod + def from_arrays(cls, arrays) -> PreChainFill: + n_levels = sum(1 for name in arrays.files if name.startswith("keys_")) + return cls( + level_edges=arrays["level_edges"], + level_keys=tuple(arrays[f"keys_{i}"] for i in range(n_levels)), + level_p0=tuple(arrays[f"p0_{i}"] for i in range(n_levels)), + level_quantiles=tuple( + arrays[f"quantiles_{i}"] for i in range(n_levels) + ), + ) + + +@dataclass(frozen=True) +class BySexFill: + """One fill per coded sex; persons of uncoded sex use the men's. + + Each part is any fill of this module, fitted on TRAIN persons of that + sex only, and fills only rows of that sex. + """ + + parts: dict + name: str = "by_sex" + + @classmethod + def fit(cls, fill_class, shares, years, birth_year, sex, *args, **kwargs): + parts, diagnostics = {}, {} + sex = np.asarray(sex) + for value in (1, 2): + rows = sex == value + extra = [ + ( + a[rows] + if isinstance(a, np.ndarray) and len(a) == len(sex) + else a + ) + for a in args + ] + part, diagnostic = fill_class.fit( + shares[rows], + years, + birth_year[rows], + sex[rows], + *extra, + **kwargs, + ) + parts[value] = part + diagnostics[str(value)] = diagnostic + return cls(parts=parts), diagnostics + + def fill( + self, shares, years, birth_year, sex, person_key, fill_mask, seed + ): + shares = np.asarray(shares, dtype=np.float64) + sex = np.asarray(sex) + out = np.where(fill_mask, np.nan, shares) + for value, part in self.parts.items(): + rows = np.flatnonzero( + (sex == value) | ((value == 1) & ~np.isin(sex, (1, 2))) + ) + if not len(rows): + continue + out[rows] = part.fill( + shares[rows], + years, + np.asarray(birth_year)[rows], + # Rows of uncoded sex take this part's sex, so its copula + # (calibrated for that sex) applies to them. + np.full(len(rows), value), + np.asarray(person_key)[rows], + fill_mask[rows], + seed, + ) + return out + + def to_bytes(self) -> bytes: + return _to_npz( + { + "kind": np.array(self.name), + **{ + f"part_{value}": np.frombuffer(part.to_bytes(), np.uint8) + for value, part in self.parts.items() + }, + } + ) + + @classmethod + def from_arrays(cls, arrays) -> BySexFill: + parts = {} + for name in arrays.files: + if name.startswith("part_"): + with np.load( + io.BytesIO(arrays[name].tobytes()), allow_pickle=False + ) as nested: + kind = str(nested["kind"]) + parts[int(name[5:])] = FILL_CLASSES[kind].from_arrays( + nested + ) + return cls(parts=parts) + + +FILL_CLASSES = { + "by_sex": BySexFill, + "odd_forest": OddForestFill, + "odd_knn": OddKnnFill, + "pre_donor": PreDonorFill, + "pre_chain": PreChainFill, +} + + +def load_fill(path: Path, *, sha256: str | None = None): + """Load a fitted fill from its ``.npz``; refuse other bytes than ``sha256``.""" + + data = Path(path).read_bytes() + if sha256 is not None: + observed = hashlib.sha256(data).hexdigest() + if observed != sha256: + raise ValueError( + f"{path} has SHA-256 {observed}, not the registered {sha256}" + ) + with np.load(io.BytesIO(data), allow_pickle=False) as arrays: + kind = str(arrays["kind"]) + return FILL_CLASSES[kind].from_arrays(arrays) diff --git a/tests/README-tiers.md b/tests/README-tiers.md index f9fdebc5..18574bca 100644 --- a/tests/README-tiers.md +++ b/tests/README-tiers.md @@ -39,9 +39,9 @@ pytest --collect-only -q -m oracle_policyengine | tail -1 | Tier | Tests at HEAD | |---|---:| -| `unit` | 6,017 | -| `artifact` | 3,388 | +| `unit` | 6,037 | +| `artifact` | 3,398 | | `integration_psid` | 1,341 | | `reproduction_legacy` | 520 | | `oracle_policyengine` | 220 | -| **Total** | **11,486** | +| **Total** | **11,516** | diff --git a/tests/cohorts/test_psid2010_epuf_fill.py b/tests/cohorts/test_psid2010_epuf_fill.py new file mode 100644 index 00000000..d8734376 --- /dev/null +++ b/tests/cohorts/test_psid2010_epuf_fill.py @@ -0,0 +1,224 @@ +"""Learned EPUF fills applied to a PSID-2010-shaped cohort (synthetic).""" + +from __future__ import annotations + +from types import SimpleNamespace + +import numpy as np +import pandas as pd +import pytest + +from populace_dynamics.cohorts import psid2010_epuf_fill as fill_module +from populace_dynamics.harness import epuf_fill_gate as g + +BIRTHS = {1: (1950, "male"), 2: (1960, "female"), 3: (1975, "male")} + + +def _cohort(): + caps = fill_module._wage_bases(np.arange(1951, 2011)) + rows = [] + for pid, (birth, _) in BIRTHS.items(): + start = max(1968, birth + 22) + for year in range(start, 2011): + cap = caps[year - 1951] + if year >= 1997 and year % 2 == 1: + continue + rows.append((pid, year, 0.3 * cap * (1 + 0.01 * pid), "observed")) + for year in range(max(start, 1997), 2011): + if year % 2 == 1: + left = [r for r in rows if r[0] == pid and r[1] == year - 1] + right = [r for r in rows if r[0] == pid and r[1] == year + 1] + if left and right: + mean = (left[0][2] + right[0][2]) / 2 + else: + mean = (left or right)[0][2] + rows.append((pid, year, mean, "gap_imputed")) + careers = pd.DataFrame( + rows, columns=["person_id", "year", "earnings", "provenance"] + ).sort_values(["person_id", "year"]) + persons = pd.DataFrame( + { + "person_id": list(BIRTHS), + "birth_year": [b for b, _ in BIRTHS.values()], + "sex": [s for _, s in BIRTHS.values()], + } + ) + return SimpleNamespace(persons=persons, careers=careers) + + +class _MeanFill: + """The assembler's neighbour mean, in dollars, on SSA's wage bases.""" + + name = "neighbour_mean" + + def fill(self, shares, years, birth, sex, key, mask, seed): + caps = fill_module._wage_bases(np.asarray(years)) + dollars = shares * caps[None, :] + out = shares.copy() + for column in np.flatnonzero(mask.any(axis=0)): + left = dollars[:, column - 1] + right = dollars[:, column + 1] + mean = np.where( + np.isnan(left), + right, + np.where(np.isnan(right), left, (left + right) / 2), + ) + rows = mask[:, column] + out[rows, column] = np.minimum( + np.nan_to_num(mean[rows]) / caps[column], 1.0 + ) + return out + + +def test_current_rule_fills_reproduce_the_assembler(): + cohort = _cohort() + result = fill_module.fill_careers( + cohort, + odd_fill=_MeanFill(), + pre_fill=g.CurrentPreFill(), + seed=7100, + odd_years=None, + ) + careers = result.careers + before = cohort.careers.set_index(["person_id", "year"]) + after = careers.set_index(["person_id", "year"]) + gap = before["provenance"] == "gap_imputed" + # Neighbours below the cap: the learned path's neighbour mean (in + # dollars, capped) is the assembler's mean. + np.testing.assert_allclose( + after.loc[before.index[gap], "earnings"], + before.loc[gap, "earnings"], + rtol=1e-9, + ) + assert ( + after.loc[before.index[gap], "provenance"] + == fill_module.EPUFFillProvenance.GAP_EPUF_DRAWN.value + ).all() + observed = before["provenance"] == "observed" + pd.testing.assert_series_equal( + after.loc[before.index[observed], "earnings"], + before.loc[observed, "earnings"], + check_names=False, + ) + pre = careers[ + careers["provenance"] + == fill_module.EPUFFillProvenance.PRE_CAREER_EPUF_DONOR.value + ] + for pid, (birth, _) in BIRTHS.items(): + years = pre.loc[pre["person_id"] == pid, "year"].tolist() + assert years == list(range(1951, max(1968, birth + 22))) + assert (pre["earnings"] == 0).all() + assert result.fills == { + "odd": "neighbour_mean", + "pre": "current_pre_career_rule", + } + + +def test_fills_see_no_pre_career_or_gap_year(): + cohort = _cohort() + seen = {} + + class Spy: + name = "spy" + + def fill(self, shares, years, birth, sex, key, mask, seed): + seen["shares"] = shares.copy() + seen["mask"] = mask.copy() + out = shares.copy() + out[mask] = 0.25 + return out + + result = fill_module.fill_careers( + cohort, odd_fill=Spy(), seed=1, odd_years=None + ) + years = np.arange(1951, 2011) + birth = np.array([b for b, _ in BIRTHS.values()]) + pre = years[None, :] < np.maximum(1968, birth + 22)[:, None] + assert np.isnan(seen["shares"][pre]).all() + assert np.isnan(seen["shares"][seen["mask"]]).all() + assert (seen["shares"][~pre & ~seen["mask"]] <= 1).all() + drawn = result.careers[result.careers["provenance"] == "gap_epuf_drawn"] + caps = fill_module._wage_bases(drawn["year"].to_numpy()) + np.testing.assert_allclose(drawn["earnings"], 0.25 * caps) + assert len(result.content_sha256) == 64 + + +def test_an_invalid_fill_is_refused(): + class Bad: + def fill(self, shares, years, birth, sex, key, mask, seed): + out = shares.copy() + out[mask] = 1.5 + return out + + with pytest.raises(ValueError, match="invalid share"): + fill_module.fill_careers(_cohort(), odd_fill=Bad(), seed=1) + + +def test_no_fill_keeps_the_careers(): + cohort = _cohort() + result = fill_module.fill_careers(cohort, seed=1) + assert len(result.careers) == len(cohort.careers) + assert result.fills == {} + + +def test_by_default_only_the_scored_gap_years_are_filled(): + cohort = _cohort() + + class Constant: + name = "constant" + + def fill(self, shares, years, birth, sex, key, mask, seed): + out = shares.copy() + out[mask] = 0.25 + return out + + result = fill_module.fill_careers(cohort, odd_fill=Constant(), seed=1) + after = result.careers.set_index(["person_id", "year"]) + before = cohort.careers.set_index(["person_id", "year"]) + gaps = before.index[before["provenance"] == "gap_imputed"] + for key in gaps: + if key[1] in fill_module.SCORED_ODD_YEARS: + assert after.loc[key, "provenance"] == "gap_epuf_drawn" + else: + # 2007 and 2009 keep the assembler's value and provenance. + assert after.loc[key, "provenance"] == "gap_imputed" + assert after.loc[key, "earnings"] == before.loc[key, "earnings"] + + +def test_a_gap_with_no_visible_neighbour_keeps_the_assemblers_value(): + cohort = _cohort() + careers = cohort.careers + # Person 3 (born 1975) starts in 1997; drop 1998 so 1997's only + # neighbours are pre-career (1996) or missing. + careers = careers[~((careers.person_id == 3) & (careers.year == 1998))] + # Person 1 gets a 2013 seam filled from a 2014 boundary year, with no + # 2012 row: neither neighbour is visible to a fill. + extra = pd.DataFrame( + [ + (1, 2013, 30_000.0, "gap_imputed"), + (1, 2014, 30_000.0, "boundary_2014"), + ], + columns=careers.columns, + ) + cohort = SimpleNamespace( + persons=cohort.persons, + careers=pd.concat([careers, extra], ignore_index=True), + ) + + class Constant: + name = "constant" + + def fill(self, shares, years, birth, sex, key, mask, seed): + out = shares.copy() + out[mask] = 0.25 + return out + + result = fill_module.fill_careers( + cohort, odd_fill=Constant(), seed=1, odd_years=None + ) + after = result.careers.set_index(["person_id", "year"]) + assert after.loc[(3, 1997), "provenance"] == "gap_imputed" + assert after.loc[(1, 2013), "provenance"] == "gap_imputed" + assert after.loc[(1, 2013), "earnings"] == 30_000.0 + assert after.loc[(1, 2014), "provenance"] == "boundary_2014" + assert after.loc[(1, 1999), "provenance"] == "gap_epuf_drawn" diff --git a/tests/estimates/test_birth_evidence_artifact.py b/tests/estimates/test_birth_evidence_artifact.py index 27151dff..fc475dd0 100644 --- a/tests/estimates/test_birth_evidence_artifact.py +++ b/tests/estimates/test_birth_evidence_artifact.py @@ -216,6 +216,8 @@ def test_post_review_sources_are_outside_historical_reducer_identity(): Path("src/populace_dynamics/harness/epuf_run.py"), Path("src/populace_dynamics/harness/epuf_fill_gate.py"), Path("src/populace_dynamics/harness/epuf_fill_scoring.py"), + Path("src/populace_dynamics/estimates/epuf_fill.py"), + Path("src/populace_dynamics/cohorts/psid2010_epuf_fill.py"), ) assert reducer.POST_REVIEW_SHARED_SOURCE_BLOBS == { Path( @@ -473,6 +475,8 @@ def test_post_review_exclusions_are_unreachable_from_birth_evidence(): "populace_dynamics.harness.epuf_run", "populace_dynamics.harness.epuf_fill_gate", "populace_dynamics.harness.epuf_fill_scoring", + "populace_dynamics.estimates.epuf_fill", + "populace_dynamics.cohorts.psid2010_epuf_fill", } assert epuf_modules.issubset(module_paths) assert epuf_modules.isdisjoint(reachable), ( diff --git a/tests/estimates/test_coordinator.py b/tests/estimates/test_coordinator.py index da1ef010..e23f211a 100644 --- a/tests/estimates/test_coordinator.py +++ b/tests/estimates/test_coordinator.py @@ -1538,12 +1538,19 @@ def test__estimator_surface__pins_complete_module_tuple(): for path in observed if path.name in ("adjusted_poverty.py", "uniform_cut_tabulation.py") ) + # The EPUF-learned career fills (gate_epuf_fill) are opt-in and outside + # the registered first-estimates surface: the gate's scoring and its fit + # script load them, and nothing on the first-estimates path imports them. + epuf_fill_surface = tuple( + path for path in observed if path.name == "epuf_fill.py" + ) first_estimates_surface = tuple( path for path in observed if path not in context_surface and path not in tabulation_surface and path not in track_u_surface + and path not in epuf_fill_surface ) assert coordinator._ESTIMATOR_SURFACE_SOURCES == expected @@ -1555,6 +1562,9 @@ def test__estimator_surface__pins_complete_module_tuple(): Path("src/populace_dynamics/estimates/adjusted_poverty.py"), Path("src/populace_dynamics/estimates/uniform_cut_tabulation.py"), ) + assert epuf_fill_surface == ( + Path("src/populace_dynamics/estimates/epuf_fill.py"), + ) assert context_surface == ( Path("src/populace_dynamics/estimates/anchor_context_coordinator.py"), Path("src/populace_dynamics/estimates/anchor_context_publication.py"), diff --git a/tests/estimates/test_epuf_fill.py b/tests/estimates/test_epuf_fill.py new file mode 100644 index 00000000..747953bd --- /dev/null +++ b/tests/estimates/test_epuf_fill.py @@ -0,0 +1,200 @@ +"""The learned EPUF career fills, on synthetic careers.""" + +from __future__ import annotations + +import hashlib + +import numpy as np +import pytest + +from populace_dynamics.estimates import epuf_fill as F +from populace_dynamics.harness import epuf_fill_gate as g + +YEARS = np.asarray(g.YEARS) + + +def _shares(seed: int, n: int = 3_000): + rng = np.random.default_rng(seed) + birth = rng.integers(1925, 1981, size=n) + sex = rng.choice([1, 2], size=n) + level = rng.normal(-1.3, 0.7, size=n) + walk = rng.normal(0, 0.25, size=(n, len(YEARS))).cumsum(axis=1) * 0.3 + shares = np.minimum(np.exp(level[:, None] + walk), 1.0) + work = rng.random((n, len(YEARS))) < 0.85 + age = YEARS[None, :] - birth[:, None] + shares = np.where(work & (age >= 15) & (age <= 85), shares, 0.0) + return np.round(shares, 6), birth, sex, np.arange(n) + 1 + + +def test_hash_uniform_is_keyed_and_order_free(): + keys = np.arange(1, 10_001) + u = F.hash_uniform("s", 7, keys, 1999) + assert ((u > 0) & (u < 1)).all() + assert abs(u.mean() - 0.5) < 0.02 + np.testing.assert_array_equal( + F.hash_uniform("s", 7, keys[::-1], 1999)[::-1], u + ) + assert not np.array_equal(u, F.hash_uniform("s", 8, keys, 1999)) + assert not np.array_equal(u, F.hash_uniform("t", 7, keys, 1999)) + assert not np.array_equal(u, F.hash_uniform("s", 7, keys, 2001)) + + +@pytest.fixture(scope="module") +def fitted(): + shares, birth, sex, key = _shares(1) + unit_years = tuple(range(1991, 2006)) + return { + "odd_forest": F.BySexFill.fit( + F.OddForestFill, + shares, + YEARS, + birth, + sex, + unit_years, + key, + n_units=40_000, + n_trees=3, + min_leaf=10, + n_jobs=1, + )[0], + "odd_knn": F.OddKnnFill.fit(shares, YEARS, birth, sex, unit_years)[0], + "pre_donor": F.PreDonorFill.fit( + shares, YEARS, birth, sex, key, k=3, bank_size=500 + )[0], + "pre_chain": F.PreChainFill.fit( + shares, YEARS, birth, sex, tuple(range(1951, 2006)) + )[0], + } + + +def _given(seed=2, n=800): + shares, birth, sex, key = _shares(seed, n) + given = np.where(g.union_mask(birth), np.nan, shares) + return given, birth, sex, key + 10_000_000 + + +@pytest.mark.parametrize( + ("name", "family"), + [ + ("odd_forest", "odd"), + ("odd_knn", "odd"), + ("pre_donor", "pre"), + ("pre_chain", "pre"), + ], +) +def test_fills_obey_the_scoring_contract(fitted, name, family): + fill = fitted[name] + given, birth, sex, key = _given() + mask = g.family_mask(family, birth) + out = fill.fill(given.copy(), YEARS, birth, sex, key, mask.copy(), 7100) + assert np.isfinite(out[mask]).all() + assert ((out[mask] >= 0) & (out[mask] <= 1)).all() + same = (out[~mask] == given[~mask]) | ( + np.isnan(out[~mask]) & np.isnan(given[~mask]) + ) + assert same.all() + again = fill.fill(given.copy(), YEARS, birth, sex, key, mask.copy(), 7100) + np.testing.assert_array_equal(np.nan_to_num(out), np.nan_to_num(again)) + other = fill.fill(given.copy(), YEARS, birth, sex, key, mask.copy(), 7101) + assert not np.array_equal(np.nan_to_num(out), np.nan_to_num(other)) + # A person's draw does not depend on the other persons' order. + order = np.random.default_rng(0).permutation(len(birth)) + permuted = fill.fill( + given[order].copy(), + YEARS, + birth[order], + sex[order], + key[order], + mask[order].copy(), + 7100, + ) + np.testing.assert_allclose( + np.nan_to_num(permuted), np.nan_to_num(out[order]) + ) + + +@pytest.mark.parametrize( + "name", ["odd_forest", "odd_knn", "pre_donor", "pre_chain"] +) +def test_artifacts_round_trip_with_reproducible_bytes(fitted, name, tmp_path): + fill = fitted[name] + blob = fill.to_bytes() + assert blob == fill.to_bytes() + path = tmp_path / f"{name}.npz" + path.write_bytes(blob) + sha = hashlib.sha256(blob).hexdigest() + loaded = F.load_fill(path, sha256=sha) + assert type(loaded) is type(fill) + assert loaded.to_bytes() == blob + with pytest.raises(ValueError, match="SHA-256"): + F.load_fill(path, sha256="0" * 64) + + +def test_donor_blocks_come_from_the_bank(fitted): + fill = fitted["pre_donor"] + given, birth, sex, key = _given(3) + mask = g.family_mask("pre", birth) + out = fill.fill(given.copy(), YEARS, birth, sex, key, mask, 7100) + donors = fill.donors(given, YEARS, birth, sex, key, mask, 7100) + for row in np.flatnonzero(donors >= 0)[:50]: + donor = donors[row] + assert fill.bank_birth_year[donor] == birth[row] + assert fill.bank_sex[donor] == sex[row] + first = int(F.block_first_year(birth[row : row + 1])[0]) + block = fill.bank_block[donor].astype(float) / 65_535 + for offset in range(F.BLOCK_WIDTH): + year = first + offset + if year in YEARS and mask[row, year - YEARS[0]]: + assert out[row, year - YEARS[0]] == pytest.approx( + block[offset] + ) + early = (YEARS < first) & mask[row] + assert (out[row, early] == 0).all() + + +def test_the_forest_copula_is_calibrated_by_band(fitted): + for part in fitted["odd_forest"].parts.values(): + rho = part.rho + assert rho.shape == (4, 6) + assert ((rho >= 0) & (rho <= 0.9)).all() + + +def test_the_donor_cache_is_keyed_by_the_bank_not_the_object(): + shares, birth, sex, key = _shares(4, n=2_000) + first = F.PreDonorFill.fit( + shares, YEARS, birth, sex, key, k=3, bank_size=5 + )[0] + second = F.PreDonorFill.fit( + shares, YEARS, birth, sex, key + 1, k=3, bank_size=5 + )[0] + assert first.bank_digest != second.bank_digest + given, b, s, k = _given(5) + mask = g.family_mask("pre", b) + a = first.donors(given, YEARS, b, s, k, mask, 7100) + c = second.donors(given, YEARS, b, s, k, mask, 7100) + # Each fill's donors index its own bank and match its recipients' group. + for fill, chosen in ((first, a), (second, c)): + rows = np.flatnonzero(chosen >= 0) + assert (chosen[rows] < len(fill.bank_sex)).all() + assert (fill.bank_birth_year[chosen[rows]] == b[rows]).all() + + +def test_uncoded_sex_takes_the_routed_parts_sex(fitted): + fill = fitted["odd_forest"] + given, birth, sex, key = _given(6, n=400) + mask = g.family_mask("odd", birth) + uncoded = sex.copy() + uncoded[::3] = 3 + as_men = sex.copy() + as_men[::3] = 1 + out = fill.fill(given.copy(), YEARS, birth, uncoded, key, mask, 7100) + expected = fill.fill(given.copy(), YEARS, birth, as_men, key, mask, 7100) + rows = np.arange(len(sex))[::3] + np.testing.assert_array_equal( + np.nan_to_num(out[rows]), np.nan_to_num(expected[rows]) + ) + + +def test_fitted_forests_have_no_empty_leaf(fitted): + for part in fitted["odd_forest"].parts.values(): + assert (np.diff(part.leaf_offsets) > 0).all() diff --git a/tests/test_epuf_fill_candidates_manifest.py b/tests/test_epuf_fill_candidates_manifest.py new file mode 100644 index 00000000..7dbf7c83 --- /dev/null +++ b/tests/test_epuf_fill_candidates_manifest.py @@ -0,0 +1,174 @@ +"""The registered candidate manifest of gate_epuf_fill. + +``runs/epuf_fill_candidates_v1.json`` names the four registered fills by +SHA-256. Their bytes live outside the repository (``~/PolicyEngine/ +epuf-data/fills``), so the hash checks of the staged files skip where they +are not staged. +""" + +from __future__ import annotations + +import hashlib +import importlib.util +import json +import os +import subprocess +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +MANIFEST = ROOT / "runs" / "epuf_fill_candidates_v1.json" +FILLS_DIR = Path( + os.environ.get( + "POPULACE_DYNAMICS_EPUF_FILLS_DIR", "~/PolicyEngine/epuf-data/fills" + ) +).expanduser() + + +@pytest.fixture(scope="module") +def manifest(): + return json.loads(MANIFEST.read_text()) + + +def _script(): + spec = importlib.util.spec_from_file_location( + "fit_epuf_fills", ROOT / "scripts" / "fit_epuf_fills.py" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_the_manifest_registers_one_primary_and_one_alternative_per_family( + manifest, +): + assert manifest["schema"] == "populace_dynamics.epuf_fill_candidates.v1" + assert manifest["registration_id"] == "2026-10-03-epuf-career-fill" + assert manifest["part"] == "train" + assert manifest["code_files_clean"] is True + roles = sorted( + (record["family"], record["role"]) + for record in manifest["fills"].values() + ) + assert roles == [ + ("odd", "alternative"), + ("odd", "primary"), + ("pre", "alternative"), + ("pre", "primary"), + ] + for record in manifest["fills"].values(): + assert len(record["sha256"]) == 64 and record["bytes"] > 0 + + +def test_the_manifest_parameters_are_the_scripts(manifest): + registered = _script().REGISTERED + assert set(registered) == set(manifest["fills"]) + for name, record in manifest["fills"].items(): + assert record["params"] == registered[name]["params"] + assert record["family"] == registered[name]["family"] + assert record["role"] == registered[name]["role"] + + +def test_the_fill_code_is_unchanged_since_the_fit(manifest): + commit = manifest["code_commit"] + probe = subprocess.run( + ["git", "-C", str(ROOT), "cat-file", "-e", f"{commit}^{{commit}}"], + capture_output=True, + ) + if probe.returncode != 0: + pytest.skip("the fit's commit is not in this clone") + for path in _script().CODE_FILES: + built = subprocess.run( + ["git", "-C", str(ROOT), "show", f"{commit}:{path}"], + capture_output=True, + check=True, + ).stdout + assert built == (ROOT / path).read_bytes(), path + + +@pytest.mark.parametrize( + "name", ["odd_forest", "odd_knn", "pre_donor", "pre_chain"] +) +def test_staged_fills_hash_to_the_manifest(manifest, name): + record = manifest["fills"][name] + path = FILLS_DIR / record["file"] + if not path.is_file(): + pytest.skip(f"{path} is not staged") + assert hashlib.sha256(path.read_bytes()).hexdigest() == record["sha256"] + + +def test_the_registered_copula_binds_for_each_coded_sex(manifest): + diagnostics = manifest["fills"]["odd_forest"]["diagnostics"] + for sex in ("1", "2"): + rho = diagnostics[sex]["rho"] + assert any(value > 0 for value in rho[int(sex)]) + + +def _score_script(): + spec = importlib.util.spec_from_file_location( + "score_epuf_fill_test", ROOT / "scripts" / "score_epuf_fill_test.py" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_the_score_script_pins_this_manifest(): + script = _score_script() + assert script.REGISTERED_MANIFEST == "runs/epuf_fill_candidates_v1.json" + assert ( + script.REGISTERED_MANIFEST_SHA256 + == hashlib.sha256(MANIFEST.read_bytes()).hexdigest() + ) + + +def test_the_score_script_refuses_before_reading_test(tmp_path, monkeypatch): + """No path here can reach TEST, even after the gate locks. + + The lock status is patched, and the fills folder is an empty or a + deliberately wrong one, so ``test_part`` is never called. + """ + + script = _score_script() + output = tmp_path / "result.json" + empty = tmp_path / "fills" + empty.mkdir() + other = tmp_path / "other.json" + other.write_text("{}") + base = ["--output", str(output), "--fills-dir", str(empty)] + with pytest.raises(ValueError, match="not the registered"): + script.main(["--manifest", str(other), *base]) + + def reached_test(**_): + raise AssertionError("the score script reached test_part") + + monkeypatch.setattr(script.scoring.g, "test_part", reached_test) + monkeypatch.setattr( + script.g, + "_gate_lock_status", + lambda _: {"locked": False, "registration_id": None}, + ) + with pytest.raises(script.g.TestPartLocked): + script.main(["--manifest", str(MANIFEST), *base]) + assert not output.exists() + assert not tmp_path.joinpath("result.json.started.json").exists() + # Locked, but a staged file is not the registered bytes: refused before + # any marker or read. + monkeypatch.setattr( + script.g, + "_gate_lock_status", + lambda _: { + "locked": True, + "registration_id": script.g.REGISTRATION_ID, + }, + ) + manifest = json.loads(MANIFEST.read_text()) + for record in manifest["fills"].values(): + (empty / record["file"]).write_bytes(b"not the registered bytes") + with pytest.raises(ValueError, match="not the registered bytes"): + script.main(["--manifest", str(MANIFEST), *base]) + assert not tmp_path.joinpath("result.json.started.json").exists() + output.write_text("{}") + with pytest.raises(FileExistsError, match="scored once"): + script.main(["--manifest", str(MANIFEST), *base]) diff --git a/tests/tier_counts.json b/tests/tier_counts.json index 7901f8e9..118a6c1e 100644 --- a/tests/tier_counts.json +++ b/tests/tier_counts.json @@ -1,8 +1,8 @@ { "schema_version": 1, "counts": { - "unit": 6017, - "artifact": 3388, + "unit": 6037, + "artifact": 3398, "integration_psid": 1341, "reproduction_legacy": 520, "oracle_policyengine": 220