-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathscripts_build_data.py
More file actions
158 lines (139 loc) · 6.01 KB
/
Copy pathscripts_build_data.py
File metadata and controls
158 lines (139 loc) · 6.01 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
"""Build app/public/data.json: per-state case arrays for in-browser Monte Carlo.
Usage: python scripts_build_data.py [QC_CSV] [PER_PDF] [PER_FY2025_PDF]
Positional arguments default to the author's cache paths so existing
invocations keep working; pass explicit paths to reproduce elsewhere.
"""
import base64
import csv
import json
import sys
from pathlib import Path
from snap_qc_sim import LEVERS, load_cases, load_official_rates
QC = "/Users/maxghenis/.cache/axiom-oracles/snap-qc/qc_pub_fy2024.csv"
PER = "/Users/maxghenis/.cache/axiom-oracles/snap-qc/snap-fy24QC-PER.pdf"
PER25 = "/Users/maxghenis/.cache/axiom-oracles/snap-qc/snap-qcfy25-per.pdf"
VERIFIED = {"CO": 856, "NY": 847, "CA": 883, "AZ": 922, "GA": 945, "MD": 722, "TX": 906}
LEVER_KEYS = ["smd", "ssed", "heat_and_eat", "bbce_resources"]
if len(sys.argv) > 4:
raise SystemExit(
"usage: python scripts_build_data.py [QC_CSV] [PER_PDF] [PER_FY2025_PDF]"
)
qc_path = sys.argv[1] if len(sys.argv) > 1 else QC
per_path = sys.argv[2] if len(sys.argv) > 2 else PER
per25_path = sys.argv[3] if len(sys.argv) > 3 else PER25
cases_by_state = load_cases(qc_path)
def _num(value: str | None) -> float | None:
value = (value or "").strip()
if not value or value == ".":
return None
return float(value)
def _self_employment_by_state(path: str) -> dict[str, list[bool]]:
"""Mirror load_cases' universe and row order for a compact client bitset."""
from snap_qc_sim.data import FIPS
flags: dict[str, list[bool]] = {}
with open(path) as source:
for row in csv.DictReader(source):
state = FIPS.get((row.get("STATE") or "").strip())
if state is None:
continue
case_flag = _num(row.get("CASE"))
if case_flag is not None and case_flag != 1:
continue
if _num(row.get("RAWBEN")) is None or (_num(row.get("HWGT")) or 0) <= 0:
continue
person = any((_num(row.get(f"SLFEMP{i}")) or 0) > 0 for i in range(1, 19))
flag = person or (_num(row.get("FSSLFEMP")) or 0) > 0
flags.setdefault(state, []).append(flag)
return flags
def _agency_findings_by_state(path: str) -> dict[str, list[list[int]]]:
"""Per case, [finding slots with an agency code, slots coded 17/19/20].
Mirrors load_cases' universe and row order. Codes 17 (computer
programming), 19 (computer-generated mass change), and 20 (arithmetic
computation) form the computation-caused class; 18 (data entry) and 21
(computer user) stay outside it.
"""
from snap_qc_sim.data import FIPS
strict = {17, 19, 20}
counts: dict[str, list[list[int]]] = {}
with open(path) as source:
for row in csv.DictReader(source):
state = FIPS.get((row.get("STATE") or "").strip())
if state is None:
continue
case_flag = _num(row.get("CASE"))
if case_flag is not None and case_flag != 1:
continue
if _num(row.get("RAWBEN")) is None or (_num(row.get("HWGT")) or 0) <= 0:
continue
codes = [
int(value)
for i in range(1, 10)
if (value := _num(row.get(f"AGENCY{i}"))) is not None
]
counts.setdefault(state, []).append(
[len(codes), sum(1 for c in codes if c in strict)]
)
return counts
def _pack_bits(flags: list[bool]) -> str:
packed = bytearray((len(flags) + 7) // 8)
for i, flag in enumerate(flags):
if flag:
packed[i // 8] |= 1 << (i % 8)
return base64.b64encode(packed).decode("ascii")
self_employment_by_state = _self_employment_by_state(qc_path)
agency_findings_by_state = _agency_findings_by_state(qc_path)
official = load_official_rates(per_path, include_national=True)
official_fy2025 = load_official_rates(per25_path, include_national=True)
missing_fy2025 = sorted(set(official) - set(official_fy2025) - {"US"})
if missing_fy2025:
raise SystemExit(f"FY2025 table is missing jurisdictions: {missing_fy2025}")
out = {}
for state, cases in sorted(cases_by_state.items()):
if state not in official:
continue
w, iss, err, hits = [], [], [], []
for c in cases:
w.append(round(c.weight, 2))
iss.append(round(c.issuance))
err.append(round(c.error))
if c.error > 0 and c.elements:
tot = len(c.elements)
hits.append([tot] + [len(c.elements & LEVERS[k]) for k in LEVER_KEYS])
else:
hits.append(0)
self_emp = self_employment_by_state.get(state, [])
if len(self_emp) != len(cases):
raise SystemExit(f"{state}: self-employment flags do not align with cases")
agency = agency_findings_by_state.get(state, [])
if len(agency) != len(cases):
raise SystemExit(f"{state}: agency finding counts do not align with cases")
# Compact: 0 for cases with no agency-coded findings, else [n, strict_n].
agency_compact = [0 if pair[0] == 0 else pair for pair in agency]
out[state] = {
"official": official[state],
"official_fy2025": official_fy2025[state],
"issuance": round(sum(c.weight * c.issuance for c in cases)),
"n": len(cases),
"verified": VERIFIED.get(state),
"w": w,
"iss": iss,
"err": err,
"hits": hits,
"self_emp": _pack_bits(self_emp),
"agency_findings": agency_compact,
}
path = Path("app/public/data.json")
path.parent.mkdir(parents=True, exist_ok=True)
payload = {
"levers": LEVER_KEYS,
"self_emp_encoding": "base64, little-endian bits within each byte",
"agency_findings_encoding": (
"per case: 0 when no finding slot carries an agency cause code, "
"else [slots with an agency code, slots coded 17/19/20]"
),
"national": {"fy2024": official.get("US"), "fy2025": official_fy2025.get("US")},
"states": out,
}
with path.open("w", encoding="utf-8") as output:
json.dump(payload, output, separators=(",", ":"))
print(f"{path}: {path.stat().st_size / 1e6:.1f}MB, {len(out)} states")