diff --git a/docs/catalog-curation-backlog.md b/docs/catalog-curation-backlog.md index 564a378a..307a399f 100644 --- a/docs/catalog-curation-backlog.md +++ b/docs/catalog-curation-backlog.md @@ -36,6 +36,46 @@ on `person/ui_claimant` under UUID period and entity metadata before folding these lineages; an alias cannot bridge either mismatch safely. +The same metadata explains two entries in the catalog's `stripped_segments` +that look like a naming bug: `week_2026-06-13` and `week_2026_06_13`. Both +come from the one 2026-06-13 row (ledger line 44), whose `source_record_id` +spells the week with ISO hyphens and whose `measure.concept` spells it with +underscores. The builder strips period tokens from both identifier fields +and records each distinct spelling, so the audit is right to list both. They +strip as `overlap`, not `derived`, because the row declares a month. Both +are recorded as reviewed overlap strips in +`tests/fixtures/series_catalog/reviewed_stripped_segments.json`. A +supersede correction that re-spells or re-periods that row will make the +audit report both as gone; the same change then updates the fixture. + +## Reviewing period strips (2026-09-30) + +`test_committed_catalog_is_current_and_valid` used to pin the exact list of +stripped period spellings, so every resolver append that recorded the next +weekly claims print failed the required "Arch checks" status, and the +resolver's admin merge went past it. The test now audits each +(spelling, concept) strip against the reviewed map in +`tests/fixtures/series_catalog/reviewed_stripped_segments.json` +(`build_series_catalog.stripped_segment_review_problems`). A strip passes +without review only when its concept already has a reviewed strip of the +same spelling template (`period_spelling_template`: the next week or month +written the same way) and every current observation behind it spells its own +declared period directly (kind `derived`). The audit reports: + +- a strip on a concept that has never stripped that template: a new series, + a new identifier shape, or a statute, cohort, or edition label that + happens to spell its row's period; +- an unreviewed `overlap` strip, or a reviewed strip that gains the + `overlap` kind; +- a reviewed strip that disappears. + +To clear a report, decide whether the segment is the row's period, then add +the pair with its kinds to the fixture, with a note when the reason is not +obvious. A concept-spelling duplicate shows up here first: a docket +placeholder such as `abs.labour.unemployment_rate` taking its first print +beside an already-observed `abs.labour.unemployment_rate.australia` is a new +concept stripping for the first time. + ## Builder polish The two LOW residuals accepted in the chronicle#145 merge disposition diff --git a/pyproject.toml b/pyproject.toml index d96fd965..332d75ba 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -29,6 +29,7 @@ dependencies = [ [project.optional-dependencies] dev = [ + "hypothesis>=6.100,<7", "pytest>=8.0.0,<9", "ruff>=0.5.0", ] @@ -56,5 +57,6 @@ target-version = "py311" [dependency-groups] dev = [ + "hypothesis>=6.100,<7", "pytest>=8.0.0,<9", ] diff --git a/scripts/build_series_catalog.py b/scripts/build_series_catalog.py index de6f4380..592b0de8 100644 --- a/scripts/build_series_catalog.py +++ b/scripts/build_series_catalog.py @@ -48,9 +48,15 @@ is invisible either way: every distinct stripped spelling is published in ``stripped_segments``, because a statute, cohort, or edition label that happens to spell the row's own period is mechanically indistinguishable -from a period label — the audit list is where a curator catches that. A -malformed period (month 13, an impossible week date) is a hard error, so -corrupt metadata can never manufacture strippable tokens. +from a period label — the audit list is where a curator catches that. The +test suite checks the list against a curated map of reviewed strips +(``stripped_segment_review_problems``): a strip that only continues a +reviewed decision (same concept, same ``period_spelling_template``, a +direct spelling of its own row's period) passes, so the next weekly or +monthly print does not re-open a question already answered; every other +strip still needs a curator. A malformed period (month 13, an impossible +week date) is a hard error, so corrupt metadata can never manufacture +strippable tokens. Aliases are curated identity statements, not derived data. Observed concept spellings of one identity become aliases automatically; everything else in @@ -305,6 +311,36 @@ def is_period_segment(segment: str) -> bool: return parse_period_token(segment) is not None +_MONTH_WORD_RE = re.compile(r"(? str: + """The way a period spelling is written, with its calendar values + abstracted: every digit becomes ``9`` and every month word, full or + abbreviated, ```` ("may" is both, so the two styles cannot be + told apart for every month). Separators and literal words (``week_``, + ``ending_``, ``after__``, ``fy``, ``q``, ``_to_``) are kept, + so two spellings share a template exactly when they write a period + the same way, whatever period they name. A template is the unit a + curator reviews in ``stripped_segment_review_problems``. + + >>> period_spelling_template("week_2026-08-29") + 'week_9999-99-99' + >>> period_spelling_template("week_2026_06_13") + 'week_9999_99_99' + >>> sorted({period_spelling_template(s) for s in ( + ... "may_2026", "june_2026", "jun_2026", "sept_2026")}) + ['_9999'] + >>> period_spelling_template("after_mpc_june_2026") + 'after_mpc__9999' + >>> period_spelling_template("february_to_april_2026") + '_to__9999' + >>> [period_spelling_template(s) for s in ("fy2024", "2026_q2", "2026")] + ['fy9999', '9999_q9', '9999'] + """ + return re.sub(r"\d", "9", _MONTH_WORD_RE.sub("", segment)) + + def period_token_variants(period: dict) -> set[str]: """Every direct spelling of ``period`` that may appear as an id segment.""" ptype, value = period.get("type"), period.get("value") @@ -675,6 +711,7 @@ def build_identities(rows: list[dict]) -> dict[tuple[str, str, str], dict]: "rid_patterns": set(), "suspects": set(), "stripped": set(), + "strip_kinds": {}, "geo_names": set(), "units": Counter(), "period_types": Counter(), @@ -704,11 +741,10 @@ def build_identities(rows: list[dict]) -> dict[tuple[str, str, str], dict]: "level/id/vintage cannot display as two places" ) for identifier in (concept_raw, rid): - ident["stripped"].update( - segment - for segment, kind in classify_segments(identifier, period) - if kind in ("derived", "overlap") - ) + for segment, kind in classify_segments(identifier, period): + if kind in ("derived", "overlap"): + ident["stripped"].add(segment) + ident["strip_kinds"].setdefault(segment, set()).add(kind) ident["units"][measure.get("unit")] += 1 ident["period_types"][period.get("type")] += 1 source = row.get("source") or {} @@ -1447,7 +1483,13 @@ def build_catalog( The plan records every registry-affecting outcome: ``mints`` (new identity bindings to append), ``supersedes`` (identities whose UUID changes — these require ``--allow-remint``), and ``dropped`` (existing - rows whose UUID would vanish from the catalog — also gated). + rows whose UUID would vanish from the catalog — also gated). It also + carries ``stripped_kinds``, keyed like the catalog's + ``stripped_segments`` (spelling -> canonical concept), giving the + ``classify_segments`` kinds (``derived``, ``overlap``) under which the + current observations stripped each spelling — the input + ``stripped_segment_review_problems`` audits. It never affects the + catalog bytes. """ raw = observations_path.read_bytes() observation_lines = raw.decode().split("\n") @@ -1552,6 +1594,7 @@ def build_catalog( "rid_patterns": set(), "suspects": set(), "stripped": set(), + "strip_kinds": {}, "geo_names": set(), "units": Counter(), "period_types": Counter(), @@ -1576,6 +1619,8 @@ def build_catalog( bucket["rid_patterns"] |= ident["rid_patterns"] bucket["suspects"] |= ident["suspects"] bucket["stripped"] |= ident["stripped"] + for segment, kinds in ident["strip_kinds"].items(): + bucket["strip_kinds"].setdefault(segment, set()).update(kinds) bucket["geo_names"] |= ident["geo_names"] if len(bucket["geo_names"]) > 1: raise SystemExit( @@ -1788,6 +1833,7 @@ def resolve_uuid( all_suspects: set[str] = set() stripped_map: dict[str, set[str]] = {} + stripped_kinds: dict[str, dict[str, set[str]]] = {} for canon_key in sorted(canonical): bucket = canonical[canon_key] concept, _, _ = canon_key @@ -1801,6 +1847,9 @@ def resolve_uuid( all_suspects.update(bucket["suspects"]) for segment in bucket["stripped"]: stripped_map.setdefault(segment, set()).add(concept) + stripped_kinds.setdefault(segment, {}).setdefault( + concept, set() + ).update(bucket["strip_kinds"][segment]) series.append({ "uuid": row_uuid, "concept": concept, @@ -2053,6 +2102,13 @@ def resolve_uuid( "ambiguous_aliases": ambiguous_aliases, "series": series, } + plan["stripped_kinds"] = { + segment: { + concept: sorted(stripped_kinds[segment][concept]) + for concept in sorted(stripped_kinds[segment]) + } + for segment in sorted(stripped_kinds) + } return catalog, plan @@ -2132,6 +2188,128 @@ def registry_agreement_problems( return problems +def _is_strippable_spelling(segment: str) -> bool: + # A grammar token, or the bare year period_token_variants derives for + # annual rows (bare years are deliberately outside the grammar). + return is_period_segment(segment) or ( + segment.isascii() and segment.isdigit() and len(segment) == 4 + and _valid_year(int(segment)) + ) + + +def stripped_segment_review_problems( + stripped_segments: dict[str, list[str]], + stripped_kinds: dict[str, dict[str, list[str]]], + reviewed: dict[str, dict[str, list[str]]], +) -> list[str]: + """Period strips in a catalog that no curator has reviewed. + + ``stripped_segments`` is the catalog's audit map (spelling -> canonical + concepts) and ``stripped_kinds`` the build plan's classification of + each of those strips (spelling -> concept -> ``classify_segments`` + kinds). ``reviewed`` has the shape of ``stripped_kinds``: the strips a + curator has looked at, with the kinds they had when reviewed. + + A strip outside ``reviewed`` needs no new review only when it continues + a decision already made: its concept already has a reviewed strip of + the SAME ``period_spelling_template`` (the next week or month written + the same way), and every current observation behind it spells its own + declared period directly (kind ``derived`` only). Everything else is + reported, one line per problem, sorted: + + * a reviewed strip that is gone (superseded observations, curation, or + a builder change); + * a reviewed strip that gained a kind it was not reviewed under; + * a strip whose concept has no reviewed strip of that template — a new + series, a new identifier shape, or a statute, cohort, or edition + label that happens to spell the row's own period, which only a + curator can tell apart; + * an unreviewed ``overlap`` strip, where identifier and declared period + disagree in granularity (the case ``classify_segments`` flags as + mechanically ambiguous); + * a stripped spelling that is not a period token, or a strip the plan + and the catalog disagree about. + + >>> reviewed = {"week_2026-08-22": {"us.dol.initial_claims.sa": ["derived"]}} + >>> segments = { + ... "week_2026-08-22": ["us.dol.initial_claims.sa"], + ... "week_2026-08-29": ["us.dol.initial_claims.sa"], + ... } + >>> kinds = {s: {"us.dol.initial_claims.sa": ["derived"]} for s in segments} + >>> stripped_segment_review_problems(segments, kinds, reviewed) + [] + >>> kinds["week_2026-08-29"]["us.dol.initial_claims.sa"] = ["overlap"] + >>> stripped_segment_review_problems(segments, kinds, reviewed) + ... # doctest: +ELLIPSIS + ["'week_2026-08-29' on 'us.dol.initial_claims.sa': unreviewed overlap..."] + """ + problems: list[str] = [] + live = { + (segment, concept) + for segment, concepts in stripped_segments.items() + for concept in concepts + } + planned = { + (segment, concept): set(kinds) + for segment, by_concept in stripped_kinds.items() + for concept, kinds in by_concept.items() + } + reviewed_kinds = { + (segment, concept): set(kinds) + for segment, by_concept in reviewed.items() + for concept, kinds in by_concept.items() + } + reviewed_templates: dict[str, set[str]] = {} + for segment, concept in reviewed_kinds: + reviewed_templates.setdefault(concept, set()).add( + period_spelling_template(segment) + ) + + def report(segment: str, concept: str, message: str) -> None: + problems.append(f"{segment!r} on {concept!r}: {message}") + + for segment, concept in reviewed_kinds.keys() - live: + report( + segment, concept, + "reviewed strip is gone from the catalog — a superseded " + "observation, curation, or a builder change; re-review and " + "update the reviewed map", + ) + for segment, concept in live ^ planned.keys(): + where = "catalog" if (segment, concept) in live else "build plan" + report(segment, concept, f"strip appears only in the {where}") + for segment, concept in live & planned.keys(): + kinds = planned[(segment, concept)] + if not _is_strippable_spelling(segment): + report(segment, concept, "stripped spelling is not a period token") + if (segment, concept) in reviewed_kinds: + gained = kinds - reviewed_kinds[(segment, concept)] + if gained: + report( + segment, concept, + f"reviewed strip gained kinds {sorted(gained)} — an " + "observation now spells this period through a " + "different declared period; review it", + ) + continue + if kinds - {"derived"}: + report( + segment, concept, + f"unreviewed overlap strip (kinds {sorted(kinds)}) — the " + "identifier names a window that only overlaps its row's " + "declared period; review it", + ) + template = period_spelling_template(segment) + if template not in reviewed_templates.get(concept, set()): + report( + segment, concept, + f"first {template!r} strip for this concept — confirm it " + "is a period label, not a statute, cohort, or edition " + "label, and add it to the reviewed map", + ) + return sorted(problems) + + def git_head_bytes(path: pathlib.Path) -> bytes | None: """The file's committed HEAD content, or None when unavailable.""" try: diff --git a/tests/fixtures/series_catalog/reviewed_stripped_segments.json b/tests/fixtures/series_catalog/reviewed_stripped_segments.json new file mode 100644 index 00000000..5afc5488 --- /dev/null +++ b/tests/fixtures/series_catalog/reviewed_stripped_segments.json @@ -0,0 +1,255 @@ +{ + "comment": "Period strips a curator has reviewed: the stripped_segments of ledger/series_catalog.json as of 55bbf3d (2026-09-03), whose spellings were pinned after review on 2026-09-04 (PolicyEngine/chronicle#244), with the classify_segments kinds each strip had then (spelling -> concept -> kinds; the current builder's plan over 55bbf3d's inputs). tests/test_build_series_catalog.py audits the committed catalog against this map with build_series_catalog.stripped_segment_review_problems: a strip outside it passes only when its concept already has a reviewed strip of the same spelling template and every observation behind it spells its own declared period directly (kind derived). When the audit reports a strip, review it (is the segment the row's period, or a statute, cohort, or edition label that happens to spell it?) and record the pair here with its kinds, adding a note when the reason is not obvious. Keys stay sorted.", + "reviewed_catalog_commit": "55bbf3d", + "notes": { + "week_2026-06-13": "One observation, the 2026-06-13 initial-claims first print (ledger line 44), spells its week twice: source_record_id us.dol.initial_claims.sa.week_2026-06-13 with ISO hyphens and measure.concept us.dol.initial_claims.sa.week_2026_06_13 with underscores. The builder strips period tokens from both identifier fields and records each distinct spelling, so this pair is one week spelled two ways, not a naming bug in the builder or the catalog. Both strip as overlap because the row declares period month 2026-06; see docs/catalog-curation-backlog.md, 'Initial claims cadence metadata'.", + "week_2026_06_13": "The underscore spelling of week_2026-06-13; see that note.", + "week_2026-07-13": "va.vba.mmwr.claims_inventory, first seen at c2aa68d: the VA MMWR publication date for the week ending 2026-07-11, the row's own period, so a period spelling of this row and not a colliding label; the strip stands. It strips as overlap (the report date falls after the week it reports), so each further MMWR print in this spelling is reported for review until the rows spell their own week.", + "week_ending_2026_06_06": "The 2026-06-06 initial-claims first print, also declared as month 2026-06: the same cadence-metadata defect as week_2026-06-13." + }, + "stripped_kinds": { + "2026-05": { + "abs.cpi_indicator.allgroups.yoy": ["derived"], + "census.housing_starts.saar": ["derived"], + "estat.jp.cpi.core_exfreshfood.yoy": ["derived"], + "fed.g17.industrial_production.total_index_mom": ["derived"], + "ons.cpih.annual_rate": ["derived"], + "statcan.36-10-0434-01.all_industries.month_to_month_percent_change": ["derived"], + "statcan.cpi.allitems.yoy": ["derived"], + "us.bea.core_pce.mom_sa": ["derived"] + }, + "2026-06": { + "abs.cpi.all_groups.yoy": ["derived"], + "bls.import_price_index.all_imports_mom": ["derived"], + "census.housing_starts.saar": ["derived"], + "eurostat.ea.hicp.flash.yoy": ["derived"], + "fed.g17.capacity_utilization.total_industry": ["derived"], + "fed.g17.industrial_production.total_index_mom": ["derived"], + "ssa.oasdi.disabled_worker_beneficiaries": ["derived"], + "us.bea.core_pce.mom_sa": ["derived"], + "us.fed.fomc.target_range_upper": ["derived"] + }, + "2026-06-18": { + "boe.bank_rate": ["overlap"] + }, + "2026-07": { + "bls.ces.home_health_care_services.employment": ["derived"], + "bls.cps.LNU02374597": ["derived"], + "bls.cps.lfpr_55_plus": ["derived"], + "bls.import_price_index.all_imports_mom": ["derived"], + "bls.laus.colorado.labor_force": ["derived"], + "census.housing_starts.saar": ["derived"], + "cms.care_compare.nursing_home_occupancy_pct": ["derived"], + "cms.nursing_home_compare.reported_total_nurse_staffing_hprd_us": ["derived"], + "eurostat.ea.hicp.flash.yoy": ["derived"], + "fed.g17.capacity_utilization.total_industry": ["derived"], + "fed.g17.industrial_production.total_index_mom": ["derived"], + "ssa.ssi.recipients.colorado": ["derived"], + "ssa.ssi.recipients.colorado.aged_65_plus": ["derived"], + "ssa.ssi.total_recipients": ["derived"] + }, + "2026_05": { + "census.housing_starts.saar": ["derived"], + "estat.jp.cpi.core_exfreshfood.yoy": ["derived"], + "fed.g17.industrial_production.total_index_mom": ["derived"], + "ons.cpih.annual_rate": ["derived"] + }, + "2026_06": { + "bls.jolts.hires_rate": ["derived"], + "census.construction_spending.total_mom": ["derived"], + "census.m3.durable_goods_new_orders_mom": ["derived"], + "census.m3.durable_goods_shipments_mom": ["derived"], + "fed.g19.consumer_credit_nonrevolving_annual_rate": ["derived"], + "fed.g19.consumer_credit_revolving_annual_rate": ["derived"], + "fed.g19.consumer_credit_total_annual_rate": ["derived"], + "ssa.ssi.recipients_aged_65_plus": ["derived"], + "us.fed.fomc.target_range_upper": ["derived"] + }, + "2026_06_18": { + "boe.bank_rate": ["overlap"] + }, + "2026_07": { + "bea.trade.goods_services_deficit": ["derived"], + "bls.cpi.owners_equivalent_rent_mom": ["derived"], + "bls.cpi.rent_primary_residence_mom": ["derived"], + "bls.cpi.services_less_energy_mom": ["derived"], + "bls.cpi.services_less_rent_shelter_mom": ["derived"], + "bls.cpi.shelter_mom": ["derived"], + "bls.cps.u6_underemployment_rate": ["derived"], + "bls.export_prices.all_commodities_mom": ["derived"], + "bls.jolts.hires_rate": ["derived"], + "census.construction_spending.total_mom": ["derived"], + "census.housing.completions_saar": ["derived"], + "census.housing.permits_saar": ["derived"], + "census.m3.durable_goods_new_orders_mom": ["derived"], + "census.m3.durable_goods_shipments_mom": ["derived"], + "census.new_residential_sales.new_single_family_houses_sold_saar": ["derived"], + "fed.g17.capacity_utilization.manufacturing": ["derived"], + "fed.g17.manufacturing_production_mom": ["derived"] + }, + "2026_q2": { + "bls.eci.private_wages_salaries_qoq": ["derived"], + "bls.eci.total_compensation_private_industry_qoq": ["derived"], + "bls.productivity.nonfarm_unit_labor_costs_qoq_prelim": ["derived"] + }, + "after_june_2026": { + "bank_of_canada.overnight_rate": ["overlap"], + "boj.policy_rate_guideline": ["overlap"], + "ecb.deposit_facility_rate": ["overlap"], + "rba.cash_rate_target": ["overlap"] + }, + "after_mpc_june_2026": { + "boe.bank_rate": ["overlap"] + }, + "april_2026": { + "census.mtis.total_business_inventories_level": ["derived"], + "eurostat.industrial_production.euro_area": ["derived"], + "ons.gdp.monthly_growth": ["derived"], + "statcan.building_permits.total_value_mom.canada": ["derived"], + "statcan.employment_insurance.regular_beneficiaries.canada": ["derived"], + "statcan.gdp_by_industry.monthly_growth": ["derived"], + "statcan.retail_trade.sales_mom.canada": ["derived"], + "statcan.wholesale_trade.sales_mom_exclusions.canada": ["derived"] + }, + "feb_2026": { + "cms.medicaid_pi.beneficiaries_disenrolled_procedural": ["derived"], + "cms.medicaid_pi.beneficiaries_disenrolled_total": ["derived"], + "cms.medicaid_pi.beneficiaries_renewed_ex_parte": ["derived"], + "cms.medicaid_pi.beneficiaries_renewed_total": ["derived"] + }, + "february_to_april_2026": { + "ons.labour.unemployment_rate": ["overlap"] + }, + "fy2024": { + "fns.snap.application_processing_timeliness_rate": ["derived"], + "fns.snap.overpayment_error_rate": ["derived"], + "fns.snap.total_payment_error_rate": ["derived"], + "fns.snap.underpayment_error_rate": ["derived"] + }, + "fy2025": { + "fns.snap.share_jurisdictions_at_or_above_6pct": ["derived"], + "fns.snap.total_payment_error_rate": ["derived"] + }, + "july_2026": { + "bls.cpi.u.core_mom": ["derived"], + "bls.cpi.u.headline_mom": ["derived"], + "bls.cps.unemployment_rate": ["derived"] + }, + "june_2026": { + "abs.labour.employment_change.australia": ["derived"], + "abs.labour.unemployment_rate.australia": ["derived"], + "bls.ces.aerospace_product_and_parts_employment": ["derived"], + "bls.ces.federal_department_of_defense_employment": ["derived"], + "bls.ces.ship_and_boat_building_employment": ["derived"], + "bls.ces.total_nonfarm_payroll_change": ["derived"], + "bls.cpi.u.core_mom": ["derived"], + "bls.cpi.u.headline_mom": ["derived"], + "bls.cps.employed_people_by_occupation.business_financial_operations": ["derived"], + "bls.cps.employed_people_by_occupation.computer_mathematical": ["derived"], + "bls.cps.employed_people_by_occupation.healthcare_support": ["derived"], + "bls.cps.employed_people_by_occupation.office_administrative_support": ["derived"], + "bls.cps.employed_people_by_occupation.production": ["derived"], + "bls.cps.employed_people_by_occupation.transportation_material_moving": ["derived"], + "bls.cps.unemployment_rate": ["derived"], + "bls.jolts.job_openings": ["derived"], + "eurostat.hicp.all_items_annual_rate.euro_area": ["derived"], + "statjp.cpi.tokyo_all_items_annual_rate": ["derived"] + }, + "may_2026": { + "abs.building_approvals.total_dwellings_mom.australia": ["derived"], + "abs.cpi.all_groups_annual_rate.australia": ["derived"], + "abs.labour.employment_change.australia": ["derived"], + "abs.labour.unemployment_rate.australia": ["derived"], + "bea.disposable_personal_income.level": ["derived"], + "bea.government_social_benefits.level": ["derived"], + "bea.government_social_benefits.medicaid": ["derived"], + "bea.government_social_benefits.medicare": ["derived"], + "bea.government_social_benefits.social_security": ["derived"], + "bea.pce.core_mom": ["derived"], + "bea.pce_price_index.monthly_change": ["derived"], + "bea.personal_current_taxes.level": ["derived"], + "bea.wages_and_salaries.level": ["derived"], + "bls.ces.average_hourly_earnings_private_monthly_change": ["derived"], + "bls.ces.total_nonfarm_payroll_change": ["derived"], + "bls.cpi.u.core_mom": ["derived"], + "bls.cpi.u.headline_mom": ["derived"], + "bls.cps.unemployment_rate": ["derived"], + "bls.import_price_index.all_imports_mom": ["derived"], + "bls.jolts.job_openings": ["derived"], + "bls.jolts.job_openings_total": ["derived"], + "bls.ppi.final_demand_monthly_change": ["derived"], + "census.housing_starts.saar": ["derived"], + "census.marts.adv44x72.monthly_change": ["derived"], + "eurostat.hicp.all_items_annual_rate.euro_area": ["derived"], + "eurostat.retail_trade.volume_mom.euro_area": ["derived"], + "eurostat.unemployment_rate.euro_area": ["derived"], + "fed.g17.capacity_utilization.total_industry": ["derived"], + "fed.g17.industrial_production.total_index_mom": ["derived"], + "ons.cpi.annual_rate": ["derived"], + "ons.hmrc.paye_payrolled_employees": ["derived"], + "ons.pusf.j5ii.public_sector_net_borrowing_ex_banks": ["derived"], + "ons.retail_sales.volume_mom": ["derived"], + "statcan.cpi.all_items_annual_rate.canada": ["derived"], + "statcan.employment_insurance.regular_beneficiaries.canada": ["derived"], + "statcan.lfs.employment_change": ["derived"], + "statcan.lfs.unemployment_rate": ["derived"], + "statjp.cpi.all_items_annual_rate.japan": ["derived"], + "statjp.household_spending.real_yoy.two_or_more_person_households": ["derived"], + "statjp.lfs.unemployment_rate.japan": ["derived"], + "treasury.mts.monthly_deficit": ["derived"] + }, + "q1_2026": { + "bea.real_gdp.saar.third_estimate": ["derived"] + }, + "week_2026-06-13": { + "us.dol.initial_claims.sa": ["overlap"] + }, + "week_2026-06-20": { + "us.dol.initial_claims.sa": ["derived"] + }, + "week_2026-06-27": { + "dol.eta.continued_claims.sa": ["derived"] + }, + "week_2026-07-04": { + "dol.eta.continued_claims.sa": ["derived"], + "us.dol.initial_claims.sa": ["derived"] + }, + "week_2026-07-11": { + "dol.eta.continued_claims.sa": ["derived"], + "us.dol.initial_claims.sa": ["derived"] + }, + "week_2026-07-13": { + "va.vba.mmwr.claims_inventory": ["overlap"] + }, + "week_2026-07-18": { + "dol.eta.continued_claims.sa": ["derived"], + "us.dol.initial_claims.sa": ["derived"] + }, + "week_2026-07-25": { + "dol.eta.continued_claims.sa": ["derived"], + "us.dol.initial_claims.sa": ["derived"] + }, + "week_2026-08-01": { + "dol.eta.continued_claims.sa": ["derived"], + "us.dol.initial_claims.sa": ["derived"] + }, + "week_2026-08-08": { + "dol.eta.continued_claims.sa": ["derived"], + "us.dol.initial_claims.sa": ["derived"] + }, + "week_2026-08-15": { + "dol.eta.continued_claims.sa": ["derived"], + "us.dol.initial_claims.sa": ["derived"] + }, + "week_2026-08-22": { + "dol.eta.continued_claims.sa": ["derived"], + "us.dol.initial_claims.sa": ["derived"] + }, + "week_2026_06_13": { + "us.dol.initial_claims.sa": ["overlap"] + }, + "week_ending_2026_06_06": { + "us.dol.initial_claims.sa": ["overlap"] + } + } +} diff --git a/tests/test_build_series_catalog.py b/tests/test_build_series_catalog.py index 584cb707..7834bfaf 100644 --- a/tests/test_build_series_catalog.py +++ b/tests/test_build_series_catalog.py @@ -10,18 +10,36 @@ from __future__ import annotations +import datetime as dt import doctest import json import pathlib import sys +import tempfile import pytest +from hypothesis import HealthCheck, given, settings +from hypothesis import strategies as st ROOT = pathlib.Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT / "scripts")) import build_series_catalog as bsc # noqa: E402 +# The curated map of reviewed period strips (spelling -> canonical concept -> +# classify_segments kinds) that test_committed_catalog_is_current_and_valid +# audits the committed catalog against; its "comment" says how to extend it. +REVIEWED_STRIPPED_SEGMENTS_PATH = ( + ROOT / "tests" / "fixtures" / "series_catalog" + / "reviewed_stripped_segments.json" +) +REVIEWED_STRIPPED_FIXTURE = json.loads( + REVIEWED_STRIPPED_SEGMENTS_PATH.read_text(encoding="utf-8") +) +REVIEWED_STRIPPED_KINDS: dict[str, dict[str, list[str]]] = ( + REVIEWED_STRIPPED_FIXTURE["stripped_kinds"] +) + PERIOD_SEGMENTS = [ "fy2026", "2026-05", @@ -774,29 +792,27 @@ def test_committed_catalog_is_current_and_valid() -> None: # digit run 2374 trips the year hint; it is an identifier, not a date, # and stays in the identity because the observation was recorded so. assert committed["suspect_segments"] == ["LNU02374597"] - # Pin extended 2026-09-04 to the catalog at 55bbf3d. Spellings first seen: - # week_2026-07-13 at c2aa68d (va.vba.mmwr.claims_inventory: the VA MMWR - # publication date for the week ending 2026-07-11, the row's own period, - # so a period spelling of this row and not a colliding label; the strip - # stands), week_2026-08-15 at 54dbabc8 and week_2026-08-22 at 55bbf3d. - # Occurrences as of 55bbf3d: both August keys map to - # dol.eta.continued_claims.sa (joined week_2026-08-15 at d77afe2) and - # us.dol.initial_claims.sa. - # EVERY stripped spelling is auditable, mapped to the canonical - # concepts it touched — a statute or edition label colliding with a - # period spelling can only be caught here. - assert sorted(committed["stripped_segments"]) == [ - "2026-05", "2026-06", "2026-06-18", "2026-07", "2026_05", - "2026_06", "2026_06_18", "2026_07", "2026_q2", "after_june_2026", - "after_mpc_june_2026", "april_2026", "feb_2026", - "february_to_april_2026", "fy2024", "fy2025", "july_2026", - "june_2026", "may_2026", "q1_2026", "week_2026-06-13", - "week_2026-06-20", "week_2026-06-27", "week_2026-07-04", - "week_2026-07-11", "week_2026-07-13", "week_2026-07-18", - "week_2026-07-25", "week_2026-08-01", "week_2026-08-08", - "week_2026-08-15", "week_2026-08-22", "week_2026_06_13", - "week_ending_2026_06_06", - ] + # EVERY stripped spelling is audited, mapped to the canonical concepts + # it touched: a statute or edition label colliding with a period + # spelling can only be caught here. Until 2026-09-30 this was an exact + # list of spellings, so every append that recorded the next weekly + # claims print (week_2026-08-29, ...) failed this required check and + # the resolver merged past it. The audit now asks the question the list + # stood for, pair by pair: has a curator already decided this concept's + # segments of this spelling template are period labels, and does every + # observation behind the strip spell its own declared period directly? + # New concepts, new templates, overlap strips, and reviewed strips that + # disappear or change kind still fail. See REVIEWED_STRIPPED_SEGMENTS_PATH. + problems = bsc.stripped_segment_review_problems( + committed["stripped_segments"], + plan["stripped_kinds"], + REVIEWED_STRIPPED_KINDS, + ) + assert problems == [], ( + "unreviewed period strips — review each and record it in " + f"{REVIEWED_STRIPPED_SEGMENTS_PATH.relative_to(ROOT)}:\n" + + "\n".join(problems) + ) assert committed["stripped_segments"]["after_mpc_june_2026"] == [ "boe.bank_rate" ] @@ -810,6 +826,713 @@ def test_committed_catalog_is_current_and_valid() -> None: assert len(committed["series"]) == 228 +# --- The stripped-segment audit --------------------------------------------- + + +def _committed_build() -> tuple[dict, dict]: + committed = json.loads(bsc.CATALOG.read_text(encoding="utf-8")) + _, plan = bsc.build_catalog( + bsc.OBSERVATIONS, + bsc.DOCKET_SEED, + bsc.ExistingCatalog(bsc.CATALOG), + bsc.UuidRegistry.load(bsc.UUID_REGISTRY), + ) + return committed, plan + + +def _pairs(stripped: dict) -> set[tuple[str, str]]: + return {(s, c) for s, concepts in stripped.items() for c in concepts} + + +def _names(problems: list[str], spelling: str, concept: str) -> list[str]: + return [p for p in problems if p.startswith(f"{spelling!r} on {concept!r}")] + + +def test_reviewed_stripped_segments_fixture_is_wellformed() -> None: + fixture = REVIEWED_STRIPPED_FIXTURE + assert set(fixture) == { + "comment", "reviewed_catalog_commit", "notes", "stripped_kinds", + } + reviewed = fixture["stripped_kinds"] + assert list(reviewed) == sorted(reviewed) + committed = json.loads(bsc.CATALOG.read_text(encoding="utf-8")) + row_concepts = {row["concept"] for row in committed["series"]} + for spelling, by_concept in reviewed.items(): + assert bsc._is_strippable_spelling(spelling), spelling + assert by_concept and list(by_concept) == sorted(by_concept), spelling + assert set(by_concept) <= row_concepts, (spelling, by_concept) + for kinds in by_concept.values(): + assert kinds and kinds == sorted(set(kinds)), (spelling, kinds) + assert set(kinds) <= {"derived", "overlap"}, (spelling, kinds) + assert set(fixture["notes"]) <= set(reviewed) + # The June 13 initial-claims pair is one observation's week, spelled + # with hyphens in its source_record_id and underscores in its concept; + # both are kept, and both are overlap strips (see the fixture notes). + for spelling in ("week_2026-06-13", "week_2026_06_13"): + assert reviewed[spelling] == {"us.dol.initial_claims.sa": ["overlap"]} + rows = [ + json.loads(line) + for line in bsc.OBSERVATIONS.read_text(encoding="utf-8").splitlines() + if "week_2026-06-13" in line or "week_2026_06_13" in line + ] + assert len(rows) == 1 + assert rows[0]["source_record_id"].endswith(".week_2026-06-13") + assert rows[0]["measure"]["concept"].endswith(".week_2026_06_13") + assert rows[0]["period"] == {"type": "month", "value": "2026-06"} + + +def test_stripped_kinds_match_an_independent_replay() -> None: + """Differential: re-derive every strip's kinds from the committed ledger + with classify_segments alone, resolving each observation to its catalog + row through that row's concept and aliases at the same geography and + entity, and require the builder's plan and the catalog to agree.""" + from check_thesis_facts_append import effective_current_rows + + committed, plan = _committed_build() + owners: dict[tuple[str, str, str], set[str]] = {} + for row in committed["series"]: + dims = ( + bsc._geo_key(row.get("geography")), + bsc._entity_key(row.get("entity")), + ) + for name in [row["concept"], *row["aliases"]]: + owners.setdefault((name, *dims), set()).add(row["concept"]) + rows = [ + json.loads(line) + for line in bsc.OBSERVATIONS.read_text(encoding="utf-8").splitlines() + if line.strip() + ] + replay: dict[str, dict[str, set[str]]] = {} + for obs in effective_current_rows(rows): + period = obs.get("period") or {} + raw = obs["measure"]["concept"] + dims = ( + bsc._geo_key(obs.get("geography") or None), + bsc._entity_key(obs.get("entity") or None), + ) + pattern_concept = bsc.concept_for(bsc.family_pattern(raw, period)) + (canonical,) = owners.get((pattern_concept, *dims)) or owners[ + (raw, *dims) + ] + for identifier in (raw, obs["source_record_id"]): + for segment, kind in bsc.classify_segments(identifier, period): + if kind != "kept": + replay.setdefault(segment, {}).setdefault( + canonical, set() + ).add(kind) + assert plan["stripped_kinds"] == { + s: {c: sorted(kinds) for c, kinds in by.items()} + for s, by in replay.items() + } + assert { + s: sorted(by) for s, by in plan["stripped_kinds"].items() + } == committed["stripped_segments"] + + +def test_committed_audit_fails_on_corrupted_catalogs() -> None: + # Each corruption must add a problem naming its pair, whatever else the + # committed catalog is waiting on (that is the committed test's job). + committed, plan = _committed_build() + segments = committed["stripped_segments"] + kinds = plan["stripped_kinds"] + audit = bsc.stripped_segment_review_problems + baseline = set(audit(segments, kinds, REVIEWED_STRIPPED_KINDS)) + + def corrupt(spelling, concept, strip_kinds, *, in_plan=True): + bad_segments = {s: list(c) for s, c in segments.items()} + bad_kinds = {s: dict(by) for s, by in kinds.items()} + if concept not in bad_segments.get(spelling, []): + bad_segments.setdefault(spelling, []).append(concept) + if in_plan: + bad_kinds.setdefault(spelling, {})[concept] = strip_kinds + return _names( + sorted( + set(audit(bad_segments, bad_kinds, REVIEWED_STRIPPED_KINDS)) + - baseline + ), + spelling, concept, + ) + + # A reviewed spelling disappears. + gone = {s: c for s, c in segments.items() if s != "after_mpc_june_2026"} + gone_kinds = {s: b for s, b in kinds.items() if s != "after_mpc_june_2026"} + assert _names( + audit(gone, gone_kinds, REVIEWED_STRIPPED_KINDS), + "after_mpc_june_2026", "boe.bank_rate", + ) + # A concept strips for the first time, as the 2026-09-24 proposal (#289) + # did when abs.labour.unemployment_rate took its first print beside the + # already-observed abs.labour.unemployment_rate.australia. + assert corrupt("2026_08", "corrupt.never_observed.rate", ["derived"]) + # A reviewed concept strips a spelling template it never used. + assert corrupt("2026_08", "us.dol.initial_claims.sa", ["derived"]) + # A reviewed concept and template, but an overlap strip. + assert corrupt("week_2026-09-19", "us.dol.initial_claims.sa", ["overlap"]) + # A reviewed strip that an observation now reaches as an overlap. + assert corrupt( + "week_2026-08-22", "us.dol.initial_claims.sa", ["derived", "overlap"] + ) + # A statute label on a reviewed concept. + assert corrupt("section_2026", "boe.bank_rate", ["derived"]) + # A non-period spelling is refused even when a curator listed it. + bogus = {**REVIEWED_STRIPPED_KINDS, "section_2026": { + "boe.bank_rate": ["derived"], + }} + bogus_segments = {**segments, "section_2026": ["boe.bank_rate"]} + bogus_kinds = {**kinds, "section_2026": {"boe.bank_rate": ["derived"]}} + assert set(audit(bogus_segments, bogus_kinds, bogus)) - baseline == { + "'section_2026' on 'boe.bank_rate': stripped spelling is not a " + "period token" + } + # The catalog claims a strip the build plan does not make. + assert corrupt( + "week_2026-09-19", "us.dol.initial_claims.sa", ["derived"], + in_plan=False, + ) + # And the routine case the exact list used to reject passes. + assert not corrupt( + "week_2026-09-19", "us.dol.initial_claims.sa", ["derived"] + ) + + +def _successor(period: dict) -> dict | None: + ptype, value = period.get("type"), str(period.get("value")) + if ptype == "week_ending": + end = dt.date.fromisoformat(value) + dt.timedelta(days=7) + return {"type": ptype, "value": end.isoformat()} + if ptype in ("month", "quarter"): + year, month = map(int, value.split("-")) + month += 1 if ptype == "month" else 3 + year, month = year + (month - 1) // 12, (month - 1) % 12 + 1 + return {"type": ptype, "value": f"{year}-{month:02d}"} + if ptype in ("fiscal_year", "year", "calendar_year", "tax_year"): + return {"type": ptype, "value": str(int(value) + 1)} + return None + + +def _respell(identifier: str, period: dict, successor: dict) -> str | None: + """``identifier`` with each strip of ``period`` rewritten as the same + template's spelling of ``successor``; None if any strip has no such + spelling (an overlap strip).""" + variants = bsc.period_token_variants(successor) + parts = [] + for segment, kind in bsc.classify_segments(identifier, period): + if kind == "kept": + parts.append(segment) + continue + template = bsc.period_spelling_template(segment) + same = [ + v for v in variants if bsc.period_spelling_template(v) == template + ] + if kind != "derived" or len(same) != 1: + return None + parts.append(same[0]) + return ".".join(parts) + + +def test_next_period_of_every_committed_series_passes_the_audit( + tmp_path: pathlib.Path, +) -> None: + """The append the exact list rejected, on real data: for every current + observation whose strips are direct spellings of its period, append the + NEXT period written the same way. The rebuilt catalog grows new + spellings and lands them on the existing identities, and with the + committed strips taken as reviewed, none of the new ones needs review. + (Whether the committed strips ARE reviewed is the committed test's + question; on a clean branch the two maps agree.)""" + from check_thesis_facts_append import effective_current_rows + + rows = [ + json.loads(line) + for line in bsc.OBSERVATIONS.read_text(encoding="utf-8").splitlines() + if line.strip() + ] + appended = [] + for obs in effective_current_rows(rows): + period = obs.get("period") or {} + successor = _successor(period) + if successor is None: + continue + rid = _respell(obs["source_record_id"], period, successor) + concept = _respell(obs["measure"]["concept"], period, successor) + if rid is None or concept is None or rid == obs["source_record_id"]: + continue + nxt = json.loads(json.dumps(obs)) + nxt.pop("assertionVersion", None) + nxt["period"] = successor + nxt["source_record_id"] = rid + nxt["measure"]["concept"] = concept + appended.append(nxt) + assert len(appended) >= 100 # most of the ledger, not a vacuous pass + observations = tmp_path / "obs.jsonl" + observations.write_text( + bsc.OBSERVATIONS.read_text(encoding="utf-8") + + "".join(json.dumps(r) + "\n" for r in appended), + encoding="utf-8", + ) + catalog, plan = bsc.build_catalog( + observations, + bsc.DOCKET_SEED, + bsc.ExistingCatalog(bsc.CATALOG), + bsc.UuidRegistry.load(bsc.UUID_REGISTRY), + ) + assert not plan["mints"] and not plan["supersedes"] + assert not plan["dropped"] and not plan["enrich_retires"] + committed = json.loads(bsc.CATALOG.read_text(encoding="utf-8")) + new_pairs = _pairs(catalog["stripped_segments"]) - _pairs( + committed["stripped_segments"] + ) + assert len(new_pairs) >= 50 + _, committed_plan = _committed_build() + assert bsc.stripped_segment_review_problems( + catalog["stripped_segments"], + plan["stripped_kinds"], + committed_plan["stripped_kinds"], + ) == [] + + +def _strip_builds( + tmp_path: pathlib.Path, + base: list[dict], + appended: list[dict], + reviewed: dict | None = None, +) -> tuple[dict, dict, list[str]]: + """Build ``base`` (its strips become the reviewed map unless one is + given), append ``appended``, rebuild on the first build's catalog, and + audit the result. UUID staging is not under test here, so the builds + skip ``main`` (and its git calls) and inherit from the catalog alone. + Returns the reviewed map, the rebuilt catalog, and the problems.""" + first, first_plan = _build(tmp_path, base, docket=SEED) + if reviewed is None: + reviewed = first_plan["stripped_kinds"] + catalog, plan = _build( + tmp_path, [*base, *appended], existing=first, docket=SEED + ) + problems = bsc.stripped_segment_review_problems( + catalog["stripped_segments"], plan["stripped_kinds"], reviewed + ) + return reviewed, catalog, problems + + +def _strip_audit( + tmp_path: pathlib.Path, + base: list[dict], + appended: list[dict], + reviewed: dict | None = None, +) -> list[str]: + return _strip_builds(tmp_path, base, appended, reviewed)[2] + + +CLAIMS = "us.dol.initial_claims.sa" + + +def _weekly(end: str, role: str = "ui_claimant") -> dict: + return _row( + CLAIMS, + rid=f"{CLAIMS}.week_{end}", + unit="thousands", + period={"type": "week_ending", "value": end}, + entity={"name": "person", "role": role}, + ) + + +def test_strip_audit_passes_the_next_print(tmp_path: pathlib.Path) -> None: + def monthly(value: str) -> dict: + return _row( + "census.m3.orders_mom", + rid=f"census.m3.orders_mom.{value.replace('-', '_')}", + period={"type": "month", "value": value}, + ) + + assert _strip_audit( + tmp_path, + [_weekly("2026-08-15"), _weekly("2026-08-22"), monthly("2026-06")], + [_weekly("2026-08-29"), _weekly("2026-09-05"), monthly("2026-07")], + ) == [] + + +def test_strip_audit_flags_a_label_on_a_new_concept( + tmp_path: pathlib.Path, +) -> None: + # A statute label that spells its row's own period: mechanically a + # period, and only a curator can say otherwise. Its concept has never + # stripped, so nothing reviewed covers it. + statute = _row( + "treasury.debt_limit.suspension_act", + rid="treasury.debt_limit.suspension_act.2026_09.first_print", + period={"type": "month", "value": "2026-09"}, + ) + problems = _strip_audit(tmp_path, [_weekly("2026-08-22")], [statute]) + assert problems == [ + "'2026_09' on 'treasury.debt_limit.suspension_act': first " + "'9999_99' strip for this concept — confirm it is a period label, " + "not a statute, cohort, or edition label, and add it to the " + "reviewed map" + ] + + +def test_strip_audit_flags_a_new_template_on_a_reviewed_concept( + tmp_path: pathlib.Path, +) -> None: + ending = _row( + CLAIMS, + rid=f"{CLAIMS}.week_ending_2026_08_29", + unit="thousands", + period={"type": "week_ending", "value": "2026-08-29"}, + entity={"name": "person", "role": "ui_claimant"}, + ) + problems = _strip_audit(tmp_path, [_weekly("2026-08-22")], [ending]) + assert len(problems) == 1 + assert _names(problems, "week_ending_2026_08_29", CLAIMS) + assert "first 'week_ending_9999_99_99' strip" in problems[0] + + +def test_strip_audit_flags_an_overlap_strip_on_a_reviewed_template( + tmp_path: pathlib.Path, +) -> None: + # The June 13 defect again: a weekly print declared as a month, on a + # second entity of the reviewed concept. + monthly = _row( + CLAIMS, + rid=f"{CLAIMS}.week_2026-09-12", + unit="thousands", + period={"type": "month", "value": "2026-09"}, + entity={"name": "person", "role": "ui_initial_claimant"}, + ) + problems = _strip_audit(tmp_path, [_weekly("2026-08-22")], [monthly]) + assert problems == [ + f"'week_2026-09-12' on '{CLAIMS}': unreviewed overlap strip " + "(kinds ['overlap']) — the identifier names a window that only " + "overlaps its row's declared period; review it" + ] + + +def test_strip_audit_flags_a_reviewed_strip_reached_by_overlap( + tmp_path: pathlib.Path, +) -> None: + # Same spelling, same concept, already reviewed — but this observation + # declares a month, so the reviewed strip gains a kind. + monthly = _row( + CLAIMS, + rid=f"{CLAIMS}.week_2026-08-22", + unit="thousands", + period={"type": "month", "value": "2026-08"}, + entity={"name": "person", "role": "ui_initial_claimant"}, + ) + problems = _strip_audit(tmp_path, [_weekly("2026-08-22")], [monthly]) + assert len(problems) == 1 + assert "reviewed strip gained kinds ['overlap']" in problems[0] + + +def test_strip_audit_flags_a_reviewed_strip_that_is_gone( + tmp_path: pathlib.Path, +) -> None: + reviewed = { + "week_2026-08-22": {CLAIMS: ["derived"]}, + "week_2026-08-29": {"dol.eta.continued_claims.sa": ["derived"]}, + } + problems = _strip_audit( + tmp_path, [_weekly("2026-08-22")], [], reviewed=reviewed + ) + assert problems == [ + "'week_2026-08-29' on 'dol.eta.continued_claims.sa': reviewed strip " + "is gone from the catalog — a superseded observation, curation, or " + "a builder change; re-review and update the reviewed map" + ] + + +# --- Properties of the stripped-segment audit -------------------------------- +# +# Invariants, for every input: +# T1 a spelling's template depends only on how it is written, never on the +# period it names (same family, any two dates -> same template); +# T2 different ways of writing a period get different templates; +# T3 every family used below spells its own period directly; +# A1 routine continuation passes: appending further periods of observed +# series, each written as that series already writes them, never +# creates a problem (end to end, through the real builder); +# A2 soundness: a strip on a concept with no reviewed strip of its +# template, or with an unreviewed overlap kind, is always reported by +# name (end to end); +# A3 persistence: every reviewed pair missing from the catalog is +# reported; +# A4 monotone in review: reviewing more live strips never adds problems, +# and reviewing every live strip as it is leaves none; +# A5 the output is sorted and independent of input ordering. +# Runs are derandomized so the required check stays deterministic. + +AUDIT_SETTINGS = settings( + derandomize=True, + deadline=None, + max_examples=60, + suppress_health_check=[HealthCheck.too_slow], +) + +_DATES = st.dates(min_value=dt.date(1901, 1, 1), max_value=dt.date(2998, 12, 1)) + + +def _month(d: dt.date) -> str: + return bsc.MONTHS_FULL[d.month - 1] + + +def _quarter(d: dt.date) -> int: + return (d.month - 1) // 3 + 1 + + +def _week(d: dt.date) -> dict: + return {"type": "week_ending", "value": d.isoformat()} + + +def _monthly(d: dt.date) -> dict: + return {"type": "month", "value": f"{d:%Y-%m}"} + + +def _quarterly(d: dt.date) -> dict: + return {"type": "quarter", "value": f"{d:%Y-%m}"} + + +# family -> (spelling of a date, the period that spelling denotes directly) +FAMILIES = { + "week_iso": (lambda d: f"week_{d.isoformat()}", _week), + "week_us": (lambda d: f"week_{d:%Y_%m_%d}", _week), + "week_ending_iso": (lambda d: f"week_ending_{d.isoformat()}", _week), + "week_ending_us": (lambda d: f"week_ending_{d:%Y_%m_%d}", _week), + "month_iso": (lambda d: f"{d:%Y-%m}", _monthly), + "month_us": (lambda d: f"{d:%Y_%m}", _monthly), + "month_name": (lambda d: f"{_month(d)}_{d.year}", _monthly), + "month_abbrev": ( + lambda d: f"{bsc.MONTHS_ABBREV[d.month - 1]}_{d.year}", _monthly, + ), + "quarter_q_first": (lambda d: f"q{_quarter(d)}_{d.year}", _quarterly), + "quarter_year_first": (lambda d: f"{d.year}_q{_quarter(d)}", _quarterly), + "fiscal": ( + lambda d: f"fy{d.year}", + lambda d: {"type": "fiscal_year", "value": d.year}, + ), + "annual": ( + lambda d: f"{d.year}", + lambda d: {"type": "year", "value": str(d.year)}, + ), +} +# Spellings only the overlap grammar produces (never a direct variant). +SHAPES_ONLY = { + "day_iso": lambda d: d.isoformat(), + "day_us": lambda d: f"{d:%Y_%m_%d}", + "after_month": lambda d: f"after_{_month(d)}_{d.year}", + "after_mpc_month": lambda d: f"after_mpc_{_month(d)}_{d.year}", + "month_range": lambda d: f"january_to_{_month(d)}_{d.year}", +} +ALL_SHAPES = {name: spell for name, (spell, _) in FAMILIES.items()} +ALL_SHAPES.update(SHAPES_ONLY) +# Full and abbreviated month names deliberately share one template. +SAME_TEMPLATE = {frozenset({"month_name", "month_abbrev"})} + + +@AUDIT_SETTINGS +@given(st.sampled_from(sorted(ALL_SHAPES)), _DATES, _DATES) +def test_template_depends_only_on_the_spelling_family(name, d1, d2) -> None: + spell = ALL_SHAPES[name] + first, second = spell(d1), spell(d2) + assert bsc._is_strippable_spelling(first) + assert bsc.period_spelling_template(first) == bsc.period_spelling_template( + second + ) + + +@AUDIT_SETTINGS +@given( + st.sampled_from(sorted(ALL_SHAPES)), + st.sampled_from(sorted(ALL_SHAPES)), + _DATES, + _DATES, +) +def test_template_separates_spelling_families(a, b, d1, d2) -> None: + same = bsc.period_spelling_template( + ALL_SHAPES[a](d1) + ) == bsc.period_spelling_template(ALL_SHAPES[b](d2)) + assert same == (a == b or frozenset({a, b}) in SAME_TEMPLATE) + + +@AUDIT_SETTINGS +@given(st.sampled_from(sorted(FAMILIES)), _DATES) +def test_every_family_spells_its_period_directly(name, d) -> None: + spell, period = FAMILIES[name] + assert spell(d) in bsc.period_token_variants(period(d)) + assert bsc.classify_segments(f"x.{spell(d)}", period(d)) == [ + ("x", "kept"), (spell(d), "derived"), + ] + + +def _step(name: str, d: dt.date, n: int) -> dt.date: + """A date in the n-th period after the one holding d, for ``name``.""" + if name.startswith("week"): + return d + dt.timedelta(days=7 * n) + if name in ("fiscal", "annual"): + return dt.date(d.year + n, 1, 1) + months = 3 if name.startswith("quarter") else 1 + total = d.year * 12 + d.month - 1 + months * n + return dt.date(total // 12, total % 12 + 1, 1) + + +CONCEPT_POOL = [f"agency{i}.series{i}.rate" for i in range(6)] + +_series = st.lists( + st.tuples( + st.sampled_from(CONCEPT_POOL), + st.sampled_from(sorted(FAMILIES)), + st.dates(min_value=dt.date(2000, 1, 1), max_value=dt.date(2090, 1, 1)), + st.integers(min_value=1, max_value=3), # observed periods + st.integers(min_value=1, max_value=3), # appended periods + st.booleans(), # the period is also spelled in measure.concept + ), + min_size=1, + max_size=4, + unique_by=lambda spec: spec[0], +) + + +def _series_rows(specs, appended: bool) -> list[dict]: + rows = [] + for concept, name, start, observed, extra, in_concept in specs: + spell, period = FAMILIES[name] + steps = range(observed, observed + extra) if appended else range( + observed + ) + for n in steps: + d = _step(name, start, n) + rows.append(_row( + f"{concept}.{spell(d)}" if in_concept else concept, + rid=f"{concept}.{spell(d)}.first_print", + period=period(d), + )) + return rows + + +@AUDIT_SETTINGS +@given(_series) +def test_routine_appends_never_need_review(specs) -> None: + with tempfile.TemporaryDirectory() as tmp: + reviewed, catalog, problems = _strip_builds( + pathlib.Path(tmp), + _series_rows(specs, appended=False), + _series_rows(specs, appended=True), + ) + assert problems == [] + # Not vacuous: every appended period adds a strip nobody reviewed. + new_pairs = _pairs(catalog["stripped_segments"]) - _pairs(reviewed) + assert len(new_pairs) >= sum(extra for *_, extra, _ in specs) + + +# The period each family's spelling only overlaps: coarser than its own. +COARSER = { + "week": _monthly, + "month": _quarterly, + "quarter": lambda d: {"type": "year", "value": str(d.year)}, + "fiscal": lambda d: {"type": "year", "value": str(d.year)}, +} + + +@AUDIT_SETTINGS +@given( + _series, + st.sampled_from(["new_concept", "new_template", "overlap"]), + st.data(), +) +def test_unreviewed_strips_are_always_reported(specs, novelty, data) -> None: + concept, family, start, _, _, _ = specs[0] + other = {"name": "person", "role": "second_entity"} + if novelty == "new_concept": + concept = "novel.series.rate" + name = data.draw(st.sampled_from(sorted(FAMILIES)), label="family") + d = data.draw(_DATES, label="date") + spelling = FAMILIES[name][0](d) + row = _row( + concept, rid=f"{concept}.{spelling}", period=FAMILIES[name][1](d) + ) + elif novelty == "new_template": + seen = bsc.period_spelling_template(FAMILIES[family][0](start)) + fresh = [ + f for f in sorted(FAMILIES) + if bsc.period_spelling_template(FAMILIES[f][0](start)) != seen + ] + name = data.draw(st.sampled_from(fresh), label="fresh family") + spelling = FAMILIES[name][0](start) + row = _row( + concept, rid=f"{concept}.{spelling}", + period=FAMILIES[name][1](start), entity=other, + ) + else: + if family == "annual": # a bare year overlaps nothing; use a month + family, concept = "month_us", "novel.series.rate" + # Either the very spelling already reviewed (it gains a kind) or a + # later period in the reviewed template (a new overlap strip): both + # must be reported once an observation reaches them by overlap. + when = _step(family, start, data.draw( + st.sampled_from([0, 7]), label="periods after the first" + )) + spelling = FAMILIES[family][0](when) + coarser = COARSER[family.split("_")[0]](when) + row = _row( + concept, rid=f"{concept}.{spelling}", period=coarser, entity=other + ) + with tempfile.TemporaryDirectory() as tmp: + problems = _strip_audit( + pathlib.Path(tmp), _series_rows(specs, appended=False), [row] + ) + assert _names(problems, spelling, concept), (novelty, spelling, problems) + + +_spelling = st.builds( + lambda name, d: ALL_SHAPES[name](d), + st.sampled_from(sorted(ALL_SHAPES)), + _DATES, +) +_KINDS = st.sampled_from([["derived"], ["overlap"], ["derived", "overlap"]]) +_live_kinds = st.dictionaries( + _spelling, + st.dictionaries(st.sampled_from(CONCEPT_POOL), _KINDS, min_size=1, + max_size=3), + max_size=6, +) + + +def _kinds_of(pairs, kinds) -> dict[str, dict[str, list[str]]]: + out: dict[str, dict[str, list[str]]] = {} + for s, c in pairs: + out.setdefault(s, {})[c] = kinds[s][c] + return out + + +@AUDIT_SETTINGS +@given(_live_kinds, st.data()) +def test_audit_is_monotone_in_review_and_order_free(kinds, data) -> None: + live = {s: sorted(by) for s, by in kinds.items()} + pairs = sorted(_pairs(live)) + subsets = st.lists(st.sampled_from(pairs), unique=True) if pairs else ( + st.just([]) + ) + chosen = data.draw(subsets, label="reviewed") + more = data.draw(subsets, label="more reviewed") + audit = bsc.stripped_segment_review_problems + fewer = audit(live, kinds, _kinds_of(chosen, kinds)) + most = audit(live, kinds, _kinds_of(set(chosen) | set(more), kinds)) + assert set(most) <= set(fewer) # A4 + assert audit(live, kinds, kinds) == [] # A4 + assert fewer == sorted(fewer) # A5 + shuffled_live = {s: live[s][::-1] for s in reversed(list(live))} + shuffled_kinds = { + s: {c: kinds[s][c] for c in reversed(list(kinds[s]))} + for s in reversed(list(kinds)) + } + assert audit( + shuffled_live, shuffled_kinds, _kinds_of(chosen[::-1], kinds) + ) == fewer # A5 + # A3: a reviewed pair that is not live is reported as gone. + ghost = {**_kinds_of(chosen, kinds), "fy1999": {"ghost.series": ["derived"]}} + reported = _names(audit(live, kinds, ghost), "fy1999", "ghost.series") + assert len(reported) == 1 and "is gone" in reported[0] + + def test_rebuild_without_prior_catalog_is_gated( tmp_path: pathlib.Path, ) -> None: diff --git a/uv.lock b/uv.lock index 3a5807ed..1d609318 100644 --- a/uv.lock +++ b/uv.lock @@ -635,6 +635,94 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/48/30/47d0bf6072f7252e6521f3447ccfa40b421b6824517f82854703d0f5a98b/hyperframe-6.1.0-py3-none-any.whl", hash = "sha256:b03380493a519fce58ea5af42e4a42317bf9bd425596f7a0835ffce80f1a42e5", size = 13007, upload-time = "2025-01-22T21:41:47.295Z" }, ] +[[package]] +name = "hypothesis" +version = "6.168.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "sortedcontainers" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/09/b7/13118bbc45d6d8b9d04e2de779e2a4ff23145ea39efa692b994fb874ca72/hypothesis-6.168.3.tar.gz", hash = "sha256:a43388f9067678fef6e13bdff325b6cfa6961a590498bb37f7ff31589c83bc75", size = 511022, upload-time = "2026-09-28T05:20:58.499Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/2b/cd/746a2fc1e5e5ba34f07ea6ee1ad07525a3dde5ff4836db3a7980af49a891/hypothesis-6.168.3-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:d20972ca134e652e9928ecd200966a8a2857adc2812c1d74b12a872e5cd503eb", size = 791536, upload-time = "2026-09-28T05:18:57.162Z" }, + { url = "https://files.pythonhosted.org/packages/c3/0f/7a158e377b69556c8e12c25c8fd811d0103e54c24a12dab1f3b9819f202e/hypothesis-6.168.3-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:eafbec09d3e87d13d8242411d1f5868b5e879f1bd6d95e233e9ff80c527bea1c", size = 787313, upload-time = "2026-09-28T05:20:18.122Z" }, + { url = "https://files.pythonhosted.org/packages/45/c6/7df9104ee359e8fcd77791781c47e70794964ca69673665e0de2a6590f4c/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:73c5627497968cc62d140e9fde1ea21e12cac7b64dc419fea8286cd34ff1ad4e", size = 1124036, upload-time = "2026-09-28T05:20:04.372Z" }, + { url = "https://files.pythonhosted.org/packages/92/0f/8a61715404a73b9a82cb1f976803428d603cdea976a9c30266c72ad6a7ce/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:29dc56e6dc6eeb0aeb08ac463f847279bde2336c4786faf7ee0af4b93cd031a7", size = 1147875, upload-time = "2026-09-28T05:19:15.495Z" }, + { url = "https://files.pythonhosted.org/packages/e4/3f/2b16e95cc3b9a069afa23830a012aefe377ce4a457a227e5a5026eb827ff/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:753bb501f8560d2e321ed3b62596b4496c56f15668b0a0669231ccc3e6c4e80d", size = 1149489, upload-time = "2026-09-28T05:19:18.886Z" }, + { url = "https://files.pythonhosted.org/packages/f6/e7/d7cf6dc068bb02732b2a6e38a35a0dcb7f2137608198fc79d740f69d197c/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:71ab606c472449cb872ef2a7acaec679ee4f651b7f0a477a04ebc85d8933ef8c", size = 1191992, upload-time = "2026-09-28T05:19:21.83Z" }, + { url = "https://files.pythonhosted.org/packages/5f/26/f1e5b25dec15998e8375d722825ec3dae1075cbe7cb1f74221b2312a2e33/hypothesis-6.168.3-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:26928956c54748e4dfa587333741ab750246123b93484955646ff316cca27eef", size = 1169911, upload-time = "2026-09-28T05:19:27.996Z" }, + { url = "https://files.pythonhosted.org/packages/03/36/901234a49147e6bf8a5b9e54b775706a223c2b52befea1106ea2aaf15d48/hypothesis-6.168.3-cp310-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:0bdfc53c041b61c854fa3761735736b991bc2bddafee45a361cfe4d43027c1ef", size = 1129406, upload-time = "2026-09-28T05:18:42.632Z" }, + { url = "https://files.pythonhosted.org/packages/55/c1/5fc9ff91ec9fba6bb527833386c66812524b5cc75fe7d4a4f9122ee1c420/hypothesis-6.168.3-cp310-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:650528e2b1e2a45e95c2df624d4b9364b8ac5a027ba84e1eea2bc5398dd01dcd", size = 1160399, upload-time = "2026-09-28T05:20:23.911Z" }, + { url = "https://files.pythonhosted.org/packages/9b/d2/49ad1ef5c55547c8abfd0fc571b3493806dda530fd5cb3d8263dbad86a72/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:209dc54cdb1b4d6d7020ad8d09e44c49756361b7183adacea4d6f5705085b595", size = 1299906, upload-time = "2026-09-28T05:18:58.417Z" }, + { url = "https://files.pythonhosted.org/packages/0d/5f/f98a26094c4f462c4215807bed0f9a6b5508b27a3c329bcba1bc8adc7ac7/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:af8ca98cd11dc7f9427bb90483abcd9688d4df2e8e63893a30a2423b027ebb11", size = 1425538, upload-time = "2026-09-28T05:20:49.074Z" }, + { url = "https://files.pythonhosted.org/packages/d0/b8/c5b4406e3a6591e51a42f85469351d1d824e0a54b84167aec9b2353759fe/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:f8122cfdd0bc0ba843063effa22ac310bdebdb3a1319bf1e63f14baa119c72ab", size = 1377084, upload-time = "2026-09-28T05:20:06.618Z" }, + { url = "https://files.pythonhosted.org/packages/e0/01/0d84a8ea469d024c15f602799c0b8410fbe03d99fdb0370449b33c3c916f/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:4bedbb379eab34f792af7ee9a05aae04e9c08bbb52e0f90f5a2110d4fe4b2fbd", size = 1281162, upload-time = "2026-09-28T05:19:42.768Z" }, + { url = "https://files.pythonhosted.org/packages/22/b5/ea6435038de1a795f005cb603531e8ccc053febc50180f3ab55583c9c60e/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:5a953115b9f5c95133ab2d04efffeec96e5658c3207923dca7285f7db3e6bbef", size = 1300433, upload-time = "2026-09-28T05:18:55.634Z" }, + { url = "https://files.pythonhosted.org/packages/6c/78/fa6c77d9f64b6d592e69e582d83cbce7d7e56897faa4762885a2123b0166/hypothesis-6.168.3-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:6173558e676ad25ed1e20507fa4024a0d816dd90f715f77006e5a28f19109026", size = 1336273, upload-time = "2026-09-28T05:19:08.273Z" }, + { url = "https://files.pythonhosted.org/packages/f1/56/4acd0d778bab818c9ea336fcba768bf43bc70db7f84cc373579298df4641/hypothesis-6.168.3-cp310-abi3-win32.whl", hash = "sha256:dc66390fb12d80585aa9222bf538ce8b7aa22cf5d118250647355a1c9e8f62f4", size = 678211, upload-time = "2026-09-28T05:19:23.355Z" }, + { url = "https://files.pythonhosted.org/packages/b0/db/0a02b146ad1c30f9716362f2e1a379dfd6b0f7be03ac1e22c9da24c9e2a9/hypothesis-6.168.3-cp310-abi3-win_amd64.whl", hash = "sha256:92325b276360fe86c5bf71a568c0d53a6d140b0de36dcf029f9164a17803bb24", size = 684906, upload-time = "2026-09-28T05:18:35.099Z" }, + { url = "https://files.pythonhosted.org/packages/a3/c0/958deaf726848f96f52250740bf39f13f476b068e22f73e2b358eaa07532/hypothesis-6.168.3-cp310-abi3-win_arm64.whl", hash = "sha256:3cf6f1eeaf41cd8d60cf1f88fde905ca1dd77c906929a507d6ac7f66f2ccba2a", size = 683292, upload-time = "2026-09-28T05:19:57.404Z" }, + { url = "https://files.pythonhosted.org/packages/f4/ab/443208c2988a24eaafbfbd6791b897ad9a8a1dd7f870d5b455d7d15a28ad/hypothesis-6.168.3-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:9029775dc2e25e3e8e596c4306926f0079158353631c7c52c285bbc5c8073c2b", size = 792239, upload-time = "2026-09-28T05:19:59.144Z" }, + { url = "https://files.pythonhosted.org/packages/cf/7b/619c1a76c52de2d3cfd68a2f9876c09464d63097db5bf86a28b62dc37bb1/hypothesis-6.168.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:0e1d91530cefdb9e0b6467c203eb509c61a11d70480a37db59cb53a2f9913102", size = 788113, upload-time = "2026-09-28T05:19:17.145Z" }, + { url = "https://files.pythonhosted.org/packages/ea/dd/33841e8833105f850214a486d83740ba52041892ca94e687a05829c0ce83/hypothesis-6.168.3-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0df00cbe8133aa220308d11fa28e3add996cf5102a39b279c1d711015fd9114b", size = 1124241, upload-time = "2026-09-28T05:18:37.589Z" }, + { url = "https://files.pythonhosted.org/packages/4b/43/c84a7900feced6a420e824ab61d316e4f59a6887d991bc143ecc294a09ed/hypothesis-6.168.3-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:fd7f75a2e23288ee82ee965a952473d9c5447c2cbc1afc94be09d2402201774e", size = 1170406, upload-time = "2026-09-28T05:19:46.365Z" }, + { url = "https://files.pythonhosted.org/packages/47/5d/6e2a32b9c3020af7a06b00fe446a2c5155ae76ac468c0b68d7188e524002/hypothesis-6.168.3-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:bc2b1c37631f2231dd0d077225aa5908fc2bd34a4c3faf475053b0bc2eb24c95", size = 1300339, upload-time = "2026-09-28T05:20:31.968Z" }, + { url = "https://files.pythonhosted.org/packages/f9/78/19f3c9eb9a5c215b9fbde57be18ef7c1c78afa509b460d4e8f86026d741a/hypothesis-6.168.3-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:73960a58f6efc8cbc57d8ad0647955f7c5b25f3d562e3eaa57afa1c69894f7aa", size = 1336484, upload-time = "2026-09-28T05:19:02.841Z" }, + { url = "https://files.pythonhosted.org/packages/88/bb/d46c4519e989e022667705d8a56e83a09ad4ea50eae07ca4bb0f4708dd23/hypothesis-6.168.3-cp311-cp311-win_amd64.whl", hash = "sha256:3f1122223759acc0fe5301c1ae505d762ee61af91db7b191a791ccdbcd965462", size = 684646, upload-time = "2026-09-28T05:20:00.842Z" }, + { url = "https://files.pythonhosted.org/packages/7a/56/44a17266bcf1add5fba7c3df5c3920396a5e924d917fb0534a6f9e2b9b9a/hypothesis-6.168.3-cp312-cp312-macosx_10_12_x86_64.whl", hash = "sha256:bff12e036b67abc5ded682f724f4d7afaff5223190270a0763695f218c079068", size = 793313, upload-time = "2026-09-28T05:20:22.045Z" }, + { url = "https://files.pythonhosted.org/packages/d8/18/1a25b11e9133cb54c02ee6cf99a61296ad461aabc7aae3399e1d0ff1f095/hypothesis-6.168.3-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:c73c6188056e6dc110260e39d141c69c3d02f23e3ba1d8610c7a8b6f2510ec96", size = 784855, upload-time = "2026-09-28T05:19:34.496Z" }, + { url = "https://files.pythonhosted.org/packages/8b/da/728d3210f0b0089337f37206a100d31ef546d1f99839e709f0a9ca443f8e/hypothesis-6.168.3-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:bb1063cb794765097b8c0add598bd34a904f44dc15db81275dd1f0ed1ac567ed", size = 1123043, upload-time = "2026-09-28T05:19:05.468Z" }, + { url = "https://files.pythonhosted.org/packages/53/86/bfb9ec344a8b3ef2fa6f59bb3d30c5b3a7d86d594107670539c34ee53d1b/hypothesis-6.168.3-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9b2d47f5d9060b049039bef0b267e88480f6468f95ee394e514e84a81ccdc5f5", size = 1169121, upload-time = "2026-09-28T05:20:40.512Z" }, + { url = "https://files.pythonhosted.org/packages/3b/e0/5a76dbafbcaf8fff075f09377cddefa17289ef05dbf51878dde88385b966/hypothesis-6.168.3-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:8367f623a98cbf90f33fd2a43b100671b152b5975bc343a578290945f52f7018", size = 1298749, upload-time = "2026-09-28T05:19:32.883Z" }, + { url = "https://files.pythonhosted.org/packages/28/15/0e65af96b353f2c4bdc283c53ed0980e5ac5009afe97ef3b23c77b364664/hypothesis-6.168.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:fd81241f4cd76fb18d481e87b9688b30a0cb5c30841730513323ba1c7aba1837", size = 1335271, upload-time = "2026-09-28T05:20:54.005Z" }, + { url = "https://files.pythonhosted.org/packages/f5/c5/37eab2155664aa544df321f6d3940b89c6702d79a72fdc98d7b5a2286f8c/hypothesis-6.168.3-cp312-cp312-win_amd64.whl", hash = "sha256:377438de53afb94347d9845b7d90db5905c6d2a8d612108e9938e0f80503bb90", size = 682211, upload-time = "2026-09-28T05:18:51.206Z" }, + { url = "https://files.pythonhosted.org/packages/0c/a2/6787da846d929e52fc3344d803299c45782dbeac287528af08380e984bf8/hypothesis-6.168.3-cp313-cp313-macosx_10_12_x86_64.whl", hash = "sha256:1b230a850de63334c16654a34a2d547e0179d36b9071d4439b3e7237f6d077e7", size = 793254, upload-time = "2026-09-28T05:19:36.294Z" }, + { url = "https://files.pythonhosted.org/packages/89/ba/2893f5ca42501f4d562cba3229c8994c3a3e5fac66ef60c1e0e0b5220a5d/hypothesis-6.168.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d63b0226cd3e0d8bdd97c3384b22a21934ed4d53246c1c93575dada616672499", size = 784699, upload-time = "2026-09-28T05:18:54.194Z" }, + { url = "https://files.pythonhosted.org/packages/b7/aa/7d7349daf75b71f6f35876f8de115e974c5c04d600e86df1b2779b0883ab/hypothesis-6.168.3-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0369f5df055f96e117ab12e5f249668ff144731ab5280bc7a205fdf81f990b89", size = 1123016, upload-time = "2026-09-28T05:20:36.372Z" }, + { url = "https://files.pythonhosted.org/packages/00/f0/7774e1ea072708ea46cb25c4aeee5f9978b6c271764809069e8a6789858e/hypothesis-6.168.3-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f076bcd0f77fdcdb7797826c879099573d03a02e228ebe649ea917081b962ac0", size = 1168989, upload-time = "2026-09-28T05:20:08.44Z" }, + { url = "https://files.pythonhosted.org/packages/a7/7d/113992abed9efbd7944e3a496a382da58152dce126fddf6419ff31a0a57d/hypothesis-6.168.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:b823ba1fcec8da730f29316d010b06d3f7e0c3828dcf91020e24c55e7d24652a", size = 1298626, upload-time = "2026-09-28T05:19:40.986Z" }, + { url = "https://files.pythonhosted.org/packages/d5/65/3659fa5e733027e5b37a27486e57f4053535e40853d23501c422bb275043/hypothesis-6.168.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:dd2849c269d674e4618590f3b48d443bd4c06b5aef3d8d869086c2e6d213d248", size = 1335184, upload-time = "2026-09-28T05:20:20.087Z" }, + { url = "https://files.pythonhosted.org/packages/b4/6c/35aab2221b5ea65125340f236e1a765a9e9f3b28d89a389ea0033fed29f7/hypothesis-6.168.3-cp313-cp313-win_amd64.whl", hash = "sha256:3ef7d26f5789e691401d5f87eafed9bd2763f0dbe47af6d6b66012a509404766", size = 682173, upload-time = "2026-09-28T05:19:12.762Z" }, + { url = "https://files.pythonhosted.org/packages/6e/79/27b0cb56ff5d2bc92458bf6fcebdb1f0dc01f57f043a178e9f9d74498169/hypothesis-6.168.3-cp314-cp314-macosx_10_12_x86_64.whl", hash = "sha256:e2d4c68729a13df9af4998d2652cfb5d541c5609b88c306880a5dfb284938ec2", size = 793287, upload-time = "2026-09-28T05:20:38.522Z" }, + { url = "https://files.pythonhosted.org/packages/e8/bd/5342f95c3bc36586777ef8cb8eba87ac7acf0184210327fb0e6d8654e295/hypothesis-6.168.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:b345f818083ec99966a43ca4f7b38feb62c6920ce28572bc2bde4948a8b7eaba", size = 784857, upload-time = "2026-09-28T05:20:42.609Z" }, + { url = "https://files.pythonhosted.org/packages/4a/5f/fc774241518a0588d362680a3084275058aab86d6bc02a2c61e1073142d0/hypothesis-6.168.3-cp314-cp314-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0608c610fc002978fc5de0f471770d8817e9455cb8649a47983c9f8bccfc1897", size = 1123330, upload-time = "2026-09-28T05:19:29.735Z" }, + { url = "https://files.pythonhosted.org/packages/b0/e6/131f16775a3dca5f4fe27f0d6ad9a9851600098a54a872d163f6ebf0b664/hypothesis-6.168.3-cp314-cp314-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9dcf6448b1ecc37f2b23f2d1b3ddfc9ff6b6910614c15a642dc82f419b96037b", size = 1169178, upload-time = "2026-09-28T05:19:53.436Z" }, + { url = "https://files.pythonhosted.org/packages/06/61/c9f5bff8b73321c9fd12f2d69666c6e4861b97b60f5f24fdd6ac293007fd/hypothesis-6.168.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:ae4f9f094041dcce5119ebd7bab71062743ab02b6e654d056b370beda78c19e2", size = 1299127, upload-time = "2026-09-28T05:20:14.486Z" }, + { url = "https://files.pythonhosted.org/packages/e4/6d/d4618f8ab12dd76c4d58ea052252759aa2bf0c4e71f4b502ba5d577e0b9f/hypothesis-6.168.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:d479985fe73af97badfb72dc6d20c6a353e486a36f6d065f600026e0cf954a86", size = 1335390, upload-time = "2026-09-28T05:20:10.326Z" }, + { url = "https://files.pythonhosted.org/packages/95/b6/727161e17cc297df1aeea945de05c532033b66751625bec192bc5b5ff707/hypothesis-6.168.3-cp314-cp314-pyemscripten_2026_0_wasm32.whl", hash = "sha256:785e2c45f8c08e274b4bf1ccae97f1a4e09407e9a68e80790cfab1d17a4a45fa", size = 624291, upload-time = "2026-09-28T05:18:36.376Z" }, + { url = "https://files.pythonhosted.org/packages/77/23/2f9b506392a0b8712440bcb781abc5713be9ffac1303b6d10f3669050882/hypothesis-6.168.3-cp314-cp314-win_amd64.whl", hash = "sha256:320920b1e3dae8611eee8a03d063cf2187446f2a17c38cfb8a7fc1466f71eee2", size = 682064, upload-time = "2026-09-28T05:20:16.336Z" }, + { url = "https://files.pythonhosted.org/packages/50/93/efabfd95eb2b69c0c9fa1e1c83c8290aa2715d46baeb5a7764faf09c133a/hypothesis-6.168.3-cp314-cp314t-macosx_10_12_x86_64.whl", hash = "sha256:35380baa981108a7f60c4eab71e46acd8d8f58440520346a1a6aba06dca7e074", size = 791875, upload-time = "2026-09-28T05:19:37.789Z" }, + { url = "https://files.pythonhosted.org/packages/1b/fc/2a0ada1623a9a33048bbf9f00a7c92513398c6b8865cba5efb854262ce30/hypothesis-6.168.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:f0aaaed00438fa6673d856aaee12a4a8afbde82a9f3c5ff487cacfb064239afe", size = 783412, upload-time = "2026-09-28T05:18:48.266Z" }, + { url = "https://files.pythonhosted.org/packages/41/31/73c615d37eeb12da0209555612c32ced628d860267c23cfd2691c6207aea/hypothesis-6.168.3-cp314-cp314t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:f27e6df1576bf838e7f484d4ae5cba92114497ca30afb917056bf1ea33e95675", size = 1121662, upload-time = "2026-09-28T05:19:26.225Z" }, + { url = "https://files.pythonhosted.org/packages/92/54/1893344a3b8bbf83fa7c9bb7e96f6836580213854f46671cb749f7b4706c/hypothesis-6.168.3-cp314-cp314t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3bc85014577982ec6d2e266edc5cd6e7a7d3674c648791ca983c67a52f89a0c4", size = 1167761, upload-time = "2026-09-28T05:18:49.85Z" }, + { url = "https://files.pythonhosted.org/packages/26/97/a61d82968febd25a0f08ffbbfabfaeead66f68fc5fa73f3c9ece1c36fc8f/hypothesis-6.168.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:ccfc29505aa1cdcc254cb9cd701fd0811fef5480addd97d4127a00df023410f7", size = 1297309, upload-time = "2026-09-28T05:19:51.711Z" }, + { url = "https://files.pythonhosted.org/packages/bc/01/fe8e4cf6d39d02f4efadaa351427fa13722086cfc54a2da3579741c5177a/hypothesis-6.168.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:32d0699566aaa93f9e97a44705d7164386f1e91de78d277f53bada35e78bcc96", size = 1334260, upload-time = "2026-09-28T05:19:00.121Z" }, + { url = "https://files.pythonhosted.org/packages/39/6a/09177ebc62f4778f94379ecade6a51299d3d8c4f4b75ba636bf3a0202364/hypothesis-6.168.3-cp314-cp314t-win_amd64.whl", hash = "sha256:28d88fa174ecbd4ecbd7bb290f06d0db084a3971c3f511ae2830b65b5e25500f", size = 681992, upload-time = "2026-09-28T05:19:48.043Z" }, + { url = "https://files.pythonhosted.org/packages/f9/f6/890bf33d608cd63348d3146a3ca83362fbf09e589c5e5230a00c403070f4/hypothesis-6.168.3-cp315-abi3.abi3t-macosx_10_12_x86_64.whl", hash = "sha256:61f5782d807b1e6aa5c9beef1054cb2037e7d1add48a64972cd6ee447778781f", size = 791231, upload-time = "2026-09-28T05:18:43.998Z" }, + { url = "https://files.pythonhosted.org/packages/03/dd/fc36b204f8aa7437c676576d9437f0422de46aa8a729e6cd5bbb0224e759/hypothesis-6.168.3-cp315-abi3.abi3t-macosx_11_0_arm64.whl", hash = "sha256:7aacf3cf40c7ce8f9e4924d5b57bc0b068beafdfed2347bcf160a28b4cfbba7b", size = 783171, upload-time = "2026-09-28T05:19:20.353Z" }, + { url = "https://files.pythonhosted.org/packages/5c/d2/3cd087d577db67c404991d7723e64d7cd75570cdc82a0f7329aed6f2086c/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:01768a03a30dc54df7fe457c0b34016c84d598fa00d5965ededab93ba4eb3408", size = 1121152, upload-time = "2026-09-28T05:18:39.095Z" }, + { url = "https://files.pythonhosted.org/packages/42/6f/a05a68cc66e7a22701c50bf94fb9ddeb36f4f5d0de3cc130066709eda621/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:1a8a4ffc6c6e6e577f2bfbcfebf7cffbb310283ba2532c729570a0a752cebaaa", size = 1144069, upload-time = "2026-09-28T05:20:44.705Z" }, + { url = "https://files.pythonhosted.org/packages/4a/f4/fdd7a093fb2fb3a0ce24256c92ad5bf67936f4669bd06d5ceae4298150ec/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:b987d73eba95183a7e59cca6d1925c588aa4307922d9852cc2b4d282a2ae4128", size = 1146692, upload-time = "2026-09-28T05:20:25.83Z" }, + { url = "https://files.pythonhosted.org/packages/f1/c8/6dbd4377e935505ae4fc8ee4c7b18c69994c4bbaca015447c470218bdbee/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:d4569c39bd97d9573e946429ed676f3b55a7c8ed80920addd67d384a47d59b38", size = 1189480, upload-time = "2026-09-28T05:20:46.734Z" }, + { url = "https://files.pythonhosted.org/packages/ba/bf/ff288b496b690000d2686dcfa7f67855e4c5a6dddb464ed8e4880be83bb2/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:6042b8707a4b25bbbfe20b110258fb7549a68951e5525608a8ce06091039e9fc", size = 1167109, upload-time = "2026-09-28T05:19:24.758Z" }, + { url = "https://files.pythonhosted.org/packages/d9/d1/3fee2bc29fc747fabbb27e174515592e390809ca465d3bbbabbcfa3235a8/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_31_riscv64.whl", hash = "sha256:f413629de94d38a7a2ad259697a6752526143cb49c2e2d7bdba19381c80693f6", size = 1126823, upload-time = "2026-09-28T05:18:52.636Z" }, + { url = "https://files.pythonhosted.org/packages/58/6a/585294392fa6a6d9719446a042a777ba98ecca265d9e48cbedc10c88df0f/hypothesis-6.168.3-cp315-abi3.abi3t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:f071737e4e775bebba07e319eb645a880d1e0186b4d24bad23255e849d86d483", size = 1155773, upload-time = "2026-09-28T05:20:51.268Z" }, + { url = "https://files.pythonhosted.org/packages/cd/4d/fa283ff79996debf1ff08593e01f8b773eb8211b19b3d6d5d491317a65bf/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_aarch64.whl", hash = "sha256:f59a3912858d0609c26054aa1c474847c797937f5e0e1c77f0d3f08032e50f09", size = 1296671, upload-time = "2026-09-28T05:18:46.937Z" }, + { url = "https://files.pythonhosted.org/packages/42/05/98f9c2f628da5afabb6c3e8df5d2d299d2d50bf306b14a6e26d3954c4951/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_armv7l.whl", hash = "sha256:f09c05a23a8025dd5cad07c2ed299a46778e72d49ff0dd5bf353bcc0f7dc1580", size = 1422001, upload-time = "2026-09-28T05:19:04.188Z" }, + { url = "https://files.pythonhosted.org/packages/b8/c9/f2109ade29a7ec284d55bca328d784743e8b13e321576fd34ee7e65f99af/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_i686.whl", hash = "sha256:44ace770bda3a0301739fc5a413c764de790c739df1f1a7218dd049f8594d9f5", size = 1374182, upload-time = "2026-09-28T05:20:27.764Z" }, + { url = "https://files.pythonhosted.org/packages/ce/6d/e05d5f72441564a3bebc71fa155deafd0fd3b015d6014ec8a00edf6e42bb/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_ppc64le.whl", hash = "sha256:03b131043608f94a2578a2a896a1a72092acb7079815eef5b8513b08447e0e62", size = 1278402, upload-time = "2026-09-28T05:20:56.332Z" }, + { url = "https://files.pythonhosted.org/packages/96/d7/6988a7f1f69c5c530687dce03a32c157e2d878e3a2e32af9269ce016b78a/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_riscv64.whl", hash = "sha256:e9784aca26eddfe99b03a0292320db742cc8a74200ef864fdce949f526f973cd", size = 1297770, upload-time = "2026-09-28T05:19:11.387Z" }, + { url = "https://files.pythonhosted.org/packages/a2/28/5bb82b60b836bd2329e1fe01ad94efc2b14dfa806f8e50ed14b795d78770/hypothesis-6.168.3-cp315-abi3.abi3t-musllinux_1_2_x86_64.whl", hash = "sha256:2b52ac363096232bebc2add117e9178d91f4248f4cdb919fd1026b5f86a4bb16", size = 1333977, upload-time = "2026-09-28T05:20:02.483Z" }, + { url = "https://files.pythonhosted.org/packages/8e/7d/d419841b8f65481ea1a50c4ba36f2670d48b8c7a51e4e8586089564cf7f6/hypothesis-6.168.3-cp315-abi3.abi3t-win32.whl", hash = "sha256:d28e3a6b511a74ce37df5274b51f36c2b274fea7365e00b16f0c81e22acd5957", size = 675394, upload-time = "2026-09-28T05:19:55.237Z" }, + { url = "https://files.pythonhosted.org/packages/d8/d2/1de6a2ad100e44817f2e9a8e8ba3eeaf4769621f4f87f2d4966c4971fcc4/hypothesis-6.168.3-cp315-abi3.abi3t-win_amd64.whl", hash = "sha256:7b9638789548361a57d984f56619ac694a914c328181d911271618409261ff4a", size = 681699, upload-time = "2026-09-28T05:19:09.957Z" }, + { url = "https://files.pythonhosted.org/packages/c0/79/be3370fd02734d6b1d950183ae9224580d19fa8d356b95348d60f1e3eeb7/hypothesis-6.168.3-cp315-abi3.abi3t-win_arm64.whl", hash = "sha256:65d78e4357ec48ed2c67825f06740ee3599be4cfe770a6092bed07108d679ac5", size = 679849, upload-time = "2026-09-28T05:19:06.884Z" }, + { url = "https://files.pythonhosted.org/packages/3f/0e/27960db1d45a94497e1681fb9d7b9140b44da4e655ba20f0d82f95d10230/hypothesis-6.168.3-pp311-pypy311_pp73-macosx_10_12_x86_64.whl", hash = "sha256:101531b4ccf7fa12965a6b295d10b64c2f8a8fab61e1f9b18c55335d8fb02575", size = 793135, upload-time = "2026-09-28T05:19:14.056Z" }, + { url = "https://files.pythonhosted.org/packages/9a/cb/be08379286c1925fcd651b92ed835f470a83b859b5c484b3431da9d851f5/hypothesis-6.168.3-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:41e0120814de5c3a6b58d8cb27cb11b8873ab33fba130cd09855a27f5a12a1ca", size = 788995, upload-time = "2026-09-28T05:20:34.164Z" }, + { url = "https://files.pythonhosted.org/packages/38/7b/6d885d139b2ed76e2f4908cf10eae56acc73bf05cbef286357c4559c9a04/hypothesis-6.168.3-pp311-pypy311_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:9d120009d145697909f2f0c532f2d568558ddd6990707f8cba6576bfb9770696", size = 1124974, upload-time = "2026-09-28T05:19:49.903Z" }, + { url = "https://files.pythonhosted.org/packages/05/cc/b302ae809b48254830cae57801d0af0b66783df515b5dd56b9f21c8dfa86/hypothesis-6.168.3-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:818b3d09bba60ef90463e47a4bc87275dba1e8aeb184d3102888440ca7c7837d", size = 1171967, upload-time = "2026-09-28T05:20:12.295Z" }, + { url = "https://files.pythonhosted.org/packages/2c/ab/51031080a6600f5b07adf5f9dfa117345c2adda07a5c24bda523e70bd1e9/hypothesis-6.168.3-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:9e3c36eba80e089ecfa28ec08049723223cb41d3cfe4d9988169cf98440e402b", size = 685641, upload-time = "2026-09-28T05:19:44.609Z" }, +] + [[package]] name = "idna" version = "3.11" @@ -1661,6 +1749,7 @@ dependencies = [ [package.optional-dependencies] dev = [ + { name = "hypothesis" }, { name = "pytest" }, { name = "ruff" }, ] @@ -1670,6 +1759,7 @@ policyengine = [ [package.dev-dependencies] dev = [ + { name = "hypothesis" }, { name = "pytest" }, ] @@ -1677,6 +1767,7 @@ dev = [ requires-dist = [ { name = "cryptography", specifier = ">=46.0.3" }, { name = "httpx", specifier = ">=0.27.0" }, + { name = "hypothesis", marker = "extra == 'dev'", specifier = ">=6.100,<7" }, { name = "microplex", specifier = ">=0.1.0" }, { name = "numpy", specifier = ">=1.24.0" }, { name = "odfpy", specifier = ">=1.4.1" }, @@ -1697,7 +1788,10 @@ requires-dist = [ provides-extras = ["dev", "policyengine"] [package.metadata.requires-dev] -dev = [{ name = "pytest", specifier = ">=8.0.0,<9" }] +dev = [ + { name = "hypothesis", specifier = ">=6.100,<7" }, + { name = "pytest", specifier = ">=8.0.0,<9" }, +] [[package]] name = "policyengine-core"