diff --git a/README.md b/README.md index e584a6a02..2539cf930 100644 --- a/README.md +++ b/README.md @@ -44,16 +44,23 @@ until you restart the kernel, load it by path with `load_country_spec(Path(...))` (always re-read), or call `microcosm.build.country_spec._load_packaged_country_spec.cache_clear()`. -## Staging build telemetry - -US fiscal refresh builds emit pre-release staging telemetry **by default**: -progress JSON is uploaded to `policyengine/populace-us-staging` while the build -runs (best-effort — a missing token or failed upload never fails the build), so -every candidate shows up on the staging dashboard before it is published. -Disable with `--no-staging`, or point elsewhere with `--staging-repo-id` / -`POPULACE_STAGING_REPO_ID`. An *empty* `POPULACE_STAGING_REPO_ID` is ignored -rather than read as off, and staging with no destination at all is an argparse -error — `--no-staging` is the only way a build produces no telemetry. +## Build progress and staging run files + +Supported US and UK build commands always start the local telemetry emitter +service. It reports live progress to the hosted collector when the operator's +existing Hugging Face login is accepted; otherwise it retains the events +locally and the build continues. This live event path is independent of the +staging run files described below. + +US fiscal refresh builds also write pre-release staging run files **by +default**. Progress JSON is uploaded to `policyengine/populace-us-staging` +while the build runs (best-effort — a missing token or failed upload never +fails the build), so every candidate shows up on the staging dashboard before +it is published. Disable these files with `--no-staging`, or point them +elsewhere with `--staging-repo-id` / `POPULACE_STAGING_REPO_ID`. An *empty* +`POPULACE_STAGING_REPO_ID` is ignored rather than read as off, and staging with +no destination at all is an argparse error. `--no-staging` does not disable +the local telemetry emitter service. The build manifest records what staging did: the run id, the destination, and how many files actually reached it, or an explicit `enabled: false` for a @@ -75,8 +82,8 @@ The UK commands (`tools/build_uk_frs_spine.py`, a shim over the package's `uk_runtime.spine_build`, and `microcosm-build-uk` / `tools/build_uk_full.py`, whose `--release-role` builds either the national or the dense line; `tools/build_uk_rowwise_candidate.py` is a stub over the same driver) stage -version 2 telemetry to `policyengine/populace-uk-staging` under the same -switch. The build command also **stages the finished dataset bundle** it built, +version 2 staging run files to `policyengine/populace-uk-staging` under the +same switch. The build command also **stages the finished dataset bundle** it built, national, dense or exact-count, under `staged//` in the private `policyengine/populace-uk-private` repository so the team can inspect it without publishing it: `releases/` and `latest.json` are untouched, the @@ -203,7 +210,8 @@ weights, calling the release tool's own gate functions (none is re-implemented), and exits `1` on any certain failure, `2` on AT-RISK only, and `0` when clean. An argparse error also exits `2` but writes no report, so the wrapper returns `64` whenever no report was written. It writes only its -report: nothing under `--out`, no staging telemetry, no receipts. A base or +report: nothing under `--out`, no staging run files, no receipts. The local +telemetry emitter service still reports dry-run progress. A base or donor that the config does not name locally is still downloaded, into the same caches the release uses. A refusal before the stop point becomes the report's certain failure. A crash while grading is reported as the dry run's own error, diff --git a/changelog.d/always-on-telemetry-emitter.added.md b/changelog.d/always-on-telemetry-emitter.added.md new file mode 100644 index 000000000..3081aaefc --- /dev/null +++ b/changelog.d/always-on-telemetry-emitter.added.md @@ -0,0 +1 @@ +Microcosm builds now start a local telemetry emitter service that reports authenticated live progress, process-tree CPU and memory, heartbeats, US engine-batch progress, and UK graph-node progress without making collector network requests in the build process. PolicyEngine members use their existing Hugging Face login automatically, while builds without an accepted organization credential remain local-only. The staging JSON writers are now named as run-bundle writers and operate independently from the hosted event emitter. SQLAlchemy ORM sessions now manage the local event spool, and Alembic applies and records its schema migrations. diff --git a/packages/microcosm-build/pyproject.toml b/packages/microcosm-build/pyproject.toml index 51d6b154e..68ac4ce97 100644 --- a/packages/microcosm-build/pyproject.toml +++ b/packages/microcosm-build/pyproject.toml @@ -7,6 +7,7 @@ requires-python = ">=3.13" license = { text = "MIT" } authors = [{ name = "PolicyEngine" }] dependencies = [ + "alembic>=1.13.3,<2", "numpy>=1.26", "pandas>=2", "scipy>=1.13", @@ -24,6 +25,7 @@ dependencies = [ "pyyaml>=6", "jsonschema>=4.23,<5", "referencing>=0.35,<1", + "sqlalchemy>=2,<3", ] [project.optional-dependencies] diff --git a/packages/microcosm-build/src/microcosm/build/__init__.py b/packages/microcosm-build/src/microcosm/build/__init__.py index 1a883d3db..b3cb80eaa 100644 --- a/packages/microcosm-build/src/microcosm/build/__init__.py +++ b/packages/microcosm-build/src/microcosm/build/__init__.py @@ -190,7 +190,7 @@ def _assert_frame_compatible(version: str, required: tuple[int, int]) -> None: LATEST_STAGING_POINTER, RUNS_INDEX, STAGING_SCHEMA_VERSION, - StagingTelemetry, + StagingRunBundleWriter, ) __version__ = "0.1.0" @@ -233,7 +233,7 @@ def _assert_frame_compatible(version: str, required: tuple[int, int]) -> None: "LATEST_STAGING_POINTER", "RUNS_INDEX", "STAGING_SCHEMA_VERSION", - "StagingTelemetry", + "StagingRunBundleWriter", "TargetCoverageRequirement", "TargetFitRequirement", "ACCEPTED_CONSUMER_ARTIFACT_SCHEMA_VERSIONS", diff --git a/packages/microcosm-build/src/microcosm/build/staging.py b/packages/microcosm-build/src/microcosm/build/staging.py index d7ebd59a3..58146c159 100644 --- a/packages/microcosm-build/src/microcosm/build/staging.py +++ b/packages/microcosm-build/src/microcosm/build/staging.py @@ -58,8 +58,8 @@ def _write_json(path: Path, payload: dict[str, Any]) -> None: @dataclass -class StagingTelemetry: - """Write and optionally upload build-run telemetry. +class StagingRunBundleWriter: + """Write and optionally upload a build's staging run bundle. Args: run_id: Stable id for this build attempt. diff --git a/packages/microcosm-build/src/microcosm/build/staging_cli.py b/packages/microcosm-build/src/microcosm/build/staging_cli.py index 633d0160b..f4e6983a1 100644 --- a/packages/microcosm-build/src/microcosm/build/staging_cli.py +++ b/packages/microcosm-build/src/microcosm/build/staging_cli.py @@ -1,4 +1,4 @@ -"""Country-neutral command-line options for staging telemetry.""" +"""Country-neutral command-line options for staging run bundles.""" from __future__ import annotations @@ -27,9 +27,9 @@ def add_staging_arguments( ``default_upload_interval_seconds`` lets a long-running command choose a slower best-effort upload cadence: the Hub allows about 128 commits per - hour per repository and every telemetry cycle is up to eight single-file - commits, so a multi-hour solve at the 30-second default exhausts the - budget and loses uploads (the UK rowwise driver runs at 300). + hour per repository and each run-bundle upload cycle performs up to eight + single-file commits, so a multi-hour solve at the 30-second default + exhausts the budget and loses uploads (the UK rowwise driver runs at 300). """ parser.add_argument( @@ -42,7 +42,7 @@ def add_staging_arguments( default=repository.repo_id(os.environ), help=( "Access-controlled Hugging Face dataset repository for best-effort " - "telemetry delivery." + "staging run-bundle delivery." ), ) parser.add_argument( @@ -68,7 +68,10 @@ def add_staging_arguments( mode.add_argument( "--no-staging", action="store_true", - help="Deliberately disable staging telemetry for this build.", + help=( + "Disable the staging run bundle for this build; hosted telemetry " + "remains active." + ), ) parser.add_argument( "--staging-read-back", @@ -107,8 +110,8 @@ def add_staged_dataset_arguments( The finished bundle follows the staging mode switch: ``--no-staging`` keeps nothing, ``--staging-local-only`` keeps the bundle and its sidecars on disk, and the default uploads it to ``repository`` under - ``staged//``. ``--no-staged-dataset`` runs telemetry alone: the - bundle is neither inventoried nor uploaded. + ``staged//``. ``--no-staged-dataset`` retains only the staging run + files: the dataset bundle is neither inventoried nor uploaded. """ parser.add_argument( @@ -125,7 +128,7 @@ def add_staged_dataset_arguments( "--no-staged-dataset", action="store_true", help=( - "Run staging telemetry alone: the finished bundle is neither " + "Keep only the staging run files: the finished bundle is neither " "inventoried nor uploaded (--no-staging already disables both)." ), ) diff --git a/packages/microcosm-build/src/microcosm/build/staging_v2.py b/packages/microcosm-build/src/microcosm/build/staging_v2.py index 1a9ec3b34..105c668b6 100644 --- a/packages/microcosm-build/src/microcosm/build/staging_v2.py +++ b/packages/microcosm-build/src/microcosm/build/staging_v2.py @@ -705,8 +705,8 @@ def _reject_record_collections(self, value: Any) -> None: self._reject_record_collections(item) -class StagingTelemetryV2: - """Record, validate, persist, and optionally upload telemetry version 2.""" +class StagingRunBundleWriterV2: + """Record, validate, persist, and optionally upload a version 2 run bundle.""" def __init__( self, @@ -743,7 +743,8 @@ def __init__( raise StagingContractError("pipeline_version must be non-empty.") if delivery_mode == "disabled": raise StagingContractError( - "Do not construct telemetry for disabled staging; record an opt-out." + "Do not construct a staging run bundle when staging is disabled; " + "record an opt-out." ) if delivery_mode == "local_and_remote" and not (repo_id or "").strip(): raise StagingContractError( diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py new file mode 100644 index 000000000..b7d97a6bd --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py @@ -0,0 +1,460 @@ +"""Best-effort client for the local telemetry emitter service. + +The build process only writes small messages to a private local socket. A +separate process owns retry, authentication, resource sampling, and network +I/O so collector availability cannot delay or fail a dataset build. +""" + +from __future__ import annotations + +import json +import os +import socket +import subprocess +import sys +import tempfile +import time +import uuid +from collections.abc import Mapping +from dataclasses import dataclass, field +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + +from microcosm.build.telemetry_emitter_constants import ( + DEFAULT_HEARTBEAT_SECONDS, + DEFAULT_SEND_TIMEOUT_SECONDS, + DEFAULT_STARTUP_TIMEOUT_SECONDS, + INVALID_ACKNOWLEDGEMENT_ERROR, + NO_LOCAL_SOCKET_WARNING, + QUEUE_WARNING, + RUNTIME_DIRECTORY_PREFIX, + SERVICE_NOT_READY_WARNING, + SERVICE_PROCESS_EXIT_TIMEOUT_SECONDS, + SERVICE_READY_TIMEOUT_SECONDS, + SERVICE_START_WARNING, + SOCKET_FILENAME, + STARTUP_POLL_SECONDS, + TELEMETRY_CACHE_PARTS, + TELEMETRY_SERVICE_MODULE, + TELEMETRY_SPOOL_FILENAME, + TEMPORARY_DIRECTORY_ALIAS, +) +from microcosm.build.telemetry_identity import runtime_identity +from microcosm.build.telemetry_protocol import ( + ACTION_CLOSE, + ACTION_EVENT, + BUILD_COMPLETED_MESSAGE, + BUILD_STARTED_MESSAGE, + CALIBRATION_EVENT_KIND, + EVENT_TYPE_CALIBRATION, + EVENT_TYPE_PROGRESS, + EVENT_TYPE_RUN, + EVENT_TYPE_STAGE, + LOCAL_ACKNOWLEDGEMENT_OK, + LOCAL_ACKNOWLEDGEMENT_READ_BYTES, + LOCAL_MESSAGE_DELIMITER, + LOCAL_PING_MESSAGE, + MAX_TELEMETRY_MESSAGE_CHARS, + SEQUENTIAL_STATUS_MAP, + STAGE_CALIBRATING, + STAGE_COMPLETE, + STAGE_CREATED, + STAGE_FAILED, + STATUS_COMPLETED, + STATUS_FAILED, + STATUS_PROGRESS, + STATUS_STARTED, + TelemetryEventType, + TelemetryStatus, +) +from microcosm.build.telemetry_sanitization import ( + sanitize_details, + sanitize_text, +) + + +def _now() -> str: + return datetime.now(UTC).isoformat() + + +def _cache_dir() -> Path: + configured = os.environ.get("XDG_CACHE_HOME", "").strip() + root = Path(configured).expanduser() if configured else Path.home() / ".cache" + return root.joinpath(*TELEMETRY_CACHE_PARTS) + + +@dataclass(frozen=True) +class TelemetryRun: + """Identity registered with the hosted collector for one build.""" + + run_id: str + country_code: str + pipeline: str + candidate_id: str | None = None + release_id: str | None = None + run_kind: str = "build" + producer_id: str = field(default_factory=lambda: uuid.uuid4().hex) + + def as_registration(self) -> dict[str, Any]: + return { + "run_id": self.run_id, + "producer_id": self.producer_id, + "country_code": self.country_code, + "pipeline": self.pipeline, + "candidate_id": self.candidate_id, + "release_id": self.release_id, + "run_kind": self.run_kind, + } + + +class LocalTelemetryEmitter: + """Non-blocking handle owned by the instrumented build process.""" + + def __init__( + self, + *, + run: TelemetryRun, + process: subprocess.Popen[bytes] | None, + socket_path: Path | None, + runtime_dir: Path | None, + send_timeout_seconds: float = DEFAULT_SEND_TIMEOUT_SECONDS, + ) -> None: + self.run = run + self._process = process + self._socket_path = socket_path + self._runtime_dir = runtime_dir + self._send_timeout_seconds = send_timeout_seconds + self._closed = False + self._warned = False + self._transition_stage: str | None = None + + @classmethod + def start( + cls, + *, + run_id: str, + country_code: str, + pipeline: str, + candidate_id: str | None = None, + release_id: str | None = None, + run_kind: str = "build", + development_collector_url: str | None = None, + heartbeat_seconds: float = DEFAULT_HEARTBEAT_SECONDS, + startup_timeout_seconds: float = DEFAULT_STARTUP_TIMEOUT_SECONDS, + spool_path: Path | str | None = None, + ) -> LocalTelemetryEmitter: + """Start the service, returning a harmless disabled handle on failure.""" + + run = TelemetryRun( + run_id=run_id, + country_code=country_code, + pipeline=pipeline, + candidate_id=candidate_id, + release_id=release_id, + run_kind=run_kind, + ) + if not hasattr(socket, "AF_UNIX"): + print(NO_LOCAL_SOCKET_WARNING, file=sys.stderr) + return cls(run=run, process=None, socket_path=None, runtime_dir=None) + + # macOS limits AF_UNIX paths to roughly 100 bytes. Its default + # temporary directory is already long, so use the short system alias. + temporary_root = ( + TEMPORARY_DIRECTORY_ALIAS if TEMPORARY_DIRECTORY_ALIAS.is_dir() else None + ) + runtime_dir = Path( + tempfile.mkdtemp(prefix=RUNTIME_DIRECTORY_PREFIX, dir=temporary_root) + ) + runtime_dir.chmod(0o700) + socket_path = runtime_dir / SOCKET_FILENAME + queue_path = ( + Path(spool_path) if spool_path else _cache_dir() / TELEMETRY_SPOOL_FILENAME + ) + command = [ + sys.executable, + "-m", + TELEMETRY_SERVICE_MODULE, + "--socket", + str(socket_path), + "--spool", + str(queue_path), + "--registration-json", + json.dumps(run.as_registration(), separators=(",", ":")), + "--parent-pid", + str(os.getpid()), + "--heartbeat-seconds", + str(heartbeat_seconds), + ] + if development_collector_url is not None: + command.extend(["--development-collector-url", development_collector_url]) + process: subprocess.Popen[bytes] | None = None + try: + process = subprocess.Popen( + command, + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + start_new_session=True, + ) + deadline = time.monotonic() + startup_timeout_seconds + while time.monotonic() < deadline: + if socket_path.exists() and _service_ready(socket_path): + emitter = cls( + run=run, + process=process, + socket_path=socket_path, + runtime_dir=runtime_dir, + ) + emitter.emit( + event_type=EVENT_TYPE_RUN, + stage_id=STAGE_CREATED, + status=STATUS_STARTED, + message=BUILD_STARTED_MESSAGE, + details={"identity": runtime_identity()}, + ) + return emitter + if process.poll() is not None: + break + time.sleep(STARTUP_POLL_SECONDS) + except Exception as error: + print( + SERVICE_START_WARNING.format( + error_type=type(error).__name__, + error=error, + ), + file=sys.stderr, + ) + else: + print(SERVICE_NOT_READY_WARNING, file=sys.stderr) + if process is not None: + _terminate_process(process) + try: + socket_path.unlink(missing_ok=True) + runtime_dir.rmdir() + except OSError: + pass + return cls(run=run, process=None, socket_path=None, runtime_dir=runtime_dir) + + @property + def available(self) -> bool: + return self._socket_path is not None and not self._closed + + def emit( + self, + *, + event_type: TelemetryEventType, + status: TelemetryStatus, + stage_id: str | None = None, + message: str | None = None, + details: Mapping[str, Any] | None = None, + ) -> None: + """Queue one event locally, never raising into the build.""" + + if not self.available: + return + self._send( + { + "action": ACTION_EVENT, + "event": { + "timestamp": _now(), + "event_type": event_type, + "stage_id": stage_id, + "status": status, + "message": ( + sanitize_text(message, limit=MAX_TELEMETRY_MESSAGE_CHARS) + if message + else None + ), + "details": sanitize_details(details or {}), + }, + } + ) + + def stage( + self, + stage_id: str, + *, + status: TelemetryStatus = STATUS_STARTED, + message: str | None = None, + **details: Any, + ) -> None: + self.emit( + event_type=EVENT_TYPE_STAGE, + stage_id=stage_id, + status=status, + message=message, + details=details, + ) + + def transition_stage( + self, + stage_id: str, + *, + status: str = "running", + message: str | None = None, + **details: Any, + ) -> None: + """Translate a sequential stage update into explicit lifecycle events.""" + + collector_status = SEQUENTIAL_STATUS_MAP.get(status, STATUS_PROGRESS) + if collector_status == STATUS_STARTED: + if self._transition_stage == stage_id: + self.stage( + stage_id, + status=STATUS_PROGRESS, + message=message, + **details, + ) + return + self._close_transition_stage() + self._transition_stage = stage_id + elif self._transition_stage == stage_id: + self._transition_stage = None + else: + self._close_transition_stage() + self.stage( + stage_id, + status=collector_status, + message=message, + **details, + ) + + def progress( + self, + stage_id: str, + *, + done: int, + total: int, + unit: str | None = None, + **details: Any, + ) -> None: + self.emit( + event_type=EVENT_TYPE_PROGRESS, + stage_id=stage_id, + status=STATUS_PROGRESS, + details={"done": done, "total": total, "unit": unit, **details}, + ) + + def calibration_progress(self, event: Mapping[str, Any]) -> None: + if event.get("kind") != CALIBRATION_EVENT_KIND: + return + self.emit( + event_type=EVENT_TYPE_CALIBRATION, + stage_id=STAGE_CALIBRATING, + status=STATUS_PROGRESS, + details=event, + ) + + def transition_calibration_progress(self, event: Mapping[str, Any]) -> None: + """Enter the sequential calibration stage, then report one epoch.""" + + if event.get("kind") != CALIBRATION_EVENT_KIND: + return + if self._transition_stage != STAGE_CALIBRATING: + self.transition_stage(STAGE_CALIBRATING) + self.calibration_progress(event) + + def fail( + self, + error: BaseException, + *, + failed_during: str | None = None, + failure_class: str = "build_failure", + ) -> None: + failed_stage = failed_during or self._transition_stage + self._close_transition_stage() + self.emit( + event_type=EVENT_TYPE_RUN, + stage_id=STAGE_FAILED, + status=STATUS_FAILED, + message=str(error)[:MAX_TELEMETRY_MESSAGE_CHARS], + details={ + "error_type": type(error).__name__, + "failure_class": failure_class, + "failed_during": failed_stage, + }, + ) + self.close() + + def complete(self) -> None: + self._close_transition_stage() + self.emit( + event_type=EVENT_TYPE_RUN, + stage_id=STAGE_COMPLETE, + status=STATUS_COMPLETED, + message=BUILD_COMPLETED_MESSAGE, + ) + self.close() + + def close(self) -> None: + """Ask the service to flush in the background, then release the handle.""" + + if self._closed: + return + self._send({"action": ACTION_CLOSE}) + self._closed = True + + def _close_transition_stage(self) -> None: + if self._transition_stage is None: + return + stage_id = self._transition_stage + self._transition_stage = None + self.stage(stage_id, status=STATUS_COMPLETED) + + def _send(self, payload: Mapping[str, Any]) -> None: + if self._socket_path is None: + return + try: + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as client: + client.settimeout(self._send_timeout_seconds) + client.connect(str(self._socket_path)) + client.sendall( + json.dumps(payload, separators=(",", ":"), allow_nan=False).encode() + + LOCAL_MESSAGE_DELIMITER + ) + acknowledgement = client.recv(LOCAL_ACKNOWLEDGEMENT_READ_BYTES) + if acknowledgement != LOCAL_ACKNOWLEDGEMENT_OK: + raise OSError(INVALID_ACKNOWLEDGEMENT_ERROR) + except Exception as error: + if not self._warned: + print( + QUEUE_WARNING.format(error_type=type(error).__name__), + file=sys.stderr, + ) + self._warned = True + + +def start_local_telemetry_emitter_service( + **run: Any, +) -> LocalTelemetryEmitter: + """Named construction seam for build entrypoints and tests.""" + + return LocalTelemetryEmitter.start(**run) + + +def _service_ready(socket_path: Path) -> bool: + try: + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as client: + client.settimeout(SERVICE_READY_TIMEOUT_SECONDS) + client.connect(str(socket_path)) + client.sendall(LOCAL_PING_MESSAGE) + return ( + client.recv(LOCAL_ACKNOWLEDGEMENT_READ_BYTES) + == LOCAL_ACKNOWLEDGEMENT_OK + ) + except OSError: + return False + + +def _terminate_process(process: subprocess.Popen[bytes]) -> None: + """Terminate and reap an emitter service that did not become usable.""" + + try: + if process.poll() is None: + process.terminate() + process.wait(timeout=SERVICE_PROCESS_EXIT_TIMEOUT_SECONDS) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=SERVICE_PROCESS_EXIT_TIMEOUT_SECONDS) + except OSError: + pass diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_constants.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_constants.py new file mode 100644 index 000000000..57a477f82 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_constants.py @@ -0,0 +1,38 @@ +"""Configuration and user-facing messages for the telemetry emitter client.""" + +from __future__ import annotations + +from pathlib import Path +from typing import Final + +TELEMETRY_SERVICE_MODULE: Final = "microcosm.build.telemetry_emitter_service" +TELEMETRY_CACHE_PARTS: Final = ("microcosm", "telemetry") +TELEMETRY_SPOOL_FILENAME: Final = "events.sqlite3" +RUNTIME_DIRECTORY_PREFIX: Final = "microcosm-telemetry-" +SOCKET_FILENAME: Final = "emitter.sock" +TEMPORARY_DIRECTORY_ALIAS: Final = Path("/tmp") + +DEFAULT_SEND_TIMEOUT_SECONDS: Final = 0.2 +DEFAULT_HEARTBEAT_SECONDS: Final = 60.0 +DEFAULT_STARTUP_TIMEOUT_SECONDS: Final = 3.0 +STARTUP_POLL_SECONDS: Final = 0.02 +SERVICE_READY_TIMEOUT_SECONDS: Final = 0.1 +SERVICE_PROCESS_EXIT_TIMEOUT_SECONDS: Final = 1.0 + +NO_LOCAL_SOCKET_WARNING: Final = ( + "warning: Microcosm telemetry is unavailable because this platform has no " + "local Unix sockets." +) +SERVICE_START_WARNING: Final = ( + "warning: the local telemetry emitter service could not start: " + "{error_type}: {error}" +) +SERVICE_NOT_READY_WARNING: Final = ( + "warning: the local telemetry emitter service did not become ready; " + "the build will continue without hosted telemetry." +) +QUEUE_WARNING: Final = ( + "warning: the local telemetry emitter service could not queue an update " + "({error_type}); the build will continue." +) +INVALID_ACKNOWLEDGEMENT_ERROR: Final = "invalid acknowledgement" diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__init__.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__init__.py new file mode 100644 index 000000000..ef5335561 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__init__.py @@ -0,0 +1,17 @@ +"""Local service that queues and delivers Microcosm telemetry.""" + +from microcosm.build.telemetry_emitter_service.collector import CollectorDelivery +from microcosm.build.telemetry_emitter_service.constants import ( + PRODUCTION_COLLECTOR_URL, +) +from microcosm.build.telemetry_emitter_service.resources import ProcessTreeSampler +from microcosm.build.telemetry_emitter_service.runtime import EmitterService +from microcosm.build.telemetry_emitter_service.spool import EventSpool + +__all__ = [ + "CollectorDelivery", + "EmitterService", + "EventSpool", + "PRODUCTION_COLLECTOR_URL", + "ProcessTreeSampler", +] diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__main__.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__main__.py new file mode 100644 index 000000000..298103831 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__main__.py @@ -0,0 +1,5 @@ +"""Module execution entry point for the telemetry emitter service.""" + +from microcosm.build.telemetry_emitter_service.main import main + +raise SystemExit(main()) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/env.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/env.py new file mode 100644 index 000000000..2cb2edcc2 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/env.py @@ -0,0 +1,28 @@ +"""Alembic environment for the local telemetry spool.""" + +from alembic import context +from sqlalchemy.engine import Connection + +from microcosm.build.telemetry_emitter_service.models import SpoolModel + +_MISSING_CONNECTION_ERROR = ( + "telemetry spool migrations require a programmatically supplied connection" +) + + +def run_migrations() -> None: + """Run migrations on the connection supplied by the emitter service.""" + + connection = context.config.attributes.get("connection") + if not isinstance(connection, Connection): + raise RuntimeError(_MISSING_CONNECTION_ERROR) + context.configure( + connection=connection, + target_metadata=SpoolModel.metadata, + render_as_batch=True, + ) + with context.begin_transaction(): + context.run_migrations() + + +run_migrations() diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/script.py.mako b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/script.py.mako new file mode 100644 index 000000000..f059f7642 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/script.py.mako @@ -0,0 +1,31 @@ +"""${message} + +Revision ID: ${up_revision} +Revises: ${down_revision | comma,n} +Create Date: ${create_date} +""" + +from __future__ import annotations + +from collections.abc import Sequence + +import sqlalchemy as sa +from alembic import op +${imports if imports else ""} + +revision: str = ${repr(up_revision)} +down_revision: str | None = ${repr(down_revision)} +branch_labels: str | Sequence[str] | None = ${repr(branch_labels)} +depends_on: str | Sequence[str] | None = ${repr(depends_on)} + + +def upgrade() -> None: + """Apply this schema revision.""" + + ${upgrades if upgrades else "pass"} + + +def downgrade() -> None: + """Reverse this schema revision.""" + + ${downgrades if downgrades else "pass"} diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/versions/20261007_01_initial_spool.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/versions/20261007_01_initial_spool.py new file mode 100644 index 000000000..87290fd65 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/versions/20261007_01_initial_spool.py @@ -0,0 +1,140 @@ +"""Create or adopt the telemetry spool schema. + +Revision ID: 20261007_01 +Revises: +""" + +from __future__ import annotations + +from collections.abc import Sequence + +import sqlalchemy as sa +from alembic import op + +revision: str = "20261007_01" +down_revision: str | None = None +branch_labels: str | Sequence[str] | None = None +depends_on: str | Sequence[str] | None = None + +_RUNS_TABLE = "telemetry_runs" +_EVENTS_TABLE = "telemetry_events" +_EVENTS_INDEX = "telemetry_events_run_sequence" +_PENDING_UPLOAD_STATE = "pending" +_LOCAL_ONLY_UPLOAD_STATE = "local_only" +_PRE_ELIGIBILITY_REASON = "created_before_upload_eligibility" + + +def _create_runs_table() -> None: + op.create_table( + _RUNS_TABLE, + sa.Column("run_id", sa.Text(), nullable=False), + sa.Column("producer_id", sa.Text(), nullable=False), + sa.Column("registration_json", sa.Text(), nullable=False), + sa.Column( + "next_sequence", + sa.Integer(), + nullable=False, + server_default="1", + ), + sa.Column( + "upload_state", + sa.Text(), + nullable=False, + server_default=_PENDING_UPLOAD_STATE, + ), + sa.Column("local_only_reason", sa.Text(), nullable=True), + sa.Column("updated_at", sa.Text(), nullable=False), + sa.PrimaryKeyConstraint("run_id", "producer_id"), + ) + + +def _adopt_runs_table(inspector: sa.Inspector) -> None: + column_names = {column["name"] for column in inspector.get_columns(_RUNS_TABLE)} + added_upload_state = "upload_state" not in column_names + if added_upload_state: + op.add_column( + _RUNS_TABLE, + sa.Column( + "upload_state", + sa.Text(), + nullable=False, + server_default=_PENDING_UPLOAD_STATE, + ), + ) + added_local_only_reason = "local_only_reason" not in column_names + if added_local_only_reason: + op.add_column( + _RUNS_TABLE, + sa.Column("local_only_reason", sa.Text(), nullable=True), + ) + + runs = sa.table( + _RUNS_TABLE, + sa.column("upload_state", sa.Text()), + sa.column("local_only_reason", sa.Text()), + ) + if added_upload_state: + op.execute(sa.update(runs).values(upload_state=_LOCAL_ONLY_UPLOAD_STATE)) + if added_upload_state or added_local_only_reason: + op.execute( + sa.update(runs) + .where(runs.c.upload_state == _LOCAL_ONLY_UPLOAD_STATE) + .values(local_only_reason=_PRE_ELIGIBILITY_REASON) + ) + + +def _create_events_table() -> None: + op.create_table( + _EVENTS_TABLE, + sa.Column("event_id", sa.Text(), nullable=False), + sa.Column("run_id", sa.Text(), nullable=False), + sa.Column("producer_id", sa.Text(), nullable=False), + sa.Column("sequence", sa.Integer(), nullable=False), + sa.Column("payload_json", sa.Text(), nullable=False), + sa.Column("created_at", sa.Text(), nullable=False), + sa.ForeignKeyConstraint( + ["run_id", "producer_id"], + [f"{_RUNS_TABLE}.run_id", f"{_RUNS_TABLE}.producer_id"], + name="fk_telemetry_events_run", + ), + sa.PrimaryKeyConstraint("event_id"), + sa.UniqueConstraint( + "run_id", + "producer_id", + "sequence", + name="uq_telemetry_events_run_producer_sequence", + ), + ) + + +def upgrade() -> None: + """Create the current schema or adopt a pre-Alembic spool.""" + + inspector = sa.inspect(op.get_bind()) + table_names = set(inspector.get_table_names()) + if _RUNS_TABLE not in table_names: + _create_runs_table() + else: + _adopt_runs_table(inspector) + + inspector = sa.inspect(op.get_bind()) + table_names = set(inspector.get_table_names()) + if _EVENTS_TABLE not in table_names: + _create_events_table() + + inspector = sa.inspect(op.get_bind()) + index_names = {index["name"] for index in inspector.get_indexes(_EVENTS_TABLE)} + if _EVENTS_INDEX not in index_names: + op.create_index( + _EVENTS_INDEX, + _EVENTS_TABLE, + ["run_id", "producer_id", "sequence"], + ) + + +def downgrade() -> None: + """Remove the local spool schema.""" + + op.drop_index(_EVENTS_INDEX, table_name=_EVENTS_TABLE) + op.drop_table(_EVENTS_TABLE) + op.drop_table(_RUNS_TABLE) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/collector.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/collector.py new file mode 100644 index 000000000..795c61a17 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/collector.py @@ -0,0 +1,273 @@ +"""Collector authentication and best-effort event delivery.""" + +from __future__ import annotations + +import json +import os +import sys +import time +import urllib.error +import urllib.request +from collections.abc import Mapping +from http import HTTPStatus +from typing import Any +from urllib.parse import urlsplit + +from huggingface_hub import get_token + +from microcosm.build.telemetry_emitter_service.constants import ( + COLLECTOR_RESPONSE_TOO_LARGE_ERROR, + COLLECTOR_URL_HTTPS_ERROR, + COLLECTOR_URL_ORIGIN_ERROR, + DEFAULT_TOKEN_LIFETIME_SECONDS, + DEVELOPMENT_COLLECTOR_LOOPBACK_ERROR, + HTTP_TIMEOUT_SECONDS, + HTTP_USER_AGENT, + INITIAL_RETRY_SECONDS, + LOCAL_ONLY_MISSING_CREDENTIAL, + LOCAL_ONLY_REJECTED_COLLECTOR_AUTHORIZATION, + LOCAL_ONLY_REJECTED_CREDENTIAL, + LOCAL_ONLY_REJECTED_REGISTRATION, + LOOPBACK_HOSTS, + MAX_HTTP_RESPONSE_BYTES, + MAX_RETRY_SECONDS, + MINIMUM_TOKEN_LIFETIME_SECONDS, + NO_CREDENTIAL_MESSAGE, + PRODUCTION_COLLECTOR_URL, + REJECTED_CREDENTIAL_MESSAGE, + RUN_EVENTS_PATH_TEMPLATE, + RUN_REGISTRATION_PATH, + TOKEN_EXCHANGE_PATH, + TOKEN_REFRESH_MARGIN_SECONDS, +) +from microcosm.build.telemetry_emitter_service.spool import EventSpool + + +class _NoRedirectHandler(urllib.request.HTTPRedirectHandler): + """Keep bearer credentials on the explicitly configured origin.""" + + def redirect_request(self, req, fp, code, msg, headers, newurl): + return None + + +def _collector_origin(value: str, *, allow_loopback_http: bool = False) -> str: + parsed = urlsplit(value) + loopback = parsed.hostname in LOOPBACK_HOSTS + valid_scheme = parsed.scheme == "https" or ( + allow_loopback_http and loopback and parsed.scheme == "http" + ) + if not parsed.hostname or not valid_scheme: + raise ValueError(COLLECTOR_URL_HTTPS_ERROR) + if ( + parsed.username + or parsed.password + or parsed.query + or parsed.fragment + or parsed.path not in {"", "/"} + ): + raise ValueError(COLLECTOR_URL_ORIGIN_ERROR) + return value.rstrip("/") + + +def _development_collector_url(value: str) -> str: + parsed = urlsplit(value) + if parsed.hostname not in LOOPBACK_HOSTS: + raise ValueError(DEVELOPMENT_COLLECTOR_LOOPBACK_ERROR) + return _collector_origin(value, allow_loopback_http=True) + + +def _decode_response(body: bytes) -> dict[str, Any]: + try: + response = json.loads(body) if body else {} + except (json.JSONDecodeError, UnicodeDecodeError): + return {} + return response if isinstance(response, dict) else {} + + +def _http_post( + url: str, + payload: Mapping[str, Any], + bearer_token: str, + *, + timeout: float = HTTP_TIMEOUT_SECONDS, +) -> tuple[int, dict[str, Any]]: + request = urllib.request.Request( + url, + data=json.dumps(payload, separators=(",", ":")).encode(), + headers={ + "Authorization": f"Bearer {bearer_token}", + "Content-Type": "application/json", + "User-Agent": HTTP_USER_AGENT, + }, + method="POST", + ) + try: + opener = urllib.request.build_opener(_NoRedirectHandler) + with opener.open(request, timeout=timeout) as response: + body = response.read(MAX_HTTP_RESPONSE_BYTES + 1) + if len(body) > MAX_HTTP_RESPONSE_BYTES: + raise OSError(COLLECTOR_RESPONSE_TOO_LARGE_ERROR) + return response.status, _decode_response(body) + except urllib.error.HTTPError as error: + body = error.read(MAX_HTTP_RESPONSE_BYTES + 1) + if len(body) > MAX_HTTP_RESPONSE_BYTES: + return error.code, {} + return error.code, _decode_response(body) + + +def _huggingface_token() -> str | None: + return ( + os.environ.get("HF_TOKEN", "").strip() + or os.environ.get("HUGGINGFACE_TOKEN", "").strip() + or get_token() + ) + + +def _registration_key(registration: Mapping[str, Any]) -> str: + return f"{registration['run_id']}:{registration['producer_id']}" + + +class CollectorDelivery: + """Authenticate queued runs and deliver idempotent event batches.""" + + def __init__( + self, + spool: EventSpool, + *, + development_collector_url: str | None = None, + ) -> None: + self.collector_url = ( + _development_collector_url(development_collector_url) + if development_collector_url is not None + else _collector_origin(PRODUCTION_COLLECTOR_URL) + ) + self.spool = spool + self._session_token: tuple[str, float] | None = None + self._registered: set[str] = set() + self._warned_no_token = False + self._warned_denied: set[str] = set() + self._next_attempt_at = 0.0 + self._retry_seconds = INITIAL_RETRY_SECONDS + + def flush_once(self) -> bool: + """Attempt one delivery pass without waiting for retry deadlines.""" + + if time.monotonic() < self._next_attempt_at: + return False + made_progress = False + for registration in self.spool.pending_runs(): + run_id = str(registration["run_id"]) + registration_key = _registration_key(registration) + token = self._collector_token(registration) + if token is None: + continue + if registration_key not in self._registered: + if not self._register(registration, token): + continue + events = self.spool.batch(run_id, str(registration["producer_id"])) + if not events: + continue + try: + status, _ = _http_post( + self.collector_url + RUN_EVENTS_PATH_TEMPLATE.format(run_id=run_id), + {"events": events}, + token, + ) + except (OSError, TimeoutError): + self._defer_retry() + continue + if status == HTTPStatus.ACCEPTED: + self.spool.acknowledge([event["event_id"] for event in events]) + made_progress = True + self._retry_seconds = INITIAL_RETRY_SECONDS + elif status == HTTPStatus.UNAUTHORIZED: + self._session_token = None + elif status == HTTPStatus.FORBIDDEN: + self._make_local_only( + registration, + LOCAL_ONLY_REJECTED_COLLECTOR_AUTHORIZATION, + ) + else: + self._defer_retry() + return made_progress + + def _collector_token(self, registration: Mapping[str, Any]) -> str | None: + if ( + self._session_token is not None + and self._session_token[1] > time.monotonic() + TOKEN_REFRESH_MARGIN_SECONDS + ): + return self._session_token[0] + hf_token = _huggingface_token() + if not hf_token: + if not self._warned_no_token: + print(NO_CREDENTIAL_MESSAGE, file=sys.stderr, flush=True) + self._warned_no_token = True + self._make_local_only(registration, LOCAL_ONLY_MISSING_CREDENTIAL) + return None + try: + status, response = _http_post( + self.collector_url + TOKEN_EXCHANGE_PATH, + {}, + hf_token, + ) + except (OSError, TimeoutError): + self._defer_retry() + return None + if status == HTTPStatus.OK and isinstance(response.get("access_token"), str): + expires_in = max( + MINIMUM_TOKEN_LIFETIME_SECONDS, + int(response.get("expires_in", DEFAULT_TOKEN_LIFETIME_SECONDS)), + ) + token = response["access_token"] + self._session_token = (token, time.monotonic() + expires_in) + return token + if status in {HTTPStatus.UNAUTHORIZED, HTTPStatus.FORBIDDEN}: + self._make_local_only(registration, LOCAL_ONLY_REJECTED_CREDENTIAL) + else: + self._defer_retry() + return None + + def _register(self, registration: Mapping[str, Any], token: str) -> bool: + registration_key = _registration_key(registration) + try: + status, _ = _http_post( + self.collector_url + RUN_REGISTRATION_PATH, + registration, + token, + ) + except (OSError, TimeoutError): + self._defer_retry() + return False + if status == HTTPStatus.CREATED: + self._registered.add(registration_key) + self._retry_seconds = INITIAL_RETRY_SECONDS + return True + if status == HTTPStatus.UNAUTHORIZED: + self._session_token = None + elif status in {HTTPStatus.FORBIDDEN, HTTPStatus.CONFLICT}: + self._make_local_only(registration, LOCAL_ONLY_REJECTED_REGISTRATION) + else: + self._defer_retry() + return False + + def _make_local_only( + self, + registration: Mapping[str, Any], + reason: str, + ) -> None: + run_id = str(registration["run_id"]) + producer_id = str(registration["producer_id"]) + registration_key = _registration_key(registration) + self.spool.make_local_only(run_id, producer_id, reason) + if reason != LOCAL_ONLY_MISSING_CREDENTIAL: + self._warn_denied(registration_key) + + def _defer_retry(self) -> None: + self._next_attempt_at = time.monotonic() + self._retry_seconds + self._retry_seconds = min(MAX_RETRY_SECONDS, self._retry_seconds * 2) + + def _warn_denied(self, registration_key: str) -> None: + if registration_key in self._warned_denied: + return + print(REJECTED_CREDENTIAL_MESSAGE, file=sys.stderr, flush=True) + self._warned_denied.add(registration_key) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/constants.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/constants.py new file mode 100644 index 000000000..e4a55b7d7 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/constants.py @@ -0,0 +1,69 @@ +"""Configuration and protocol values for the telemetry emitter service.""" + +from __future__ import annotations + +from typing import Final + +RETENTION_DAYS: Final = 7 +MAX_QUEUED_BYTES: Final = 100 * 1024 * 1024 +BATCH_SIZE: Final = 100 +PRODUCTION_COLLECTOR_URL: Final = ( + "https://microcosm-telemetry-389282473430.us-central1.run.app" +) + +LOOPBACK_HOSTS: Final = frozenset({"localhost", "127.0.0.1", "::1"}) +TOKEN_EXCHANGE_PATH: Final = "/v1/auth/huggingface/exchange" +RUN_REGISTRATION_PATH: Final = "/v1/runs" +RUN_EVENTS_PATH_TEMPLATE: Final = "/v1/runs/{run_id}/events" +HTTP_USER_AGENT: Final = "microcosm-telemetry-emitter/1" +MAX_HTTP_RESPONSE_BYTES: Final = 1_048_576 +HTTP_TIMEOUT_SECONDS: Final = 5.0 + +UPLOAD_STATE_PENDING: Final = "pending" +UPLOAD_STATE_LOCAL_ONLY: Final = "local_only" +LOCAL_ONLY_MISSING_CREDENTIAL: Final = "missing_huggingface_credential" +LOCAL_ONLY_REJECTED_CREDENTIAL: Final = "huggingface_credential_rejected" +LOCAL_ONLY_REJECTED_REGISTRATION: Final = "run_registration_rejected" +LOCAL_ONLY_REJECTED_COLLECTOR_AUTHORIZATION: Final = "collector_authorization_rejected" +LOCAL_ONLY_PRE_ELIGIBILITY: Final = "created_before_upload_eligibility" + +NO_CREDENTIAL_MESSAGE: Final = ( + "Microcosm telemetry is local-only: no ambient Hugging Face credential was " + "found. The dataset build will continue." +) +REJECTED_CREDENTIAL_MESSAGE: Final = ( + "Microcosm telemetry is local-only for this run: the ambient Hugging Face " + "credential was not accepted as a PolicyEngine organization member. The " + "dataset build will continue." +) +COLLECTOR_URL_HTTPS_ERROR: Final = "collector URL must be an HTTPS origin" +COLLECTOR_URL_ORIGIN_ERROR: Final = ( + "collector URL must be an origin without credentials or path data" +) +DEVELOPMENT_COLLECTOR_LOOPBACK_ERROR: Final = ( + "development collector URL must use a loopback address" +) +COLLECTOR_RESPONSE_TOO_LARGE_ERROR: Final = "collector response exceeds 1 MiB" + +DEFAULT_TOKEN_LIFETIME_SECONDS: Final = 3_600 +MINIMUM_TOKEN_LIFETIME_SECONDS: Final = 60 +TOKEN_REFRESH_MARGIN_SECONDS: Final = 30 +INITIAL_RETRY_SECONDS: Final = 1.0 +MAX_RETRY_SECONDS: Final = 60.0 + +DATABASE_TIMEOUT_SECONDS: Final = 5 +PRUNE_INTERVAL_SECONDS: Final = 60.0 + +DEFAULT_HEARTBEAT_SECONDS: Final = 60.0 +DEFAULT_DRAIN_SECONDS: Final = 15.0 +MINIMUM_HEARTBEAT_SECONDS: Final = 1.0 +SOCKET_LISTEN_BACKLOG: Final = 16 +SOCKET_ACCEPT_TIMEOUT_SECONDS: Final = 0.5 +SOCKET_CONNECTION_TIMEOUT_SECONDS: Final = 0.25 +WORKER_INTERVAL_SECONDS: Final = 1.0 +DRAIN_RETRY_SECONDS: Final = 0.5 + +EVENT_OBJECT_ERROR: Final = "event must be an object" +UNSUPPORTED_ACTION_ERROR: Final = "unsupported local telemetry action" +LOCAL_MESSAGE_TOO_LARGE_ERROR: Final = "local telemetry message exceeds 1 MiB" +FAILURE_CLASS_UNEXPECTED_PROCESS_EXIT: Final = "unexpected_process_exit" diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/database.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/database.py new file mode 100644 index 000000000..75d32d96b --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/database.py @@ -0,0 +1,34 @@ +"""SQLAlchemy engine and session construction for the telemetry spool.""" + +from __future__ import annotations + +from pathlib import Path + +from sqlalchemy import URL, create_engine +from sqlalchemy.engine import Engine +from sqlalchemy.orm import Session, sessionmaker + +from microcosm.build.telemetry_emitter_service.constants import ( + DATABASE_TIMEOUT_SECONDS, +) + + +def sqlite_database_url(path: Path | str) -> URL: + """Return a SQLAlchemy URL for a local SQLite spool path.""" + + return URL.create("sqlite+pysqlite", database=str(Path(path))) + + +def create_spool_engine(path: Path | str) -> Engine: + """Create the SQLAlchemy engine used by one emitter service process.""" + + return create_engine( + sqlite_database_url(path), + connect_args={"timeout": DATABASE_TIMEOUT_SECONDS}, + ) + + +def create_spool_session_factory(engine: Engine) -> sessionmaker[Session]: + """Create short-lived ORM sessions bound to the spool engine.""" + + return sessionmaker(engine, expire_on_commit=False) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/main.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/main.py new file mode 100644 index 000000000..2872bf91a --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/main.py @@ -0,0 +1,53 @@ +"""Command-line construction for the telemetry emitter service.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +from microcosm.build.telemetry_emitter_service.collector import CollectorDelivery +from microcosm.build.telemetry_emitter_service.constants import ( + DEFAULT_HEARTBEAT_SECONDS, +) +from microcosm.build.telemetry_emitter_service.resources import ProcessTreeSampler +from microcosm.build.telemetry_emitter_service.runtime import EmitterService +from microcosm.build.telemetry_emitter_service.spool import EventSpool + + +def build_parser() -> argparse.ArgumentParser: + """Create the service command-line parser.""" + + parser = argparse.ArgumentParser() + parser.add_argument("--socket", type=Path, required=True) + parser.add_argument("--spool", type=Path, required=True) + parser.add_argument("--development-collector-url") + parser.add_argument("--registration-json", required=True) + parser.add_argument("--parent-pid", type=int, required=True) + parser.add_argument( + "--heartbeat-seconds", + type=float, + default=DEFAULT_HEARTBEAT_SECONDS, + ) + return parser + + +def main(argv: list[str] | None = None) -> int: + """Run the telemetry emitter service until the build disconnects.""" + + args = build_parser().parse_args(argv) + registration = json.loads(args.registration_json) + spool = EventSpool(args.spool) + service = EmitterService( + socket_path=args.socket, + registration=registration, + spool=spool, + delivery=CollectorDelivery( + spool, + development_collector_url=args.development_collector_url, + ), + sampler=ProcessTreeSampler(args.parent_pid), + heartbeat_seconds=args.heartbeat_seconds, + ) + service.run() + return 0 diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/migrations.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/migrations.py new file mode 100644 index 000000000..c1c69c61c --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/migrations.py @@ -0,0 +1,60 @@ +"""Programmatic Alembic migration entry points for the telemetry spool.""" + +from __future__ import annotations + +from collections.abc import Iterator +from contextlib import contextmanager +from importlib.resources import as_file, files +from pathlib import Path + +from alembic import command +from alembic.config import Config +from alembic.migration import MigrationContext +from alembic.script import ScriptDirectory +from sqlalchemy.engine import Connection, Engine + +from microcosm.build.telemetry_emitter_service.database import create_spool_engine + +_MIGRATION_TARGET = "head" + + +@contextmanager +def alembic_config( + *, + connection: Connection | None = None, +) -> Iterator[Config]: + """Yield Alembic configuration with migrations on the filesystem.""" + + migration_resources = files(__package__).joinpath("alembic") + with as_file(migration_resources) as migration_directory: + config = Config() + config.set_main_option("script_location", str(migration_directory)) + if connection is not None: + config.attributes["connection"] = connection + yield config + + +def upgrade_spool_database(engine: Engine) -> None: + """Apply every committed spool migration to an engine.""" + + with engine.begin() as connection: + with alembic_config(connection=connection) as config: + command.upgrade(config, _MIGRATION_TARGET) + + +def current_database_revision(path: Path | str) -> str | None: + """Return the Alembic revision recorded by one spool database.""" + + engine = create_spool_engine(path) + try: + with engine.connect() as connection: + return MigrationContext.configure(connection).get_current_revision() + finally: + engine.dispose() + + +def migration_head_revision() -> str | None: + """Return the single current head from the packaged migration history.""" + + with alembic_config() as config: + return ScriptDirectory.from_config(config).get_current_head() diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/models.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/models.py new file mode 100644 index 000000000..87f1c678d --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/models.py @@ -0,0 +1,130 @@ +"""SQLAlchemy ORM models for the local telemetry spool.""" + +from __future__ import annotations + +import json +from typing import Any + +from sqlalchemy import ( + ForeignKeyConstraint, + Index, + Integer, + Text, + UniqueConstraint, +) +from sqlalchemy.orm import DeclarativeBase, Mapped, mapped_column, relationship +from sqlalchemy.types import TypeDecorator + +from microcosm.build.telemetry_emitter_service.constants import ( + UPLOAD_STATE_PENDING, +) + + +class JsonObjectText(TypeDecorator[dict[str, Any]]): + """Store JSON objects in the spool's existing text-column format.""" + + impl = Text + cache_ok = True + + def process_bind_param( + self, + value: dict[str, Any] | None, + dialect, + ) -> str | None: + if value is None: + return None + return _serialize_json_object(value) + + def process_result_value( + self, + value: str | None, + dialect, + ) -> dict[str, Any] | None: + if value is None: + return None + decoded = json.loads(value) + if not isinstance(decoded, dict): + raise ValueError("telemetry spool JSON must contain an object") + return decoded + + +def serialized_json_length(value: dict[str, Any]) -> int: + """Return the stored character count for one mapped JSON object.""" + + return len(_serialize_json_object(value)) + + +def _serialize_json_object(value: dict[str, Any]) -> str: + return json.dumps( + value, + separators=(",", ":"), + sort_keys=True, + allow_nan=False, + ) + + +class SpoolModel(DeclarativeBase): + """Declarative base for the local telemetry spool.""" + + +class TelemetryRunRecord(SpoolModel): + """One telemetry producer registered for one build run.""" + + __tablename__ = "telemetry_runs" + + run_id: Mapped[str] = mapped_column(Text, primary_key=True) + producer_id: Mapped[str] = mapped_column(Text, primary_key=True) + registration: Mapped[dict[str, Any]] = mapped_column( + "registration_json", + JsonObjectText, + nullable=False, + ) + next_sequence: Mapped[int] = mapped_column(Integer, nullable=False, default=1) + upload_state: Mapped[str] = mapped_column( + Text, + nullable=False, + default=UPLOAD_STATE_PENDING, + ) + local_only_reason: Mapped[str | None] = mapped_column(Text) + updated_at: Mapped[str] = mapped_column(Text, nullable=False) + events: Mapped[list[TelemetryEventRecord]] = relationship( + back_populates="run", + cascade="all, delete-orphan", + ) + + +class TelemetryEventRecord(SpoolModel): + """One ordered telemetry event awaiting collector acknowledgement.""" + + __tablename__ = "telemetry_events" + __table_args__ = ( + UniqueConstraint( + "run_id", + "producer_id", + "sequence", + name="uq_telemetry_events_run_producer_sequence", + ), + ForeignKeyConstraint( + ["run_id", "producer_id"], + ["telemetry_runs.run_id", "telemetry_runs.producer_id"], + name="fk_telemetry_events_run", + ), + Index( + "telemetry_events_run_sequence", + "run_id", + "producer_id", + "sequence", + ), + ) + + event_id: Mapped[str] = mapped_column(Text, primary_key=True) + run_id: Mapped[str] = mapped_column(Text, nullable=False) + producer_id: Mapped[str] = mapped_column(Text, nullable=False) + sequence: Mapped[int] = mapped_column(Integer, nullable=False) + payload: Mapped[dict[str, Any]] = mapped_column( + "payload_json", + JsonObjectText, + nullable=False, + ) + created_at: Mapped[str] = mapped_column(Text, nullable=False) + run: Mapped[TelemetryRunRecord] = relationship(back_populates="events") diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/resources.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/resources.py new file mode 100644 index 000000000..d23d6f46c --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/resources.py @@ -0,0 +1,137 @@ +"""Process-tree resource sampling for telemetry events.""" + +from __future__ import annotations + +import os +from pathlib import Path +from typing import Any + +try: + import psutil +except ModuleNotFoundError: # Base installs use the standard-library fallback. + psutil = None + + +def _empty_sample(peak_rss: int) -> dict[str, Any]: + return { + "cpu_user_seconds": 0.0, + "cpu_system_seconds": 0.0, + "rss_bytes": 0, + "peak_rss_bytes": peak_rss, + } + + +class ProcessTreeSampler: + """Collect cumulative CPU and resident memory for a build process tree.""" + + def __init__(self, parent_pid: int) -> None: + self.parent_pid = parent_pid + self._peak_rss = 0 + self._parent_create_time = self._create_time() + + def _create_time(self) -> float | str | None: + if psutil is not None: + try: + return psutil.Process(self.parent_pid).create_time() + except (psutil.Error, OSError): + return None + stat_path = Path(f"/proc/{self.parent_pid}/stat") + try: + return stat_path.read_text().rsplit(")", 1)[1].split()[19] + except (OSError, IndexError): + return None + + def parent_alive(self) -> bool: + """Return whether the sampled process still has its original identity.""" + + if psutil is not None: + try: + process = psutil.Process(self.parent_pid) + return ( + self._parent_create_time is not None + and process.create_time() == self._parent_create_time + and process.is_running() + and process.status() != psutil.STATUS_ZOMBIE + ) + except (psutil.Error, OSError): + return False + try: + os.kill(self.parent_pid, 0) + except OSError: + return False + current = self._create_time() + # /proc supplies a process-start tick that protects against PID reuse. + # On platforms without it, existence is the best standard-library test. + return self._parent_create_time is None or current == self._parent_create_time + + def sample(self) -> dict[str, Any]: + """Sample the parent and every currently visible child process.""" + + if psutil is None: + return self._fallback_sample() + processes = [] + try: + parent = psutil.Process(self.parent_pid) + processes = [parent, *parent.children(recursive=True)] + except (psutil.Error, OSError): + pass + user = 0.0 + system = 0.0 + rss = 0 + for process in processes: + try: + cpu = process.cpu_times() + user += float(cpu.user) + float(getattr(cpu, "children_user", 0.0)) + system += float(cpu.system) + float( + getattr(cpu, "children_system", 0.0) + ) + rss += int(process.memory_info().rss) + except (psutil.Error, OSError): + continue + self._peak_rss = max(self._peak_rss, rss) + return { + "cpu_user_seconds": user, + "cpu_system_seconds": system, + "rss_bytes": rss, + "peak_rss_bytes": self._peak_rss, + } + + def _fallback_sample(self) -> dict[str, Any]: + """Sample Linux process trees when the country engine is not installed.""" + + process_rows: dict[int, tuple[int, float, float, int]] = {} + try: + clock_ticks = float(os.sysconf("SC_CLK_TCK")) + page_size = int(os.sysconf("SC_PAGE_SIZE")) + except (AttributeError, OSError, ValueError): + return _empty_sample(self._peak_rss) + for stat_path in Path("/proc").glob("[0-9]*/stat"): + try: + pid = int(stat_path.parent.name) + fields = stat_path.read_text().rsplit(")", 1)[1].split() + process_rows[pid] = ( + int(fields[1]), + int(fields[11]) / clock_ticks, + int(fields[12]) / clock_ticks, + int(fields[21]) * page_size, + ) + except (OSError, ValueError, IndexError): + continue + selected = {self.parent_pid} + changed = True + while changed: + changed = False + for pid, (parent, *_rest) in process_rows.items(): + if parent in selected and pid not in selected: + selected.add(pid) + changed = True + user = sum(process_rows[pid][1] for pid in selected if pid in process_rows) + system = sum(process_rows[pid][2] for pid in selected if pid in process_rows) + rss = sum(process_rows[pid][3] for pid in selected if pid in process_rows) + self._peak_rss = max(self._peak_rss, rss) + return { + "cpu_user_seconds": user, + "cpu_system_seconds": system, + "rss_bytes": rss, + "peak_rss_bytes": self._peak_rss, + } diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/runtime.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/runtime.py new file mode 100644 index 000000000..8f9e1c847 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/runtime.py @@ -0,0 +1,215 @@ +"""Unix-socket runtime for the telemetry emitter service.""" + +from __future__ import annotations + +import json +import os +import socket +import threading +import time +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +from microcosm.build.telemetry_emitter_service.collector import CollectorDelivery +from microcosm.build.telemetry_emitter_service.constants import ( + DEFAULT_DRAIN_SECONDS, + DRAIN_RETRY_SECONDS, + EVENT_OBJECT_ERROR, + FAILURE_CLASS_UNEXPECTED_PROCESS_EXIT, + LOCAL_MESSAGE_TOO_LARGE_ERROR, + MINIMUM_HEARTBEAT_SECONDS, + SOCKET_ACCEPT_TIMEOUT_SECONDS, + SOCKET_CONNECTION_TIMEOUT_SECONDS, + SOCKET_LISTEN_BACKLOG, + UNSUPPORTED_ACTION_ERROR, + WORKER_INTERVAL_SECONDS, +) +from microcosm.build.telemetry_emitter_service.resources import ProcessTreeSampler +from microcosm.build.telemetry_emitter_service.spool import EventSpool +from microcosm.build.telemetry_emitter_service.timestamps import utc_now +from microcosm.build.telemetry_protocol import ( + ACTION_CLOSE, + ACTION_EVENT, + ACTION_PING, + EVENT_TYPE_HEARTBEAT, + EVENT_TYPE_RUN, + LOCAL_ACKNOWLEDGEMENT_ERROR, + LOCAL_ACKNOWLEDGEMENT_OK, + LOCAL_MESSAGE_DELIMITER, + LOCAL_SOCKET_READ_BYTES, + MAX_LOCAL_MESSAGE_BYTES, + STAGE_COMPLETE, + STAGE_CREATED, + STAGE_FAILED, + STATUS_FAILED, + STATUS_PROGRESS, + UNEXPECTED_PROCESS_EXIT_MESSAGE, +) + + +def _heartbeat_event(stage_id: str) -> dict[str, Any]: + return { + "timestamp": utc_now(), + "event_type": EVENT_TYPE_HEARTBEAT, + "stage_id": stage_id, + "status": STATUS_PROGRESS, + "message": None, + "details": {}, + } + + +def _unexpected_exit_event(stage_id: str) -> dict[str, Any]: + return { + "timestamp": utc_now(), + "event_type": EVENT_TYPE_RUN, + "stage_id": STAGE_FAILED, + "status": STATUS_FAILED, + "message": UNEXPECTED_PROCESS_EXIT_MESSAGE, + "details": { + "failure_class": FAILURE_CLASS_UNEXPECTED_PROCESS_EXIT, + "failed_during": stage_id, + }, + } + + +class EmitterService: + """Private local socket server with a concurrent delivery worker.""" + + def __init__( + self, + *, + socket_path: Path, + registration: Mapping[str, Any], + spool: EventSpool, + delivery: CollectorDelivery, + sampler: ProcessTreeSampler, + heartbeat_seconds: float, + drain_seconds: float = DEFAULT_DRAIN_SECONDS, + ) -> None: + self.socket_path = socket_path + self.registration = dict(registration) + self.spool = spool + self.delivery = delivery + self.sampler = sampler + self.heartbeat_seconds = max( + MINIMUM_HEARTBEAT_SECONDS, + heartbeat_seconds, + ) + self.drain_seconds = max(0.0, drain_seconds) + self._stop = threading.Event() + self._last_stage = STAGE_CREATED + + def run(self) -> None: + """Serve local messages until the client closes or exits.""" + + self.spool.register(self.registration) + self.socket_path.parent.mkdir(parents=True, exist_ok=True) + try: + self.socket_path.unlink(missing_ok=True) + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as server: + server.bind(str(self.socket_path)) + os.chmod(self.socket_path, 0o600) + server.listen(SOCKET_LISTEN_BACKLOG) + server.settimeout(SOCKET_ACCEPT_TIMEOUT_SECONDS) + worker = threading.Thread(target=self._worker, daemon=True) + worker.start() + self._serve(server) + worker.join(timeout=self.drain_seconds + WORKER_INTERVAL_SECONDS) + finally: + self.socket_path.unlink(missing_ok=True) + try: + self.socket_path.parent.rmdir() + except OSError: + pass + + def _serve(self, server: socket.socket) -> None: + while not self._stop.is_set(): + try: + connection, _ = server.accept() + except TimeoutError: + continue + with connection: + connection.settimeout(SOCKET_CONNECTION_TIMEOUT_SECONDS) + response = self._serve_connection(connection) + try: + connection.sendall(response) + except OSError: + pass + + def _serve_connection(self, connection: socket.socket) -> bytes: + try: + message = self._read_message(connection) + self._handle(message) + except Exception: + return LOCAL_ACKNOWLEDGEMENT_ERROR + return LOCAL_ACKNOWLEDGEMENT_OK + + @staticmethod + def _read_message(connection: socket.socket) -> dict[str, Any]: + chunks = bytearray() + while len(chunks) <= MAX_LOCAL_MESSAGE_BYTES: + data = connection.recv(LOCAL_SOCKET_READ_BYTES) + if not data: + break + chunks.extend(data) + if LOCAL_MESSAGE_DELIMITER in data: + break + if len(chunks) > MAX_LOCAL_MESSAGE_BYTES: + raise ValueError(LOCAL_MESSAGE_TOO_LARGE_ERROR) + return json.loads(bytes(chunks).split(LOCAL_MESSAGE_DELIMITER, 1)[0]) + + def _handle(self, message: Mapping[str, Any]) -> None: + action = message.get("action") + if action == ACTION_EVENT: + event = message.get("event") + if not isinstance(event, Mapping): + raise ValueError(EVENT_OBJECT_ERROR) + stage_id = event.get("stage_id") + if isinstance(stage_id, str) and stage_id not in { + STAGE_COMPLETE, + STAGE_FAILED, + }: + self._last_stage = stage_id + self.spool.append( + self.registration, + event, + resources=self.sampler.sample(), + ) + elif action == ACTION_CLOSE: + self._stop.set() + elif action == ACTION_PING: + return + else: + raise ValueError(UNSUPPORTED_ACTION_ERROR) + + def _worker(self) -> None: + next_heartbeat = time.monotonic() + self.heartbeat_seconds + while not self._stop.wait(WORKER_INTERVAL_SECONDS): + now = time.monotonic() + # Sample every worker iteration so short-lived build children are + # much less likely to disappear between stage and heartbeat events. + self.sampler.sample() + if now >= next_heartbeat: + self.spool.append( + self.registration, + _heartbeat_event(self._last_stage), + resources=self.sampler.sample(), + ) + next_heartbeat = now + self.heartbeat_seconds + self.delivery.flush_once() + if not self.sampler.parent_alive(): + self.spool.append( + self.registration, + _unexpected_exit_event(self._last_stage), + resources=self.sampler.sample(), + ) + self._stop.set() + break + self._drain() + + def _drain(self) -> None: + deadline = time.monotonic() + self.drain_seconds + while self.spool.has_deliverable() and time.monotonic() < deadline: + if not self.delivery.flush_once(): + self._stop.wait(DRAIN_RETRY_SECONDS) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py new file mode 100644 index 000000000..1ef43c526 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py @@ -0,0 +1,258 @@ +"""Durable local queue for telemetry events.""" + +from __future__ import annotations + +import threading +import time +import uuid +from collections.abc import Mapping +from datetime import UTC, datetime, timedelta +from pathlib import Path +from typing import Any + +from sqlalchemy import func, select +from sqlalchemy.orm import Session + +from microcosm.build.telemetry_emitter_service.constants import ( + BATCH_SIZE, + MAX_QUEUED_BYTES, + PRUNE_INTERVAL_SECONDS, + RETENTION_DAYS, + UPLOAD_STATE_LOCAL_ONLY, + UPLOAD_STATE_PENDING, +) +from microcosm.build.telemetry_emitter_service.database import ( + create_spool_engine, + create_spool_session_factory, +) +from microcosm.build.telemetry_emitter_service.migrations import ( + upgrade_spool_database, +) +from microcosm.build.telemetry_emitter_service.models import ( + TelemetryEventRecord, + TelemetryRunRecord, + serialized_json_length, +) +from microcosm.build.telemetry_emitter_service.timestamps import utc_now +from microcosm.build.telemetry_protocol import TELEMETRY_SCHEMA_VERSION + + +class EventSpool: + """Small SQLite queue shared by successive emitter service processes.""" + + def __init__(self, path: Path | str) -> None: + self.path = Path(path) + self.path.parent.mkdir(parents=True, exist_ok=True) + try: + self.path.parent.chmod(0o700) + except OSError: + pass + self._engine = create_spool_engine(self.path) + upgrade_spool_database(self._engine) + self._session_factory = create_spool_session_factory(self._engine) + self._lock = threading.RLock() + self._last_prune_at = 0.0 + self.prune() + + def register(self, registration: Mapping[str, Any]) -> None: + """Create or refresh a producer registration.""" + + run_id = str(registration["run_id"]) + producer_id = str(registration["producer_id"]) + with self._lock, self._session_factory.begin() as session: + existing = session.get(TelemetryRunRecord, (run_id, producer_id)) + if existing is not None: + if existing.registration.get("producer_id") != registration.get( + "producer_id" + ): + raise ValueError(f"run_id {run_id!r} already has another producer") + existing.registration = dict(registration) + existing.updated_at = utc_now() + return + session.add( + TelemetryRunRecord( + run_id=run_id, + producer_id=producer_id, + registration=dict(registration), + next_sequence=1, + upload_state=UPLOAD_STATE_PENDING, + local_only_reason=None, + updated_at=utc_now(), + ) + ) + + def append( + self, + registration: Mapping[str, Any], + event: Mapping[str, Any], + *, + resources: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + """Append an event and assign its stable producer sequence.""" + + run_id = str(registration["run_id"]) + producer_id = str(registration["producer_id"]) + with self._lock, self._session_factory.begin() as session: + run = session.get(TelemetryRunRecord, (run_id, producer_id)) + if run is None: + raise KeyError(run_id) + sequence = run.next_sequence + event_id = uuid.uuid4().hex + payload = { + "schema_version": TELEMETRY_SCHEMA_VERSION, + "event_id": event_id, + "run_id": run_id, + "producer_id": producer_id, + "sequence": sequence, + "timestamp": event.get("timestamp") or utc_now(), + "event_type": event["event_type"], + "stage_id": event.get("stage_id"), + "status": event["status"], + "message": event.get("message"), + "details": event.get("details") or {}, + "resources": resources, + } + session.add( + TelemetryEventRecord( + event_id=event_id, + run_id=run_id, + producer_id=producer_id, + sequence=sequence, + payload=payload, + created_at=utc_now(), + ) + ) + run.next_sequence = sequence + 1 + run.updated_at = utc_now() + if time.monotonic() - self._last_prune_at >= PRUNE_INTERVAL_SECONDS: + self.prune() + return payload + + def pending_runs(self) -> list[dict[str, Any]]: + """Return registrations that have events eligible for delivery.""" + + statement = ( + select(TelemetryRunRecord) + .join(TelemetryRunRecord.events) + .where(TelemetryRunRecord.upload_state == UPLOAD_STATE_PENDING) + .order_by(TelemetryRunRecord.updated_at) + .distinct() + ) + with self._lock, self._session_factory() as session: + runs = session.scalars(statement).all() + return [dict(run.registration) for run in runs] + + def make_local_only( + self, + run_id: str, + producer_id: str, + reason: str, + ) -> None: + """Permanently exclude one producer's queued events from upload.""" + + with self._lock, self._session_factory.begin() as session: + run = session.get(TelemetryRunRecord, (run_id, producer_id)) + if run is None: + return + run.upload_state = UPLOAD_STATE_LOCAL_ONLY + run.local_only_reason = reason + run.updated_at = utc_now() + + def has_deliverable(self) -> bool: + """Return whether any queued event remains eligible for delivery.""" + + statement = ( + select(TelemetryEventRecord) + .join(TelemetryEventRecord.run) + .where(TelemetryRunRecord.upload_state == UPLOAD_STATE_PENDING) + .limit(1) + ) + with self._lock, self._session_factory() as session: + return session.scalar(statement) is not None + + def batch( + self, + run_id: str, + producer_id: str, + limit: int = BATCH_SIZE, + ) -> list[dict[str, Any]]: + """Return the next ordered batch for one producer.""" + + statement = ( + select(TelemetryEventRecord) + .where( + TelemetryEventRecord.run_id == run_id, + TelemetryEventRecord.producer_id == producer_id, + ) + .order_by(TelemetryEventRecord.sequence) + .limit(limit) + ) + with self._lock, self._session_factory() as session: + events = session.scalars(statement).all() + return [dict(event.payload) for event in events] + + def acknowledge(self, event_ids: list[str]) -> None: + """Remove events acknowledged by the collector.""" + + if not event_ids: + return + statement = select(TelemetryEventRecord).where( + TelemetryEventRecord.event_id.in_(event_ids) + ) + with self._lock, self._session_factory.begin() as session: + for event in session.scalars(statement): + session.delete(event) + + def has_pending(self) -> bool: + """Return whether any event remains in local storage.""" + + statement = select(TelemetryEventRecord).limit(1) + with self._lock, self._session_factory() as session: + return session.scalar(statement) is not None + + def prune(self) -> None: + """Enforce the age and total-size retention limits.""" + + cutoff = (datetime.now(UTC) - timedelta(days=RETENTION_DAYS)).isoformat() + with self._lock, self._session_factory.begin() as session: + expired_events = session.scalars( + select(TelemetryEventRecord).where( + TelemetryEventRecord.created_at < cutoff + ) + ) + for event in expired_events: + session.delete(event) + session.flush() + stored_characters = session.scalar( + select( + func.coalesce( + func.sum(func.length(TelemetryEventRecord.payload)), + 0, + ) + ) + ) + excess = int(stored_characters or 0) - MAX_QUEUED_BYTES + if excess > 0: + self._remove_oldest_bytes(session, excess) + expired_runs = session.scalars( + select(TelemetryRunRecord).where( + TelemetryRunRecord.updated_at < cutoff, + ~TelemetryRunRecord.events.any(), + ) + ) + for run in expired_runs: + session.delete(run) + self._last_prune_at = time.monotonic() + + @staticmethod + def _remove_oldest_bytes(session: Session, excess: int) -> None: + statement = select(TelemetryEventRecord).order_by( + TelemetryEventRecord.created_at, + TelemetryEventRecord.sequence, + ) + removed = 0 + for event in session.scalars(statement): + session.delete(event) + removed += serialized_json_length(event.payload) + if removed >= excess: + break diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/timestamps.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/timestamps.py new file mode 100644 index 000000000..3f1f39a91 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/timestamps.py @@ -0,0 +1,11 @@ +"""Timestamp helpers shared by the telemetry emitter service.""" + +from __future__ import annotations + +from datetime import UTC, datetime + + +def utc_now() -> str: + """Return the current UTC timestamp in ISO 8601 format.""" + + return datetime.now(UTC).isoformat() diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_identity.py b/packages/microcosm-build/src/microcosm/build/telemetry_identity.py new file mode 100644 index 000000000..0fa5e5b80 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_identity.py @@ -0,0 +1,53 @@ +"""Runtime identity attached to the first telemetry event.""" + +from __future__ import annotations + +import os +import subprocess +from importlib import metadata +from platform import platform +from typing import Any, Final + +_IDENTITY_DISTRIBUTIONS: Final = ( + "microcosm-build", + "microcosm-graph", + "policyengine-us", + "policyengine-uk", +) +_GIT_IDENTITY_TIMEOUT_SECONDS: Final = 1.0 + + +def runtime_identity() -> dict[str, Any]: + """Describe source, host capacity, and installed runtime versions.""" + + try: + commit = subprocess.run( + ["git", "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + timeout=_GIT_IDENTITY_TIMEOUT_SECONDS, + ).stdout.strip() + except (OSError, subprocess.SubprocessError): + commit = None + try: + memory_bytes = int(os.sysconf("SC_PHYS_PAGES")) * int( + os.sysconf("SC_PAGE_SIZE") + ) + except (AttributeError, OSError, ValueError): + memory_bytes = None + versions = {} + for distribution in _IDENTITY_DISTRIBUTIONS: + try: + versions[distribution] = metadata.version(distribution) + except metadata.PackageNotFoundError: + continue + return { + "git_commit": commit, + "host": { + "platform": platform(), + "cpu_count": os.cpu_count(), + "memory_bytes": memory_bytes, + }, + "runtime": versions, + } diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_protocol.py b/packages/microcosm-build/src/microcosm/build/telemetry_protocol.py new file mode 100644 index 000000000..1bc3f6839 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_protocol.py @@ -0,0 +1,66 @@ +"""Shared constants for local and hosted Microcosm telemetry messages.""" + +from __future__ import annotations + +from types import MappingProxyType +from typing import Final, Literal + +type TelemetryAction = Literal["event", "close", "ping"] +type TelemetryEventType = Literal[ + "run", "stage", "progress", "calibration", "heartbeat" +] +type TelemetryStatus = Literal["started", "progress", "completed", "failed"] + +TELEMETRY_SCHEMA_VERSION: Final = 1 + +ACTION_EVENT: Final = "event" +ACTION_CLOSE: Final = "close" +ACTION_PING: Final = "ping" + +EVENT_TYPE_RUN: Final = "run" +EVENT_TYPE_STAGE: Final = "stage" +EVENT_TYPE_PROGRESS: Final = "progress" +EVENT_TYPE_CALIBRATION: Final = "calibration" +EVENT_TYPE_HEARTBEAT: Final = "heartbeat" + +STATUS_STARTED: Final = "started" +STATUS_PROGRESS: Final = "progress" +STATUS_COMPLETED: Final = "completed" +STATUS_FAILED: Final = "failed" + +STAGE_CREATED: Final = "created" +STAGE_CALIBRATING: Final = "calibrating" +STAGE_COMPLETE: Final = "complete" +STAGE_FAILED: Final = "failed" + +CALIBRATION_EVENT_KIND: Final = "calibration_epoch" +BUILD_STARTED_MESSAGE: Final = "Microcosm build started." +BUILD_COMPLETED_MESSAGE: Final = "Microcosm build completed." +UNEXPECTED_PROCESS_EXIT_MESSAGE: Final = ( + "The build process exited without reporting completion." +) + +LOCAL_ACKNOWLEDGEMENT_OK: Final = b"ok\n" +LOCAL_ACKNOWLEDGEMENT_ERROR: Final = b"error\n" +LOCAL_MESSAGE_DELIMITER: Final = b"\n" +LOCAL_PING_MESSAGE: Final = b'{"action":"ping"}\n' + +MAX_TELEMETRY_TEXT_CHARS: Final = 2_000 +MAX_TELEMETRY_MESSAGE_CHARS: Final = 500 +MAX_TELEMETRY_DETAILS_DEPTH: Final = 6 +MAX_TELEMETRY_COLLECTION_ITEMS: Final = 200 +MAX_TELEMETRY_DETAILS_BYTES: Final = 8_192 +MAX_LOCAL_MESSAGE_BYTES: Final = 1_048_576 +LOCAL_SOCKET_READ_BYTES: Final = 65_536 +LOCAL_ACKNOWLEDGEMENT_READ_BYTES: Final = 16 + +SEQUENTIAL_STATUS_MAP: Final = MappingProxyType( + { + "failed": STATUS_FAILED, + "completed": STATUS_COMPLETED, + "passed": STATUS_COMPLETED, + "progress": STATUS_PROGRESS, + "started": STATUS_STARTED, + "running": STATUS_STARTED, + } +) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_sanitization.py b/packages/microcosm-build/src/microcosm/build/telemetry_sanitization.py new file mode 100644 index 000000000..cabd12f33 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_sanitization.py @@ -0,0 +1,94 @@ +"""Size bounds and credential redaction for telemetry payloads.""" + +from __future__ import annotations + +import json +import re +from collections.abc import Mapping +from pathlib import Path +from typing import Any, Final + +from microcosm.build.telemetry_protocol import ( + MAX_TELEMETRY_COLLECTION_ITEMS, + MAX_TELEMETRY_DETAILS_BYTES, + MAX_TELEMETRY_DETAILS_DEPTH, + MAX_TELEMETRY_TEXT_CHARS, +) + +_SENSITIVE_KEY_PARTS: Final = ( + "authorization", + "credential", + "password", + "secret", + "token", + "traceback", +) +_SECRET_TEXT: Final = re.compile( + r"(?i)(?:bearer\s+[^\s]+|(?:token|secret|password|credential)\s*[:=]\s*[^\s]+|hf_[A-Za-z0-9_-]{8,})" +) +_REDACTED_VALUE: Final = "[redacted]" +_MAXIMUM_DEPTH_VALUE: Final = "[maximum depth]" +_DETAILS_TRUNCATED_KEY: Final = "telemetry_details_truncated" + + +def sanitize_text(value: str, *, limit: int = MAX_TELEMETRY_TEXT_CHARS) -> str: + """Redact credential-like text and apply a character limit.""" + + return _SECRET_TEXT.sub(_REDACTED_VALUE, value)[:limit] + + +def sanitize_json(value: Any, *, depth: int = 0) -> Any: + """Return bounded JSON-compatible telemetry without model objects.""" + + if depth >= MAX_TELEMETRY_DETAILS_DEPTH: + return _MAXIMUM_DEPTH_VALUE + if isinstance(value, Mapping): + result = {} + for key, item in list(value.items())[:MAX_TELEMETRY_COLLECTION_ITEMS]: + name = str(key) + if any(part in name.lower() for part in _SENSITIVE_KEY_PARTS): + result[name] = _REDACTED_VALUE + else: + result[name] = sanitize_json(item, depth=depth + 1) + return result + if isinstance(value, (list, tuple)): + return [ + sanitize_json(item, depth=depth + 1) + for item in value[:MAX_TELEMETRY_COLLECTION_ITEMS] + ] + if isinstance(value, Path): + return str(value) + if isinstance(value, float): + return value if value == value and abs(value) != float("inf") else None + if isinstance(value, str): + return sanitize_text(value) + if isinstance(value, (int, bool)) or value is None: + return value + item = getattr(value, "item", None) + if callable(item): + return sanitize_json(item()) + return sanitize_text(str(value)) + + +def sanitize_details(value: Mapping[str, Any]) -> dict[str, Any]: + """Return a redacted details object within the wire-size limit.""" + + sanitized = sanitize_json(value) + assert isinstance(sanitized, dict) + if ( + len(json.dumps(sanitized, separators=(",", ":")).encode()) + <= MAX_TELEMETRY_DETAILS_BYTES + ): + return sanitized + compact: dict[str, Any] = {_DETAILS_TRUNCATED_KEY: True} + for key, item in sanitized.items(): + if not (isinstance(item, (str, int, float, bool)) or item is None): + continue + candidate = {**compact, key: item} + if ( + len(json.dumps(candidate, separators=(",", ":")).encode()) + > MAX_TELEMETRY_DETAILS_BYTES + ): + break + compact[key] = item + return compact diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py index ff684bde8..cc89d2eca 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py @@ -174,10 +174,12 @@ from .rowwise_staging import ( STAGED_DATASET_PHASES, STAGING_UPLOAD_INTERVAL_SECONDS, - create_staging_telemetry, - fail_staging_telemetry, - finalize_staging_telemetry, + create_staging_run_bundle, + emit_calibration_progress, + fail_staging_run_bundle, + finalize_staging_run_bundle, gate_statuses, + graph_progress, preflight_staged_dataset, replace_manifest, stage, @@ -185,6 +187,7 @@ stage_sample, staging_delivery, staging_epoch_every, + start_telemetry_emitter, thinned_epochs, ) from .size_checkpoint import uk_size_checkpoint_identity @@ -212,6 +215,33 @@ ) +def _run_graph_with_progress(compiled, **kwargs): + """Run a graph and emit one lightweight update per completed node.""" + + completed = 0 + previous = time.monotonic() + total = len(compiled.order) + + def observe(node_id, _population) -> None: + nonlocal completed, previous + now = time.monotonic() + completed += 1 + graph_progress( + node_id=node_id, + done=completed, + total=total, + elapsed_seconds=now - previous, + ) + previous = now + + return run_graph( + compiled, + _population_observer=observe, + _population_observer_detach=False, + **kwargs, + ) + + def _target_geographies(value: str) -> tuple[str, ...] | None: if value == "all": return None @@ -571,6 +601,9 @@ def _solve_observer(args: argparse.Namespace, telemetry): from .solve_progress import uk_solve_progress_callback sinks = [uk_solve_progress_callback(_stderr_progress)] + sinks.append( + thinned_epochs(emit_calibration_progress, every=staging_epoch_every(args)) + ) if telemetry is not None: sinks.append( thinned_epochs( @@ -1118,11 +1151,11 @@ def _close_attempt( manifest=manifest, output_paths={"manifest": output / MANIFEST_FILENAME}, run_id=state.build_id if telemetry is None else telemetry.run_id, - telemetry=telemetry, + staging_bundle=telemetry, ) append_phase(state, STAGED_DATASET_PHASES[staged_dataset["status"]]) try: - finalize_staging_telemetry(args, telemetry) + finalize_staging_run_bundle(args, telemetry) finally: manifest["staging_delivery"] = staging_delivery(telemetry) manifest["staged_dataset"] = staged_dataset @@ -1136,7 +1169,7 @@ def _close_attempt( output / f"{stem}.h5", repository_hint=REPOSITORY ) else: - finalize_staging_telemetry(args, telemetry) + finalize_staging_run_bundle(args, telemetry) spool_path = record_candidate_attempt( state=state, started_at=attempt["started_at"], @@ -1283,7 +1316,7 @@ def _execute_full_build( ): if endpoint not in node_ids: continue - checkpoint = run_graph( + checkpoint = _run_graph_with_progress( compile_graph(_through(graph, endpoint)), sources=sources, store=store, @@ -1295,7 +1328,7 @@ def _execute_full_build( # Persist preflight outcomes before any solver can reject them. stage(telemetry, "target_compilation", "started") preflight_graph = _through(graph, "uk.full.gates.preflight") - preflight = run_graph( + preflight = _run_graph_with_progress( compile_graph(preflight_graph), sources=sources, store=store, @@ -1331,7 +1364,7 @@ def _execute_full_build( epoch_every=staging_epoch_every(args), resumed_from_checkpoint=args.resume_size_checkpoint is not None, ) - manifest = run_graph( + manifest = _run_graph_with_progress( compile_graph(_through(graph, "uk.full.gates.calibrated")), sources=sources, store=store, @@ -1408,7 +1441,7 @@ def _execute_full_build( if not enforcement["artifact_permitted"]: return 1 stage(telemetry, "output_bundle", "started") - manifest = run_graph( + manifest = _run_graph_with_progress( compile_graph(graph), sources=sources, store=store, @@ -1488,7 +1521,7 @@ def _execute_full_build( spine_provenance=prepared.spine_provenance, comparison_sources=tuple(comparisons), ) - final = run_graph( + final = _run_graph_with_progress( compile_graph(graph), sources={ **sources, @@ -1874,7 +1907,7 @@ def _execute_national_build( resume = args.resume # 1. The bound checkpoint: its provenance artifact is the solve's admission. stage(telemetry, "input_loading", "started") - checkpoint = run_graph( + checkpoint = _run_graph_with_progress( compile_graph(_through(graph, "uk.full.spine_checkpoint")), sources=sources, store=store, @@ -1895,7 +1928,7 @@ def _execute_national_build( ) # 2. The register: compiled inside the graph, frozen beside the outputs. stage(telemetry, "target_compilation", "started") - targets = run_graph( + targets = _run_graph_with_progress( compile_graph(_through(graph, NATIONAL_TARGETS_NODE)), sources=sources, store=store, @@ -1948,7 +1981,7 @@ def _execute_national_build( epochs=int(args.epochs), epoch_every=staging_epoch_every(args), ) - manifest = run_graph( + manifest = _run_graph_with_progress( compile_graph(_through(graph, NATIONAL_GATES_NODE)), sources=sources, store=store, @@ -2101,7 +2134,7 @@ def _execute_national_build( size_bytes=paths["dataset"].stat().st_size, ) continued = add_uk_national_readback(graph, population=national.population) - final = run_graph( + final = _run_graph_with_progress( compile_graph(continued), sources={**sources, "exported_dataset": paths["dataset"]}, store=store, @@ -2241,7 +2274,7 @@ def _close_national_attempt( manifest=manifest, output_paths=paths, run_id=state.build_id if telemetry is None else telemetry.run_id, - telemetry=telemetry, + staging_bundle=telemetry, ) append_phase(state, STAGED_DATASET_PHASES[staged_dataset["status"]]) evaluation = national_role.evaluate_against_incumbent( @@ -2254,7 +2287,7 @@ def _close_national_attempt( out_dir=output, ) try: - finalize_staging_telemetry(args, telemetry) + finalize_staging_run_bundle(args, telemetry) finally: delivery = staging_delivery(telemetry) build_record = json.loads(paths["build_record"].read_text()) @@ -2288,7 +2321,7 @@ def _close_national_attempt( }, ) else: - finalize_staging_telemetry(args, telemetry) + finalize_staging_run_bundle(args, telemetry) spool_path = record_candidate_attempt( state=state, started_at=attempt["started_at"], @@ -2309,26 +2342,26 @@ def _close_national_attempt( def _national_main(args: argparse.Namespace) -> int: """The national role's envelope: the seam's attempt id and pipeline, the graph build.""" posture = posture_of(args) - national_role.require_bound_input(args) - if args.dry_run: - return national_role.national_dry_run( - args, - operation_inventory=lambda: prepare_national_build( - args - ).national.operation_inventory(), - ) - # Argument refusals cost nothing, the occupied output directory included; - # the credential check reaches the Hub, so it runs last, still before any - # input is read. - refuse_occupied_national_output(args) - preflight_staged_dataset(args) started_at = time.perf_counter() started_ts = datetime.now(UTC) - digest = preflight_digest(posture.pipeline) + build_id = new_uk_calibration_attempt_id(timestamp=started_ts) + emitter = start_telemetry_emitter( + args, + build_id=build_id, + run_kind="dry_run" if args.dry_run else "calibration", + ) + emitter.transition_stage( + "preflight", + message="Validating UK national build inputs and configuration.", + ) + try: + digest = preflight_digest(posture.pipeline) + except BaseException as error: + fail_staging_run_bundle(None, error) + raise state = AttemptState( - # The attempt id is minted before telemetry opens so the staging run - # id and the Logbook row agree, as the seam minted it. - build_id=new_uk_calibration_attempt_id(timestamp=started_ts), + # The staging run id and Logbook row use the same attempt id. + build_id=build_id, identity_digest=digest, input_pins_digest=digest, phases_reached=["attempt_started"], @@ -2339,8 +2372,31 @@ def _national_main(args: argparse.Namespace) -> int: } }, ) - predecessor = resolve_predecessor(args.logbook_prev_row_digest) - telemetry = create_staging_telemetry(args, build_id=state.build_id) + if args.dry_run: + try: + status = national_role.national_dry_run( + args, + operation_inventory=lambda: prepare_national_build( + args + ).national.operation_inventory(), + ) + except BaseException as error: + fail_staging_run_bundle(None, error) + raise + finalize_staging_run_bundle(args, None) + return status + + try: + national_role.require_bound_input(args) + # These checks must precede input loading. The emitter is already + # running so a refusal is visible as a failed attempt. + refuse_occupied_national_output(args) + preflight_staged_dataset(args) + predecessor = resolve_predecessor(args.logbook_prev_row_digest) + telemetry = create_staging_run_bundle(args, build_id=state.build_id) + except BaseException as error: + fail_staging_run_bundle(None, error) + raise attempt = { "state": state, "started_at": started_at, @@ -2365,13 +2421,13 @@ def _national_main(args: argparse.Namespace) -> int: pipeline=posture.pipeline, disposition="discarded", ) - fail_staging_telemetry(telemetry, interrupt) + fail_staging_run_bundle(telemetry, interrupt) raise except Exception as error: _record_failure( args, error, state=state, attempt=attempt, pipeline=posture.pipeline ) - fail_staging_telemetry(telemetry, error) + fail_staging_run_bundle(telemetry, error) print(f"UK national build failed: {error}", file=sys.stderr) return 1 @@ -2454,22 +2510,29 @@ def main(argv: list[str] | None = None) -> int: return _national_main(args) if args.candidate_clone_counts is not None and not args.dry_run: raise ValueError("--candidate-clone-counts is valid only with --dry-run.") - if args.dry_run: - # Dry runs plan without solving or writing and record no Logbook - # row on any path, so they need no chain configuration. - return _dry_run(args) - # Argument refusals above cost nothing; the credential check reaches the - # Hub, so it runs last, still before any input is read. - preflight_staged_dataset(args) started_at = time.perf_counter() started_ts = datetime.now(UTC) - digest = preflight_digest(posture.pipeline) + build_id = new_candidate_build_id( + seed=args.seed, + timestamp=started_ts, + rung=UK_SAMPLE_RUNG_TOKENS[args.sample_fraction], + ) + emitter = start_telemetry_emitter( + args, + build_id=build_id, + run_kind="dry_run" if args.dry_run else "calibration", + ) + emitter.transition_stage( + "preflight", + message="Validating UK dense build inputs and configuration.", + ) + try: + digest = preflight_digest(posture.pipeline) + except BaseException as error: + fail_staging_run_bundle(None, error) + raise state = AttemptState( - build_id=new_candidate_build_id( - seed=args.seed, - timestamp=started_ts, - rung=UK_SAMPLE_RUNG_TOKENS[args.sample_fraction], - ), + build_id=build_id, identity_digest=digest, input_pins_digest=digest, phases_reached=["attempt_started"], @@ -2480,11 +2543,26 @@ def main(argv: list[str] | None = None) -> int: } }, ) - # Logbook chain configuration is validated before any terminal work: a - # malformed or conflicting head refuses the run with no row and no side - # effects. - predecessor = resolve_predecessor(args.logbook_prev_row_digest) - telemetry = create_staging_telemetry(args, build_id=state.build_id) + if args.dry_run: + # Dry runs plan without solving or writing and record no Logbook row, + # but their hosted telemetry still has a complete lifecycle. + try: + status = _dry_run(args) + except BaseException as error: + fail_staging_run_bundle(None, error) + raise + finalize_staging_run_bundle(args, None) + return status + + try: + # The credential and chain checks still precede input loading, while + # the emitter records their failures. + preflight_staged_dataset(args) + predecessor = resolve_predecessor(args.logbook_prev_row_digest) + telemetry = create_staging_run_bundle(args, build_id=state.build_id) + except BaseException as error: + fail_staging_run_bundle(None, error) + raise attempt = { "state": state, "started_at": started_at, @@ -2508,7 +2586,7 @@ def main(argv: list[str] | None = None) -> int: prepared=prepared, disposition="discarded", ) - fail_staging_telemetry(telemetry, interrupt) + fail_staging_run_bundle(telemetry, interrupt) raise except Exception as error: _record_failure( @@ -2519,7 +2597,7 @@ def main(argv: list[str] | None = None) -> int: pipeline=posture.pipeline, prepared=prepared, ) - fail_staging_telemetry(telemetry, error) + fail_staging_run_bundle(telemetry, error) print(f"UK full build failed: {error}", file=sys.stderr) return 1 diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/national_role.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/national_role.py index 5d6f6cf94..5ffc89ce7 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/national_role.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/national_role.py @@ -34,7 +34,7 @@ parse_sha256sums, refresh_sha256sums_entry, ) -from microcosm.build.staging_v2 import StagingTelemetryV2 +from microcosm.build.staging_v2 import StagingRunBundleWriterV2 from microcosm.build.uk_runtime.calibration_run import runtime_provenance from microcosm.build.uk_runtime.diagnostics import uk_fit_by_family from microcosm.build.uk_runtime.frs_release import load_uk_frs_release @@ -480,7 +480,7 @@ def evaluate_against_incumbent( incumbent: Mapping[str, Any] | None, inputs: Mapping[str, Any], output_paths: Mapping[str, Path], - telemetry: StagingTelemetryV2 | None, + telemetry: StagingRunBundleWriterV2 | None, calibration_year: int, out_dir: Path, ) -> dict[str, Any]: diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py index ccb031c11..2f0ababfc 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py @@ -1,6 +1,6 @@ -"""Staging telemetry and staged-dataset delivery of the UK rowwise roles. +"""Staging run bundles and staged-dataset delivery for UK rowwise roles. -Two destinations, one run id: reviewed aggregate telemetry goes to the +Two destinations, one run id: reviewed aggregate build files go to the staging repository under ``runs//``, the finished bundle a manifest vouches for to the private artifact repository under ``staged//``. Both are best-effort evidence about the build, never a release, and the @@ -32,9 +32,13 @@ from microcosm.build.staging_storage import HuggingFaceDatasetStorage from microcosm.build.staging_v2 import ( StagingContractError, - StagingTelemetryV2, + StagingRunBundleWriterV2, disabled_staging_delivery, ) +from microcosm.build.telemetry_emitter import ( + LocalTelemetryEmitter, + start_local_telemetry_emitter_service, +) from microcosm.build.uk_runtime.rowwise_cli import ( BUDGET_ITERS, MANIFEST_FILENAME, @@ -50,14 +54,17 @@ "STAGING_MAX_EPOCH_ROWS", "STAGING_UPLOAD_INTERVAL_SECONDS", "add_staging_artifact", - "create_staging_telemetry", - "fail_staging_telemetry", - "finalize_staging_telemetry", + "create_staging_run_bundle", + "emit_calibration_progress", + "fail_staging_run_bundle", + "finalize_staging_run_bundle", "fit_summary", "gate_statuses", + "graph_progress", "preflight_staged_dataset", "publish_staged_files", "replace_manifest", + "start_telemetry_emitter", "stage", "stage_dataset", "staged_dataset_mode", @@ -66,13 +73,13 @@ "thinned_epochs", ] -# Best-effort telemetry upload cadence. The Hub allows about 128 commits per +# Best-effort staging-bundle upload cadence. The Hub allows about 128 commits per # hour per repository and one cycle is up to eight single-file commits, so # the shared 30-second default exhausts the budget on a multi-hour solve # and loses uploads (the v20 national run did); five minutes keeps a # 1,500-epoch run well inside it. STAGING_UPLOAD_INTERVAL_SECONDS = 300.0 -# Staging telemetry keeps one row per forwarded epoch in +# The staging bundle keeps one row per forwarded epoch in # calibration_progress.json and one event in events.ndjson, both under the # contract's 5 MiB remote cap. A size run at 2,000 epochs solves the dense # pool, up to ten full-length L0 probes and the refit: about 24,000 epochs, @@ -92,10 +99,11 @@ _STAGING_EPOCH_EVERY = STAGING_EPOCH_EVERY _STAGING_MAX_EPOCH_ROWS = STAGING_MAX_EPOCH_ROWS _STAGED_DATASET_PHASES = STAGED_DATASET_PHASES +_ACTIVE_EMITTER: LocalTelemetryEmitter | None = None def _hub_api() -> Any: - """The Hub client used for telemetry and the staged dataset (test seam).""" + """The Hub client used for the run bundle and staged dataset (test seam).""" from huggingface_hub import HfApi @@ -183,16 +191,17 @@ def _require_write_credential(storage: HuggingFaceDatasetStorage, *, hint: str) ) -def create_staging_telemetry( +def create_staging_run_bundle( args: argparse.Namespace, *, build_id: str -) -> StagingTelemetryV2 | None: +) -> StagingRunBundleWriterV2 | None: + posture = posture_of(args) + run_id = args.staging_run_id or build_id if args.no_staging: return None local_only = bool(args.staging_local_only) out_dir = args.out.expanduser().resolve() - posture = posture_of(args) - return StagingTelemetryV2( - run_id=args.staging_run_id or build_id, + return StagingRunBundleWriterV2( + run_id=run_id, country_code="GB", operation_id=posture.staging_operation_id, pipeline_id=posture.pipeline, @@ -207,16 +216,45 @@ def create_staging_telemetry( ) -_create_staging_telemetry = create_staging_telemetry +def start_telemetry_emitter( + args: argparse.Namespace, + *, + build_id: str, + run_kind: str = "calibration", +) -> LocalTelemetryEmitter: + """Start or reuse the process-local emitter for one UK build attempt.""" + + global _ACTIVE_EMITTER + posture = posture_of(args) + run_id = args.staging_run_id or build_id + if ( + _ACTIVE_EMITTER is not None + and _ACTIVE_EMITTER.available + and _ACTIVE_EMITTER.run.run_id == run_id + ): + return _ACTIVE_EMITTER + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.close() + _ACTIVE_EMITTER = start_local_telemetry_emitter_service( + run_id=run_id, + country_code="GB", + pipeline=posture.pipeline, + candidate_id=args.staging_candidate_id or build_id, + run_kind=run_kind, + ) + return _ACTIVE_EMITTER + + +_create_staging_run_bundle = create_staging_run_bundle def stage( - telemetry: StagingTelemetryV2 | None, + staging_bundle: StagingRunBundleWriterV2 | None, stage_id: str, event_status: str = "started", **details: Any, ) -> None: - """Forward one stage event to the best-effort telemetry. + """Report one stage independently to the emitter and staging bundle. A contract or content refusal of the event is reported and the event dropped; the build must never abort on its own progress report. Each @@ -224,13 +262,25 @@ def stage( the ones that follow. """ - if telemetry is None: + emitter_details = dict(details) + message = emitter_details.pop("message", None) + emitter_details.pop("force_upload", None) + if "status" in emitter_details: + emitter_details["result_status"] = emitter_details.pop("status") + if _ACTIVE_EMITTER is not None: + _ACTIVE_EMITTER.transition_stage( + stage_id, + status=event_status, + message=message, + **emitter_details, + ) + if staging_bundle is None: return try: - telemetry.stage(stage_id, event_status=event_status, **details) + staging_bundle.stage(stage_id, event_status=event_status, **details) except StagingContractError as error: print( - f"warning: staging telemetry refused the {stage_id!r} stage event " + f"warning: staging run bundle refused the {stage_id!r} stage event " f"({type(error).__name__}: {error}); the event is not staged, the " "build continues.", file=sys.stderr, @@ -241,10 +291,41 @@ def stage( _stage = stage +def emit_calibration_progress(event: Mapping[str, Any]) -> None: + """Report calibration progress only through the local emitter service.""" + + if _ACTIVE_EMITTER is not None: + _ACTIVE_EMITTER.transition_calibration_progress(event) + + +def graph_progress( + *, + node_id: str, + done: int, + total: int, + elapsed_seconds: float, +) -> None: + """Send graph work progress only through the local emitter service.""" + + if _ACTIVE_EMITTER is None: + return + _ACTIVE_EMITTER.emit( + event_type="progress", + stage_id=node_id, + status="completed", + details={ + "done": done, + "total": total, + "unit": "graph_nodes", + "elapsed_seconds": elapsed_seconds, + }, + ) + + def stage_sample( - telemetry: StagingTelemetryV2 | None, *, sample_fraction: float + staging_bundle: StagingRunBundleWriterV2 | None, *, sample_fraction: float ) -> None: - """Record the run's sampling evidence on the staging telemetry. + """Record the run's sampling evidence in the staging bundle. The contract's only sampling statement is ``{"mode": "full"}``, the f100 rung; a rung below f100 stages a null sample, as the spine builder does. @@ -253,9 +334,9 @@ def stage_sample( own fraction), so the two release roles stage the same evidence. """ - if telemetry is None or float(sample_fraction) != 1.0: + if staging_bundle is None or float(sample_fraction) != 1.0: return - telemetry.set_sample({"mode": "full"}) + staging_bundle.set_sample({"mode": "full"}) def staging_epoch_every(args: argparse.Namespace) -> int: @@ -307,7 +388,7 @@ def callback(event: dict[str, object]) -> None: except StagingContractError as error: disabled = True print( - "warning: staging telemetry refused a calibration progress row " + "warning: staging run bundle refused a calibration progress row " f"({type(error).__name__}); epoch progress is no longer forwarded, " "the solve continues.", file=sys.stderr, @@ -331,44 +412,51 @@ def gate_statuses(gate_report: Mapping[str, Any]) -> dict[str, str]: _gate_statuses = gate_statuses -def fail_staging_telemetry( - telemetry: StagingTelemetryV2 | None, error: BaseException +def fail_staging_run_bundle( + staging_bundle: StagingRunBundleWriterV2 | None, error: BaseException ) -> None: - if telemetry is None or telemetry.status != "running": + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.fail(error) + if staging_bundle is None: + return + if staging_bundle.status != "running": return try: - telemetry.fail(error) - telemetry.validate_local_bundle() + staging_bundle.fail(error) + staging_bundle.validate_local_bundle() except Exception: pass -_fail_staging_telemetry = fail_staging_telemetry +_fail_staging_run_bundle = fail_staging_run_bundle -def finalize_staging_telemetry( - args: argparse.Namespace, telemetry: StagingTelemetryV2 | None +def finalize_staging_run_bundle( + args: argparse.Namespace, staging_bundle: StagingRunBundleWriterV2 | None ) -> None: - if telemetry is None: - return - try: - telemetry.complete(message="UK rowwise candidate staging run completed.") - except StagingContractError as error: - _warn_telemetry("could not close the staging run", error) - return - try: - if args.staging_read_back: - # Requested explicitly, so a failed read-back is the run's failure, - # as on the national command. - telemetry.verify_remote() - finally: + if staging_bundle is not None: try: - telemetry.validate_local_bundle() + staging_bundle.complete( + message="UK rowwise candidate staging run completed." + ) except StagingContractError as error: - _warn_telemetry("the local staging bundle does not validate", error) + _warn_telemetry("could not close the staging run", error) + else: + try: + if args.staging_read_back: + # Requested explicitly, so a failed read-back is the run's + # failure, as on the national command. + staging_bundle.verify_remote() + finally: + try: + staging_bundle.validate_local_bundle() + except StagingContractError as error: + _warn_telemetry("the local staging bundle does not validate", error) + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.complete() -_finalize_staging_telemetry = finalize_staging_telemetry +_finalize_staging_run_bundle = finalize_staging_run_bundle def _warn_telemetry(what: str, error: BaseException) -> None: @@ -380,10 +468,12 @@ def _warn_telemetry(what: str, error: BaseException) -> None: ) -def staging_delivery(telemetry: StagingTelemetryV2 | None) -> dict[str, Any]: - if telemetry is None: +def staging_delivery( + staging_bundle: StagingRunBundleWriterV2 | None, +) -> dict[str, Any]: + if staging_bundle is None: return disabled_staging_delivery("--no-staging") - return telemetry.delivery_summary + return staging_bundle.delivery_summary _staging_delivery = staging_delivery @@ -395,7 +485,7 @@ def stage_dataset( manifest: Mapping[str, Any], output_paths: Mapping[str, Path], run_id: str, - telemetry: StagingTelemetryV2 | None, + staging_bundle: StagingRunBundleWriterV2 | None, ) -> dict[str, Any]: """Stage the published bundle under ``staged//``; record, never raise. @@ -412,7 +502,13 @@ def stage_dataset( repository = ( None if mode == "local_only" else str(args.staged_dataset_repo_id).strip() ) - stage(telemetry, "dataset_staging", "started", mode=mode, repository=repository) + stage( + staging_bundle, + "dataset_staging", + "started", + mode=mode, + repository=repository, + ) statuses = gate_statuses(getattr(args, "_gate_report", {}) or {}) bundle = StagedDatasetBundle.from_manifest( output_paths["manifest"].parent, @@ -429,11 +525,11 @@ def stage_dataset( ) telemetry_reference = ( None - if telemetry is None + if staging_bundle is None else { - "repository": telemetry.repo_id, - "prefix": telemetry.repo_run_prefix, - "mode": telemetry.delivery_mode, + "repository": staging_bundle.repo_id, + "prefix": staging_bundle.repo_run_prefix, + "mode": staging_bundle.delivery_mode, } ) write_sidecars( @@ -457,7 +553,7 @@ def stage_dataset( ) print(_staged_dataset_line(delivery), file=sys.stderr, flush=True) stage( - telemetry, + staging_bundle, "dataset_staging", "completed", status=delivery["status"], @@ -467,14 +563,14 @@ def stage_dataset( file_count=len(delivery["files"]), ) add_staging_artifact( - telemetry, + staging_bundle, "staged_dataset", delivery, artifact_kind="build_metadata", classification="non_row_level", ) add_staging_artifact( - telemetry, + staging_bundle, "fit_summary", fit_summary( manifest, @@ -511,26 +607,26 @@ def _staged_dataset_line(delivery: Mapping[str, Any]) -> str: def add_staging_artifact( - telemetry: StagingTelemetryV2 | None, + staging_bundle: StagingRunBundleWriterV2 | None, logical_name: str, payload: Mapping[str, Any], *, artifact_kind: str, classification: str, ) -> None: - """Attach a reviewed aggregate JSON artifact to the telemetry run. + """Attach a reviewed aggregate JSON artifact to the staging run bundle. A content-policy refusal is reported and skipped: the telemetry is best-effort and must never fail a finished build. """ - if telemetry is None: + if staging_bundle is None: return with tempfile.TemporaryDirectory(prefix=".staging-artifact.") as scratch: source = Path(scratch) / f"{logical_name}.json" source.write_text(json_text(payload), encoding="utf-8") try: - telemetry.add_artifact( + staging_bundle.add_artifact( logical_name, source, artifact_kind=artifact_kind, diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py index 523a1f226..b0d92ccfe 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py @@ -54,9 +54,13 @@ validate_staging_arguments, ) from microcosm.build.staging_v2 import ( - StagingTelemetryV2, + StagingRunBundleWriterV2, disabled_staging_delivery, ) +from microcosm.build.telemetry_emitter import ( + LocalTelemetryEmitter, + start_local_telemetry_emitter_service, +) from microcosm.build.uk_runtime.age_tail import UKAgeTailStageTransform from microcosm.build.uk_runtime.battery_bindings import UK_GATE_REGISTRY from microcosm.build.uk_runtime.calibration_run import ( @@ -162,6 +166,7 @@ from microcosm.graph import ContentStore, compile_graph, run_graph _PIPELINE = "uk-frs-spine" +_ACTIVE_EMITTER: LocalTelemetryEmitter | None = None _REPOSITORY = next( ( parent @@ -1011,13 +1016,18 @@ def checkpoint_metadata(self) -> dict[str, object]: return dict(hook()) -def _staging_stage_observer(telemetry: StagingTelemetryV2) -> StageObserver: - """Translate a shared stage observation into staging telemetry.""" +def _stage_observer( + staging_bundle: StagingRunBundleWriterV2 | None, + emitter: LocalTelemetryEmitter | None, +) -> StageObserver: + """Report a stage independently to the emitter and staging bundle.""" def observe(observation: StageObservation) -> None: - telemetry.stage( + _record_stage( + staging_bundle, + emitter, observation.stage_id, - event_status=observation.status, + observation.status, elapsed_seconds=observation.elapsed_seconds, entity_row_counts=dict(observation.entity_row_counts), produced_column_count=observation.produced_column_count, @@ -1026,6 +1036,49 @@ def observe(observation: StageObservation) -> None: return observe +def _record_stage( + staging_bundle: StagingRunBundleWriterV2 | None, + emitter: LocalTelemetryEmitter | None, + stage_id: str, + status: str, + **details: object, +) -> None: + """Send one stage update to two independent destinations.""" + + if emitter is not None: + emitter.transition_stage(stage_id, status=status, **details) + if staging_bundle is not None: + staging_bundle.stage(stage_id, event_status=status, **details) + + +def _graph_progress_observer(emitter: LocalTelemetryEmitter | None, *, total: int): + """Report graph-node completion without copying the observed population.""" + + completed = 0 + previous = time.monotonic() + + def observe(node_id, _population) -> None: + nonlocal completed, previous + if emitter is None: + return + now = time.monotonic() + completed += 1 + emitter.emit( + event_type="progress", + stage_id=node_id, + status="completed", + details={ + "done": completed, + "total": total, + "unit": "graph_nodes", + "elapsed_seconds": now - previous, + }, + ) + previous = now + + return observe + + class _GraphSourceTransform: """Build a file-reading stage from only the node's declared source paths.""" @@ -1202,15 +1255,18 @@ def _exception_chain_contains(error: BaseException, text: str) -> bool: return False -def _create_staging_telemetry( - args: argparse.Namespace, *, state: AttemptState -) -> StagingTelemetryV2 | None: +def _create_staging_run_bundle( + args: argparse.Namespace, + *, + state: AttemptState, +) -> StagingRunBundleWriterV2 | None: + run_id = args.staging_run_id or state.build_id if args.no_staging: return None local_dir = args.staging_dir or args.spine_h5.parent / "staging" local_only = args.staging_local_only - return StagingTelemetryV2( - run_id=args.staging_run_id or state.build_id, + return StagingRunBundleWriterV2( + run_id=run_id, country_code="GB", operation_id="uk_frs_spine", pipeline_id=_PIPELINE, @@ -1224,6 +1280,29 @@ def _create_staging_telemetry( ) +def _start_telemetry_emitter( + args: argparse.Namespace, *, state: AttemptState +) -> LocalTelemetryEmitter: + global _ACTIVE_EMITTER + run_id = args.staging_run_id or state.build_id + if ( + _ACTIVE_EMITTER is not None + and _ACTIVE_EMITTER.available + and _ACTIVE_EMITTER.run.run_id == run_id + ): + return _ACTIVE_EMITTER + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.close() + _ACTIVE_EMITTER = start_local_telemetry_emitter_service( + run_id=run_id, + country_code="GB", + pipeline=_PIPELINE, + candidate_id=args.staging_candidate_id or state.build_id, + run_kind="smoke" if args.smoke else "spine", + ) + return _ACTIVE_EMITTER + + def _telemetry_sample( args: argparse.Namespace, sampling: Mapping[str, object] | None ) -> dict[str, object] | None: @@ -1233,11 +1312,11 @@ def _telemetry_sample( def _staging_delivery( - args: argparse.Namespace, telemetry: StagingTelemetryV2 | None + args: argparse.Namespace, staging_bundle: StagingRunBundleWriterV2 | None ) -> dict[str, object]: - if telemetry is None: + if staging_bundle is None: return disabled_staging_delivery("--no-staging") - return telemetry.delivery_summary + return staging_bundle.delivery_summary @dataclass(frozen=True) @@ -1275,7 +1354,7 @@ def prepare_uk_spine_execution( roster, the private-input checks, the stage implementations (licensed or synthetic fixture), the sampling root, the optional observation wrap, the declared graph sources, and the gate nodes bound to the rules engine. - ``observer`` is the staging telemetry's stage observer when telemetry is + ``observer`` is the staging run bundle's stage observer when that bundle is enabled; ``main`` supplies it, and a caller without telemetry passes none. """ @@ -1666,12 +1745,11 @@ def main(argv: list[str] | None = None) -> int: rung = UK_SAMPLE_RUNG_TOKENS[args.sample_fraction] started_at = time.perf_counter() started_ts = datetime.now(UTC) - predecessor = resolve_predecessor(args.logbook_prev_row_digest) - digest = preflight_digest(_PIPELINE) + predecessor = None state = AttemptState( build_id=_new_build_id(started_ts), - identity_digest=digest, - input_pins_digest=digest, + identity_digest="unresolved-preflight-digest", + input_pins_digest="unresolved-preflight-digest", phases_reached=["attempt_started"], gate_verdicts={ "pipeline": { @@ -1682,9 +1760,14 @@ def main(argv: list[str] | None = None) -> int: ) code_pin = "unresolved-local-git-code-pin" spool_dir = args.spine_h5.parent / "logbook-spool" - telemetry: StagingTelemetryV2 | None = None + staging_bundle: StagingRunBundleWriterV2 | None = None spine_battery: GateBatteryRun | None = None + emitter = _start_telemetry_emitter(args, state=state) try: + predecessor = resolve_predecessor(args.logbook_prev_row_digest) + digest = preflight_digest(_PIPELINE) + state.identity_digest = digest + state.input_pins_digest = digest _validate_args(args) # A crash between the H5 write and the sidecar writes must never # leave a stale sidecar beside a fresh H5 (adversarial-review @@ -1702,24 +1785,24 @@ def main(argv: list[str] | None = None) -> int: stale_outputs.append(args.emit_nonzero_shares) for stale in stale_outputs: stale.unlink(missing_ok=True) - telemetry = _create_staging_telemetry(args, state=state) - if telemetry is not None: + staging_bundle = _create_staging_run_bundle(args, state=state) + if staging_bundle is not None: initial_sample = _telemetry_sample(args, None) if initial_sample is not None: - telemetry.set_sample(initial_sample) - telemetry.stage( - "configuration", - event_status="completed", - smoke=args.smoke, - sample_mode=("fraction" if args.sample_fraction != 1.0 else "full"), - ) + staging_bundle.set_sample(initial_sample) + _record_stage( + staging_bundle, + emitter, + "configuration", + "completed", + smoke=args.smoke, + sample_mode=("fraction" if args.sample_fraction != 1.0 else "full"), + ) code_pin = git_code_pin(_REPOSITORY) append_phase(state, "configured") prepared = prepare_uk_spine_execution( args, - observer=( - _staging_stage_observer(telemetry) if telemetry is not None else None - ), + observer=_stage_observer(staging_bundle, emitter), ) spec = prepared.spec graph = prepared.graph @@ -1749,13 +1832,14 @@ def main(argv: list[str] | None = None) -> int: canonical_json_bytes(run_config) ).hexdigest() append_phase(state, "inputs_pinned") - if telemetry is not None: - telemetry.stage( - "input_verification", - event_status="completed", - stage_count=len(stage_names), - input_artifact_count=len(artifact_pins) + len(input_artifact_pins), - ) + _record_stage( + staging_bundle, + emitter, + "input_verification", + "completed", + stage_count=len(stage_names), + input_artifact_count=len(artifact_pins) + len(input_artifact_pins), + ) stochastic_contract = prepared.stochastic_contract frs_release = prepared.frs_release spine_gate_path = _spine_gate_report_path(args.spine_h5) @@ -1786,6 +1870,10 @@ def main(argv: list[str] | None = None) -> int: kernels=prepared.kernels, resume="auto", decisions=(), + _population_observer=_graph_progress_observer( + _ACTIVE_EMITTER, total=len(compiled_graph.order) + ), + _population_observer_detach=False, ) graph_manifest.save(checkpoint_root / "spine.graph.json") final_version = compiled_graph.versions[stage_names[-1]] @@ -1801,23 +1889,24 @@ def main(argv: list[str] | None = None) -> int: ) stored_evidence = spine_sidecar_evidence(stored_stages) sampling = stored_evidence["sampling"] - if telemetry is not None: - sample = _telemetry_sample(args, sampling) - if sample is not None: - telemetry.set_sample(sample) - telemetry.stage( - "sampling", - event_status="completed", - realized_household_rows=( - len(frame.table("household")) - if sampling is None - else sampling.get( - "realized_household_rows", - sampling.get("post_household_count"), - ) - ), - ) - telemetry.stage("validation", event_status="started") + sample = _telemetry_sample(args, sampling) + if staging_bundle is not None and sample is not None: + staging_bundle.set_sample(sample) + _record_stage( + staging_bundle, + emitter, + "sampling", + "completed", + realized_household_rows=( + len(frame.table("household")) + if sampling is None + else sampling.get( + "realized_household_rows", + sampling.get("post_household_count"), + ) + ), + ) + _record_stage(staging_bundle, emitter, "validation", "started") if spine_battery is not None: materialize_spine_gate_reports( graph_manifest, @@ -1827,24 +1916,25 @@ def main(argv: list[str] | None = None) -> int: ) if spine_battery is not None: append_phase(state, "spine_gates_evaluated") - if telemetry is not None: - telemetry.stage( - "validation", - event_status="completed", - entity_row_counts=_entity_row_counts(frame), - ) + _record_stage( + staging_bundle, + emitter, + "validation", + "completed", + entity_row_counts=_entity_row_counts(frame), + ) append_phase(state, "spine_built") - if telemetry is not None: - telemetry.stage("spine_h5_creation", event_status="started") + _record_stage(staging_bundle, emitter, "spine_h5_creation", "started") output = write_uk_national_frame(frame, args.spine_h5) if args.smoke: _mark_non_release_h5(output, build_id=state.build_id) - if telemetry is not None: - telemetry.stage( - "spine_h5_creation", - event_status="completed", - size_bytes=output.stat().st_size, - ) + _record_stage( + staging_bundle, + emitter, + "spine_h5_creation", + "completed", + size_bytes=output.stat().st_size, + ) append_phase(state, "spine_written") if args.checkpoint_dir is not None: args.checkpoint_dir.mkdir(parents=True, exist_ok=True) @@ -1866,8 +1956,7 @@ def main(argv: list[str] | None = None) -> int: "report_kind": str(json.loads(replay_bytes).get("report_kind", "")), "sha256": hashlib.sha256(replay_bytes).hexdigest(), } - if telemetry is not None: - telemetry.stage("sidecar_creation", event_status="started") + _record_stage(staging_bundle, emitter, "sidecar_creation", "started") sidecar = _build_sidecar( frame=frame, stages=stages, @@ -1888,7 +1977,7 @@ def main(argv: list[str] | None = None) -> int: else "development" ), synthetic_fixture=synthetic_fixture, - staging_delivery=_staging_delivery(args, telemetry), + staging_delivery=_staging_delivery(args, staging_bundle), spine_gate_report=( { "path": str(spine_gate_path), @@ -1910,12 +1999,13 @@ def main(argv: list[str] | None = None) -> int: if fit_weight_records: sidecar["fit_weight_records"] = fit_weight_records atomic_write_json(sidecar_path, sidecar) - if telemetry is not None: - telemetry.stage( - "sidecar_creation", - event_status="completed", - size_bytes=sidecar_path.stat().st_size, - ) + _record_stage( + staging_bundle, + emitter, + "sidecar_creation", + "completed", + size_bytes=sidecar_path.stat().st_size, + ) append_phase(state, "build_sidecar_written") if args.emit_nonzero_shares is not None: final_columns = list( @@ -1934,8 +2024,8 @@ def main(argv: list[str] | None = None) -> int: }, ) append_phase(state, "nonzero_shares_written") - if telemetry is not None: - telemetry.complete( + if staging_bundle is not None: + staging_bundle.complete( message=( "Non-release smoke verification completed." if args.smoke @@ -1943,10 +2033,12 @@ def main(argv: list[str] | None = None) -> int: ) ) if args.staging_read_back: - telemetry.verify_remote() - telemetry.validate_local_bundle() - sidecar["staging_delivery"] = telemetry.delivery_summary + staging_bundle.verify_remote() + staging_bundle.validate_local_bundle() + sidecar["staging_delivery"] = staging_bundle.delivery_summary atomic_write_json(sidecar_path, sidecar) + if emitter.available: + emitter.complete() state.artifact_location = local_artifact_reference( output, repository_hint=_REPOSITORY, @@ -1993,12 +2085,14 @@ def main(argv: list[str] | None = None) -> int: materialize_blocked_spine_gate_report(error, battery=spine_battery) except GateBatteryBlockedError as blocked: error = blocked - if telemetry is not None and telemetry.status == "running": + if staging_bundle is not None and staging_bundle.status == "running": try: - telemetry.fail(error) - telemetry.validate_local_bundle() + staging_bundle.fail(error) + staging_bundle.validate_local_bundle() except Exception: pass + if emitter.available: + emitter.fail(error) if _is_sampled(args) and _exception_chain_contains( error, _RUNG_NAMED_EDGE_SIGNATURE ): diff --git a/packages/microcosm-build/tests/engine_free/shared/test_staging.py b/packages/microcosm-build/tests/engine_free/shared/test_staging.py index be9490272..3d6b62c03 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_staging.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_staging.py @@ -1,7 +1,8 @@ +import inspect import json import microcosm.build.staging as staging_module -from microcosm.build.staging import StagingTelemetry +from microcosm.build.staging import StagingRunBundleWriter from microcosm.build.staging_storage import BestEffortUploadSession from test_support.paths import paths_for @@ -10,6 +11,10 @@ V1_FIXTURE = _TEST_PATHS.tests / "fixtures" / "staging" / "v1" +def test_staging_run_bundle_writer_has_no_emitter_dependency() -> None: + assert "emitter" not in inspect.signature(StagingRunBundleWriter).parameters + + class FakeApi: def __init__(self) -> None: self.uploads: list[tuple[str, str, str, str]] = [] @@ -26,8 +31,8 @@ def upload_file( return None -def test_staging_telemetry_writes_local_progress(tmp_path) -> None: - telemetry = StagingTelemetry( +def test_staging_run_bundle_writer_writes_local_progress(tmp_path) -> None: + telemetry = StagingRunBundleWriter( run_id="run-a", candidate_release_id="populace-us-2024-abc-20260618T000000Z", run_dir=tmp_path / "run-a", @@ -51,9 +56,9 @@ def test_staging_telemetry_writes_local_progress(tmp_path) -> None: assert len(events) >= 3 -def test_staging_telemetry_uploads_repo_paths(tmp_path) -> None: +def test_staging_run_bundle_writer_uploads_repo_paths(tmp_path) -> None: api = FakeApi() - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-b", candidate_release_id="populace-us-2024-def-20260618T000000Z", run_dir=tmp_path / "run-b", @@ -96,7 +101,7 @@ class FakeApiWithDownload(FakeApi): def hf_hub_download(self, *, repo_id, filename, repo_type, **kwargs): return str(existing) - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="newer", candidate_release_id="populace-us-2024-new-20260620T000000Z", run_dir=tmp_path / "newer", @@ -120,7 +125,7 @@ def upload_file(self, **kwargs): def hf_hub_download(self, **kwargs): raise RuntimeError("401 Unauthorized") - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-1", candidate_release_id="run-1", run_dir=tmp_path / "run", @@ -143,7 +148,7 @@ def hf_hub_download(self, **kwargs): def test_uploads_succeeded_counts_files_that_reached_the_repo(tmp_path): api = FakeApi() - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-c", candidate_release_id="run-c", run_dir=tmp_path / "run-c", @@ -162,7 +167,7 @@ def test_blank_path_prefix_falls_back_to_the_default(tmp_path): # A blank or slash-only prefix would put run files at the repo root, where # the dashboard's runs/ paths cannot find them. for blank in ("", " ", "/", " / "): - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-e", candidate_release_id="run-e", run_dir=tmp_path / "run-e", @@ -173,7 +178,7 @@ def test_blank_path_prefix_falls_back_to_the_default(tmp_path): def test_path_prefix_is_trimmed_but_otherwise_respected(tmp_path): - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-f", candidate_release_id="run-f", run_dir=tmp_path / "run-f", @@ -184,7 +189,7 @@ def test_path_prefix_is_trimmed_but_otherwise_respected(tmp_path): def test_uploads_succeeded_is_zero_for_a_local_only_run(tmp_path): - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-d", candidate_release_id="run-d", run_dir=tmp_path / "run-d", @@ -215,7 +220,7 @@ def hf_hub_download(self, **kwargs): raise FileNotFoundError run_dir = tmp_path / "v1-us-fixture" - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="v1-us-fixture", candidate_release_id="populace-us-2024-v1-fixture", run_dir=run_dir, diff --git a/packages/microcosm-build/tests/engine_free/shared/test_staging_v2.py b/packages/microcosm-build/tests/engine_free/shared/test_staging_v2.py index a4030f8ad..357c08acc 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_staging_v2.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_staging_v2.py @@ -1,5 +1,6 @@ import argparse import hashlib +import inspect import json import subprocess import sys @@ -26,7 +27,7 @@ StagingContentError, StagingContractError, StagingReadBackError, - StagingTelemetryV2, + StagingRunBundleWriterV2, disabled_staging_delivery, validate_staging_delivery, validate_v2_bundle, @@ -80,7 +81,7 @@ def _sample() -> dict: return {"mode": "full"} -def _recorder(tmp_path, **kwargs) -> StagingTelemetryV2: +def _recorder(tmp_path, **kwargs) -> StagingRunBundleWriterV2: defaults = { "run_id": "uk-smoke-5-42", "country_code": "GB", @@ -96,7 +97,11 @@ def _recorder(tmp_path, **kwargs) -> StagingTelemetryV2: "clock": Clock(), } defaults.update(kwargs) - return StagingTelemetryV2(**defaults) + return StagingRunBundleWriterV2(**defaults) + + +def test_staging_run_bundle_writer_has_no_emitter_dependency() -> None: + assert "emitter" not in inspect.signature(StagingRunBundleWriterV2).parameters def test_version_2_local_bundle_has_explicit_schemas_and_ordered_events(tmp_path): diff --git a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py new file mode 100644 index 000000000..bb85c366c --- /dev/null +++ b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py @@ -0,0 +1,686 @@ +import json +import os +import socket +import tempfile +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from types import SimpleNamespace + +from alembic import command +from sqlalchemy import Column, Integer, MetaData, String, Table, create_engine +from sqlalchemy.orm import Session + +from microcosm.build.telemetry_emitter import ( + LocalTelemetryEmitter, + TelemetryRun, +) +from microcosm.build.telemetry_emitter_service import ( + CollectorDelivery, + EmitterService, + EventSpool, +) +from microcosm.build.telemetry_emitter_service import collector as collector_module +from microcosm.build.telemetry_emitter_service import resources as resources_module +from microcosm.build.telemetry_emitter_service import spool as spool_module +from microcosm.build.telemetry_emitter_service.constants import ( + PRODUCTION_COLLECTOR_URL, +) +from microcosm.build.telemetry_emitter_service.database import ( + create_spool_engine, + sqlite_database_url, +) +from microcosm.build.telemetry_emitter_service.migrations import ( + alembic_config, + current_database_revision, + migration_head_revision, +) +from microcosm.build.telemetry_emitter_service.models import ( + SpoolModel, + TelemetryEventRecord, + TelemetryRunRecord, + serialized_json_length, +) +from microcosm.build.telemetry_protocol import ( + BUILD_COMPLETED_MESSAGE, + BUILD_STARTED_MESSAGE, + MAX_TELEMETRY_DETAILS_BYTES, +) +from microcosm.build.telemetry_sanitization import ( + sanitize_details, + sanitize_json, + sanitize_text, +) + + +def _registration(run_id: str = "run-a") -> dict[str, object]: + return TelemetryRun( + run_id=run_id, + country_code="US", + pipeline="us_fiscal_refresh", + candidate_id="candidate-a", + producer_id="producer-a", + ).as_registration() + + +def _event(stage_id: str = "compile_targets") -> dict[str, object]: + return { + "timestamp": "2026-10-02T10:00:00+00:00", + "event_type": "stage", + "stage_id": stage_id, + "status": "started", + "message": "Compiling targets.", + "details": {"batches": 12}, + } + + +def test_development_collector_must_be_on_loopback() -> None: + assert collector_module._development_collector_url("http://127.0.0.1:8080") == ( + "http://127.0.0.1:8080" + ) + for value in ("https://collector.example", "http://192.0.2.1:8080"): + try: + collector_module._development_collector_url(value) + except ValueError as error: + assert "loopback" in str(error) + else: + raise AssertionError("non-loopback development collector was accepted") + + +def test_production_collector_cannot_be_replaced_by_environment( + tmp_path, monkeypatch +) -> None: + monkeypatch.setenv( + "MICROCOSM_TELEMETRY_COLLECTOR_URL", + "https://untrusted.example", + ) + + delivery = CollectorDelivery(EventSpool(tmp_path / "events.sqlite3")) + + assert delivery.collector_url == PRODUCTION_COLLECTOR_URL + + +def test_token_bearing_http_post_does_not_follow_redirects() -> None: + paths: list[str] = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + paths.append(self.path) + self.send_response(307) + self.send_header("Location", "/credential-leak") + self.send_header("Content-Length", "0") + self.end_headers() + + def log_message(self, format, *args): + return + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + server_thread = threading.Thread(target=server.serve_forever) + server_thread.start() + try: + status, _ = collector_module._http_post( + f"http://127.0.0.1:{server.server_port}/exchange", + {"run_id": "run-a"}, + "hf-private-token", + ) + finally: + server.shutdown() + server_thread.join(timeout=2) + server.server_close() + + assert status == 307 + assert paths == ["/exchange"] + + +def test_outbound_payload_redacts_credentials_and_tracebacks() -> None: + assert sanitize_text("Bearer hf_abcdefghijk") == "[redacted]" + assert sanitize_json( + { + "HF_TOKEN": "hf_abcdefghijk", + "traceback": "private stack", + "message": "credential=top-secret", + } + ) == { + "HF_TOKEN": "[redacted]", + "traceback": "[redacted]", + "message": "[redacted]", + } + bounded = sanitize_details( + {"done": 2, **{f"large_{index}": "x" * 2_000 for index in range(10)}} + ) + assert bounded["done"] == 2 + assert bounded["telemetry_details_truncated"] is True + assert len(json.dumps(bounded).encode()) <= MAX_TELEMETRY_DETAILS_BYTES + + +def test_event_spool_assigns_stable_sequences_and_acknowledges(tmp_path) -> None: + spool_path = tmp_path / "events.sqlite3" + spool = EventSpool(spool_path) + registration = _registration() + spool.register(registration) + + first = spool.append(registration, _event()) + second = spool.append(registration, _event("calibrate")) + + assert first["sequence"] == 1 + assert second["sequence"] == 2 + assert first["producer_id"] == "producer-a" + assert [event["sequence"] for event in spool.batch("run-a", "producer-a")] == [ + 1, + 2, + ] + + spool.acknowledge([first["event_id"]]) + assert [event["sequence"] for event in spool.batch("run-a", "producer-a")] == [2] + assert current_database_revision(spool_path) == migration_head_revision() + + +def test_alembic_schema_matches_sqlalchemy_models(tmp_path) -> None: + spool_path = tmp_path / "events.sqlite3" + EventSpool(spool_path) + engine = create_spool_engine(spool_path) + try: + with engine.begin() as connection: + with alembic_config(connection=connection) as config: + command.check(config) + finally: + engine.dispose() + + +def test_sequential_stage_updates_close_the_previous_stage() -> None: + emitter = LocalTelemetryEmitter( + run=TelemetryRun( + run_id="run-a", + country_code="US", + pipeline="us_fiscal_refresh", + ), + process=None, + socket_path=Path("/unused"), + runtime_dir=None, + ) + messages: list[dict[str, object]] = [] + emitter._send = messages.append + + emitter.transition_stage("load", message="Loading.") + emitter.transition_stage("compile", message="Compiling.") + emitter.transition_stage("compile", status="completed", batches=4) + + events = [message["event"] for message in messages] + assert [(event["stage_id"], event["status"]) for event in events] == [ + ("load", "started"), + ("load", "completed"), + ("compile", "started"), + ("compile", "completed"), + ] + + +def test_event_spool_keeps_repeated_run_producers_separate(tmp_path) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + first = _registration() + second = {**first, "producer_id": "producer-b"} + spool.register(first) + spool.register(second) + + spool.append(first, _event()) + spool.append(second, _event()) + + assert spool.batch("run-a", "producer-a")[0]["sequence"] == 1 + assert spool.batch("run-a", "producer-b")[0]["sequence"] == 1 + assert len(spool.pending_runs()) == 2 + + +def test_pre_eligibility_spool_is_not_uploaded_after_upgrade(tmp_path) -> None: + path = tmp_path / "events.sqlite3" + registration = _registration() + metadata = MetaData() + runs = Table( + "telemetry_runs", + metadata, + Column("run_id", String, primary_key=True), + Column("producer_id", String, primary_key=True), + Column("registration_json", String, nullable=False), + Column("next_sequence", Integer, nullable=False, default=1), + Column("updated_at", String, nullable=False), + ) + events = Table( + "telemetry_events", + metadata, + Column("event_id", String, primary_key=True), + Column("run_id", String, nullable=False), + Column("producer_id", String, nullable=False), + Column("sequence", Integer, nullable=False), + Column("payload_json", String, nullable=False), + Column("created_at", String, nullable=False), + ) + engine = create_engine(sqlite_database_url(path)) + metadata.create_all(engine) + with engine.begin() as connection: + connection.execute( + runs.insert().values( + run_id="run-a", + producer_id="producer-a", + registration_json=json.dumps(registration), + next_sequence=2, + updated_at="2026-10-02T10:00:00+00:00", + ) + ) + connection.execute( + events.insert().values( + event_id="old-event", + run_id="run-a", + producer_id="producer-a", + sequence=1, + payload_json=json.dumps({"event_id": "old-event"}), + created_at="2026-10-02T10:00:00+00:00", + ) + ) + engine.dispose() + + spool = EventSpool(path) + + assert spool.has_pending() + assert spool.pending_runs() == [] + assert current_database_revision(path) == migration_head_revision() + + +def test_current_pre_alembic_spool_is_adopted_without_losing_events(tmp_path) -> None: + path = tmp_path / "events.sqlite3" + registration = _registration() + engine = create_engine(sqlite_database_url(path)) + SpoolModel.metadata.create_all(engine) + with Session(engine) as session, session.begin(): + session.add( + TelemetryRunRecord( + run_id="run-a", + producer_id="producer-a", + registration=registration, + next_sequence=2, + upload_state="pending", + local_only_reason=None, + updated_at="2026-10-02T10:00:00+00:00", + ) + ) + session.add( + TelemetryEventRecord( + event_id="existing-event", + run_id="run-a", + producer_id="producer-a", + sequence=1, + payload={"event_id": "existing-event", "sequence": 1}, + created_at="2026-10-02T10:00:00+00:00", + ) + ) + engine.dispose() + + spool = EventSpool(path) + + assert spool.pending_runs() == [registration] + assert spool.batch("run-a", "producer-a") == [ + {"event_id": "existing-event", "sequence": 1} + ] + assert current_database_revision(path) == migration_head_revision() + + +def test_event_spool_prunes_oldest_events_to_size_limit( + tmp_path, + monkeypatch, +) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + spool.register(registration) + first = spool.append(registration, _event("first")) + second = spool.append(registration, _event("second")) + monkeypatch.setattr( + spool_module, + "MAX_QUEUED_BYTES", + serialized_json_length(second), + ) + + spool.prune() + + assert first["event_id"] != second["event_id"] + assert spool.batch("run-a", "producer-a") == [second] + + +def test_collector_delivery_exchanges_hf_token_then_flushes( + tmp_path, monkeypatch +) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + spool.register(registration) + queued = spool.append(registration, _event()) + requests: list[tuple[str, dict[str, object], str]] = [] + + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: "hf-secret") + + def fake_post(url, payload, token, *, timeout=5.0): + requests.append((url, payload, token)) + if url.endswith("/v1/auth/huggingface/exchange"): + return 200, {"access_token": "collector-token", "expires_in": 3600} + if url.endswith("/v1/runs"): + return 201, {"registered": True} + return 202, {"accepted": 1, "duplicates": 0} + + monkeypatch.setattr(collector_module, "_http_post", fake_post) + + delivery = CollectorDelivery( + spool, + development_collector_url="http://127.0.0.1:8080", + ) + assert delivery.flush_once() + assert requests[0][2] == "hf-secret" + assert requests[0][1] == {} + assert requests[1] == ( + "http://127.0.0.1:8080/v1/runs", + registration, + "collector-token", + ) + assert requests[2][2] == "collector-token" + assert requests[2][1] == {"events": [queued]} + assert not spool.has_pending() + + +def test_non_org_credential_keeps_event_local(tmp_path, monkeypatch, capsys) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + spool.register(registration) + spool.append(registration, _event()) + requests: list[dict[str, object]] = [] + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: "hf-outsider") + + def reject(url, payload, token, *, timeout=5.0): + requests.append(payload) + return 403, {"detail": "not a member"} + + monkeypatch.setattr(collector_module, "_http_post", reject) + delivery = CollectorDelivery( + spool, + development_collector_url="http://127.0.0.1:8080", + ) + + assert not delivery.flush_once() + assert not delivery.flush_once() + assert spool.has_pending() + assert spool.pending_runs() == [] + assert requests == [{}] + assert capsys.readouterr().err.count("local-only for this run") == 1 + + +def test_identity_provider_outage_keeps_events_eligible_for_retry( + tmp_path, monkeypatch +) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + spool.register(registration) + spool.append(registration, _event()) + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: "hf-member") + monkeypatch.setattr( + collector_module, + "_http_post", + lambda *args, **kwargs: (503, {"detail": "temporarily unavailable"}), + ) + + assert not CollectorDelivery(spool).flush_once() + assert spool.pending_runs() == [registration] + + +def test_missing_credential_never_contacts_collector_and_stays_local_only( + tmp_path, monkeypatch +) -> None: + spool_path = tmp_path / "events.sqlite3" + spool = EventSpool(spool_path) + registration = _registration() + spool.register(registration) + spool.append(registration, _event()) + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: None) + monkeypatch.setattr( + collector_module, + "_http_post", + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("network request") + ), + ) + + delivery = CollectorDelivery(spool) + assert not delivery.flush_once() + assert spool.pending_runs() == [] + + reopened = EventSpool(spool_path) + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: "hf-later") + assert reopened.pending_runs() == [] + + +def test_resource_sampler_has_a_base_install_fallback(monkeypatch) -> None: + monkeypatch.setattr(resources_module, "psutil", None) + sampler = resources_module.ProcessTreeSampler(os.getpid()) + + assert sampler.parent_alive() + sample = sampler.sample() + assert sample.keys() == { + "cpu_user_seconds", + "cpu_system_seconds", + "rss_bytes", + "peak_rss_bytes", + } + assert all(value >= 0 for value in sample.values()) + + +def test_resource_sampler_keeps_reaped_child_cpu_monotonic(monkeypatch) -> None: + state = {"child_alive": True} + + class FakeProcess: + def __init__(self, pid): + self.pid = pid + + def create_time(self): + return float(self.pid) + + def children(self, recursive=False): + assert recursive is True + if self.pid == 10 and state["child_alive"]: + return [FakeProcess(11)] + return [] + + def cpu_times(self): + if self.pid == 10: + return SimpleNamespace( + user=10.0, + system=2.0, + children_user=100.0 if not state["child_alive"] else 0.0, + children_system=20.0 if not state["child_alive"] else 0.0, + ) + return SimpleNamespace( + user=80.0, + system=15.0, + children_user=0.0, + children_system=0.0, + ) + + def memory_info(self): + return SimpleNamespace(rss=100) + + monkeypatch.setattr( + resources_module, + "psutil", + SimpleNamespace(Process=FakeProcess, Error=OSError, STATUS_ZOMBIE="zombie"), + ) + sampler = resources_module.ProcessTreeSampler(10) + + before = sampler.sample() + state["child_alive"] = False + after = sampler.sample() + + assert before["cpu_user_seconds"] == 90.0 + assert before["cpu_system_seconds"] == 17.0 + assert after["cpu_user_seconds"] == 110.0 + assert after["cpu_system_seconds"] == 22.0 + + +def test_emitter_startup_timeout_terminates_and_reaps_service( + tmp_path, monkeypatch +) -> None: + class FakeProcess: + def __init__(self): + self.terminated = False + self.waited = False + + def poll(self): + return None + + def terminate(self): + self.terminated = True + + def wait(self, timeout=None): + self.waited = True + return 0 + + process = FakeProcess() + monkeypatch.setattr( + "microcosm.build.telemetry_emitter.subprocess.Popen", + lambda *args, **kwargs: process, + ) + + emitter = LocalTelemetryEmitter.start( + run_id="startup-timeout", + country_code="US", + pipeline="test-pipeline", + development_collector_url="http://127.0.0.1:8080", + spool_path=tmp_path / "events.sqlite3", + startup_timeout_seconds=0, + ) + + assert not emitter.available + assert process.terminated + assert process.waited + + +class _FakeSampler: + def sample(self): + return { + "cpu_user_seconds": 1.0, + "cpu_system_seconds": 0.5, + "rss_bytes": 100, + "peak_rss_bytes": 120, + } + + def parent_alive(self): + return True + + +class _FakeDelivery: + def flush_once(self): + return False + + +def test_local_socket_acknowledges_after_durable_queue(tmp_path) -> None: + socket_path = ( + Path(tempfile.mkdtemp(prefix="microcosm-test-", dir="/tmp")) / "e.sock" + ) + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + emitter = EmitterService( + socket_path=socket_path, + registration=registration, + spool=spool, + delivery=_FakeDelivery(), + sampler=_FakeSampler(), + heartbeat_seconds=60, + drain_seconds=0, + ) + thread = threading.Thread(target=emitter.run) + thread.start() + deadline = time.monotonic() + 2 + while not socket_path.exists() and time.monotonic() < deadline: + time.sleep(0.01) + + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as client: + client.connect(str(socket_path)) + client.sendall( + json.dumps({"action": "event", "event": _event()}).encode() + b"\n" + ) + assert client.recv(16) == b"ok\n" + + assert spool.batch("run-a", "producer-a")[0]["resources"]["rss_bytes"] == 100 + + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as client: + client.connect(str(socket_path)) + client.sendall(b'{"action":"close"}\n') + assert client.recv(16) == b"ok\n" + thread.join(timeout=2) + assert not thread.is_alive() + + +def test_subprocess_exchanges_ambient_token_and_delivers_events( + tmp_path, monkeypatch +) -> None: + requests: list[tuple[str, str, dict[str, object]]] = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + length = int(self.headers.get("Content-Length", "0")) + payload = json.loads(self.rfile.read(length)) + requests.append((self.path, self.headers["Authorization"], payload)) + if self.path.endswith("/v1/auth/huggingface/exchange"): + response = {"access_token": "collector-token", "expires_in": 3600} + status = 200 + elif self.path == "/v1/runs": + response = {"registered": True} + status = 201 + else: + response = { + "accepted": len(payload["events"]), + "duplicates": 0, + } + status = 202 + body = json.dumps(response).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def log_message(self, format, *args): + return + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + server_thread = threading.Thread(target=server.serve_forever) + server_thread.start() + monkeypatch.setenv("HF_TOKEN", "hf-ambient-test-token") + emitter = LocalTelemetryEmitter.start( + run_id="subprocess-run", + country_code="US", + pipeline="test-pipeline", + development_collector_url=f"http://127.0.0.1:{server.server_port}", + spool_path=tmp_path / "subprocess.sqlite3", + heartbeat_seconds=60, + ) + assert emitter.available + emitter.stage("compile", message="Started.") + emitter.complete() + assert emitter._process is not None + emitter._process.wait(timeout=10) + server.shutdown() + server_thread.join(timeout=2) + server.server_close() + + assert requests[0][0] == "/v1/auth/huggingface/exchange" + assert requests[0][1] == "Bearer hf-ambient-test-token" + assert requests[0][2] == {} + assert requests[1][0] == "/v1/runs" + assert requests[1][1] == "Bearer collector-token" + ingestion = [request for request in requests if request[0].endswith("/events")] + assert ingestion + events = [ + event + for _, authorization, payload in ingestion + for event in payload["events"] + if authorization == "Bearer collector-token" + ] + assert [event["sequence"] for event in events] == list(range(1, len(events) + 1)) + identity = events[0]["details"]["identity"] + assert identity["host"]["cpu_count"] is not None + assert "runtime" in identity + assert events[0]["message"] == BUILD_STARTED_MESSAGE + assert events[-1]["message"] == BUILD_COMPLETED_MESSAGE + assert events[-1]["status"] == "completed" diff --git a/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py b/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py index ce94fbd8d..97c21f4dd 100644 --- a/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py +++ b/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py @@ -108,15 +108,20 @@ def _stored_fit_weight_records( } -def test_staging_stage_observer_translates_shared_observation() -> None: +def test_stage_observer_reports_to_emitter_and_staging_bundle_independently() -> None: tool = _load_tool() - calls = [] + bundle_calls = [] + emitter_calls = [] - class RecordingTelemetry: + class RecordingBundle: def stage(self, stage_id: str, **payload: object) -> None: - calls.append((stage_id, payload)) + bundle_calls.append((stage_id, payload)) - observer = tool._staging_stage_observer(RecordingTelemetry()) + class RecordingEmitter: + def transition_stage(self, stage_id: str, **payload: object) -> None: + emitter_calls.append((stage_id, payload)) + + observer = tool._stage_observer(RecordingBundle(), RecordingEmitter()) observer( StageObservation( stage_id="frs_spine", @@ -127,17 +132,21 @@ def stage(self, stage_id: str, **payload: object) -> None: ) ) - assert calls == [ + details = { + "elapsed_seconds": 1.25, + "entity_row_counts": {"household": 2}, + "produced_column_count": 4, + } + assert bundle_calls == [ ( "frs_spine", { "event_status": "completed", - "elapsed_seconds": 1.25, - "entity_row_counts": {"household": 2}, - "produced_column_count": 4, + **details, }, ) ] + assert emitter_calls == [("frs_spine", {"status": "completed", **details})] def _write_tab(root: Path, table: str, rows: list[dict[str, object]]) -> None: @@ -1588,6 +1597,56 @@ def test_driver_refuses_missing_spi_tab(tmp_path: Path) -> None: tool._validate_args(args) +def test_driver_reports_validation_failure_through_early_emitter( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + tool = _load_tool() + raw_dir = tmp_path / "raw" + raw_dir.mkdir() + hmrc_ods = tmp_path / "Collated_Tables_3_1_to_3_11_2324.ods" + hmrc_ods.write_text("synthetic\n", encoding="utf-8") + events = [] + + class FakeEmitter: + available = True + + def __init__(self, run_id: str) -> None: + self.run = SimpleNamespace(run_id=run_id) + + def close(self) -> None: + events.append(("close",)) + + def fail(self, error: BaseException) -> None: + events.append(("failed", str(error))) + + def start_emitter(**run): + events.append(("started", run["run_id"])) + return FakeEmitter(run["run_id"]) + + monkeypatch.setattr(tool, "start_local_telemetry_emitter_service", start_emitter) + monkeypatch.setattr(tool, "preflight_digest", lambda pipeline: "0" * 64) + monkeypatch.delenv("POPULACE_LOGBOOK_PREV_ROW_DIGEST", raising=False) + + status = tool.main( + [ + "--frs-raw-dir", + str(raw_dir), + "--spine-h5", + str(tmp_path / "spine.h5"), + "--spi-tab", + str(tmp_path / "missing-put2223uk.tab"), + "--hmrc-ods", + str(hmrc_ods), + "--no-staging", + ] + ) + + assert status == 1 + assert events[0][0] == "started" + assert events[1][0] == "failed" + assert "--spi-tab must be an existing file" in events[1][1] + + def test_driver_refuses_misnamed_spi_tab(tmp_path: Path) -> None: tool = _load_tool() raw_dir = tmp_path / "raw" diff --git a/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py b/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py index 96a19b1e0..de62bb5be 100644 --- a/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py +++ b/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py @@ -500,6 +500,112 @@ def test_dry_run_has_no_files_or_kernel_execution(tmp_path, monkeypatch, capsys) assert not args.out.exists() +def test_dense_dry_run_starts_and_finishes_hosted_telemetry(tmp_path, monkeypatch): + args = arguments(tmp_path, "--dry-run") + events = [] + + class FakeEmitter: + def transition_stage(self, stage_id, **details): + events.append(("stage", stage_id, details)) + + monkeypatch.setattr(cli, "parse_args", lambda argv: args) + monkeypatch.setattr( + cli, + "start_telemetry_emitter", + lambda requested, *, build_id, run_kind: ( + events.append(("start", build_id, run_kind)) or FakeEmitter() + ), + ) + monkeypatch.setattr( + cli, "_dry_run", lambda requested: events.append(("plan",)) or 0 + ) + monkeypatch.setattr( + cli, + "finalize_staging_run_bundle", + lambda requested, telemetry: events.append(("complete", telemetry)), + ) + + assert cli.main([]) == 0 + assert [event[0] for event in events] == ["start", "stage", "plan", "complete"] + assert events[0][2] == "dry_run" + + +def test_dense_preflight_failure_is_reported_by_early_emitter(tmp_path, monkeypatch): + args = arguments(tmp_path) + events = [] + + class FakeEmitter: + def transition_stage(self, stage_id, **details): + events.append(("stage", stage_id, details)) + + monkeypatch.setattr(cli, "parse_args", lambda argv: args) + monkeypatch.setattr( + cli, + "start_telemetry_emitter", + lambda requested, *, build_id, run_kind: ( + events.append(("start", build_id, run_kind)) or FakeEmitter() + ), + ) + + def refuse_preflight(requested): + events.append(("preflight",)) + raise RuntimeError("staging credential unavailable") + + monkeypatch.setattr(cli, "preflight_staged_dataset", refuse_preflight) + monkeypatch.setattr( + cli, + "fail_staging_run_bundle", + lambda telemetry, error: events.append(("failed", telemetry, str(error))), + ) + monkeypatch.setattr( + cli, + "create_staging_run_bundle", + lambda *args, **kwargs: pytest.fail("preflight failure reached staging setup"), + ) + + with pytest.raises(RuntimeError, match="staging credential unavailable"): + cli.main([]) + assert [event[0] for event in events] == [ + "start", + "stage", + "preflight", + "failed", + ] + assert events[-1][1] is None + + +def test_hosted_stage_reporting_survives_staging_bundle_refusal(monkeypatch, capsys): + from microcosm.build.staging_v2 import StagingContentError + from microcosm.build.uk_runtime import rowwise_staging + + events = [] + + class RefusingBundle: + def stage(self, *args, **kwargs): + raise StagingContentError("synthetic staging bundle refusal") + + class Emitter: + def transition_stage(self, stage_id, **details): + events.append((stage_id, details)) + + monkeypatch.setattr(rowwise_staging, "_ACTIVE_EMITTER", Emitter()) + + rowwise_staging.stage( + RefusingBundle(), + "target_compilation", + "started", + selected_target_count=10, + ) + + assert events == [ + ( + "target_compilation", + {"status": "started", "message": None, "selected_target_count": 10}, + ) + ] + assert "synthetic staging bundle refusal" in capsys.readouterr().err + + def test_rejected_output_inside_source_never_writes_failure_sidecar( tmp_path, monkeypatch ): @@ -878,11 +984,11 @@ def test_invalid_local_telemetry_bundle_is_a_warning_not_the_runs_failure( from microcosm.build.staging_v2 import StagingContractError from microcosm.build.uk_runtime import rowwise_staging - class Invalid(rowwise_staging.StagingTelemetryV2): + class Invalid(rowwise_staging.StagingRunBundleWriterV2): def validate_local_bundle(self): raise StagingContractError("synthetic bundle defect") - monkeypatch.setattr(rowwise_staging, "StagingTelemetryV2", Invalid) + monkeypatch.setattr(rowwise_staging, "StagingRunBundleWriterV2", Invalid) status, out = run_dense_main(tmp_path, monkeypatch, staging="--staging-local-only") assert status == 0 err = capsys.readouterr().err @@ -901,11 +1007,11 @@ def test_telemetry_content_refusal_never_aborts_the_solve( from microcosm.build.staging_v2 import StagingContentError, validate_v2_bundle from microcosm.build.uk_runtime import rowwise_staging - class Refusing(rowwise_staging.StagingTelemetryV2): + class Refusing(rowwise_staging.StagingRunBundleWriterV2): def calibration_progress(self, event): raise StagingContentError("Staging file exceeds the 5242880-byte limit.") - monkeypatch.setattr(rowwise_staging, "StagingTelemetryV2", Refusing) + monkeypatch.setattr(rowwise_staging, "StagingRunBundleWriterV2", Refusing) status, out = run_dense_main( tmp_path, monkeypatch, staging="--staging-local-only", on_prepare=_drive_epochs ) diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_exact_k_ladder_launcher.py b/packages/microcosm-build/tests/engine_free/us/test_us_exact_k_ladder_launcher.py index 484e548f5..bb1623eb3 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_exact_k_ladder_launcher.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_exact_k_ladder_launcher.py @@ -365,7 +365,7 @@ def __init__(self, **kwargs): monkeypatch.setattr( launcher.fiscal_release, - "StagingTelemetry", + "StagingRunBundleWriter", UnexpectedTelemetry, ) argv = launcher._builder_argv( @@ -376,13 +376,27 @@ def __init__(self, **kwargs): ) parsed = launcher.fiscal_release._parse_args(argv) - telemetry = launcher.fiscal_release._staging_telemetry( + class Emitter: + def transition_stage(self, *args, **kwargs): + pass + + def transition_calibration_progress(self, *args, **kwargs): + pass + + def fail(self, *args, **kwargs): + pass + + def complete(self): + pass + + progress = launcher.fiscal_release._build_progress( parsed, release_root=tmp_path / "out", release_id=config.release_id, + emitter=Emitter(), ) assert parsed.no_staging is True - assert telemetry is None + assert progress.staging_bundle is None assert constructed is False assert not (tmp_path / "out" / "staging").exists() diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py b/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py index fe38277f8..73cc7bfad 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py @@ -6,6 +6,7 @@ import os import sys from dataclasses import replace +from datetime import UTC, datetime from pathlib import Path from types import SimpleNamespace @@ -94,6 +95,72 @@ def test_release_and_fiscal_scorer_signatures_have_no_membership_switches() -> N } +def test_telemetry_attempt_id_is_available_before_release_inputs_are_loaded() -> None: + builder = _load_builder_module() + timestamp = datetime(2026, 10, 5, 12, 30, tzinfo=UTC) + + assert ( + builder._telemetry_run_id( + SimpleNamespace(staging_run_id="staged-attempt", release_id="release"), + timestamp=timestamp, + ) + == "staged-attempt" + ) + assert ( + builder._telemetry_run_id( + SimpleNamespace(staging_run_id=None, release_id="release"), + timestamp=timestamp, + ) + == "release" + ) + generated = builder._telemetry_run_id( + SimpleNamespace(staging_run_id=None, release_id=None), + timestamp=timestamp, + ) + assert generated.startswith("populace-us-build-20261005T123000Z-") + assert len(generated.rsplit("-", 1)[-1]) == 8 + + +def test_us_emitter_starts_before_dirty_worktree_refusal(monkeypatch) -> None: + builder = _load_builder_module() + calls = [] + + class FakeEmitter: + available = True + + def transition_stage(self, stage_id, **details): + calls.append(("stage", stage_id, details)) + + def fail(self, error, **details): + calls.append(("failed", type(error).__name__, details)) + + monkeypatch.setattr( + builder, + "_parse_args", + lambda argv: SimpleNamespace( + staging_run_id="observed-run", + release_id="release-id", + dry_run_gates_report=None, + ), + ) + monkeypatch.setattr(builder._ReleaseDryRun, "start", lambda args, argv: None) + monkeypatch.setattr(builder, "_git_dirty", lambda: True) + + def start_emitter(**run): + calls.append(("started", run)) + return FakeEmitter() + + monkeypatch.setattr(builder, "start_local_telemetry_emitter_service", start_emitter) + + with pytest.raises(SystemExit, match="dirty git worktree"): + builder.main([]) + + assert calls[0][0] == "started" + assert calls[0][1]["run_id"] == "observed-run" + assert calls[1][0:2] == ("stage", "preflight") + assert calls[2][0:2] == ("failed", "SystemExit") + + @pytest.mark.parametrize( "removed_option", ( @@ -6768,6 +6835,8 @@ class LiveTelemetry: run_id = "live-telemetry-test" repo_id = "policyengine/populace-us-staging" uploads_succeeded = 3 + staging_bundle = object() + staging_opt_out_reason = None def stage(self, stage, **details): captured.setdefault("telemetry_events", []).append(("stage", stage)) @@ -6800,7 +6869,7 @@ def complete(self): live_telemetry = LiveTelemetry() monkeypatch.setattr( builder, - "_staging_telemetry", + "_build_progress", lambda *args, **kwargs: live_telemetry, ) if terminal_mode in { @@ -13321,7 +13390,28 @@ def test_us_release_id_guard() -> None: raise AssertionError("Expected non-US release id to fail.") -def test_staging_telemetry_defaults_on_and_no_staging_disables(tmp_path, monkeypatch): +class _TestEmitter: + available = True + + def __init__(self) -> None: + self.events = [] + + def transition_stage(self, stage, **details): + self.events.append(("stage", stage, details)) + + def transition_calibration_progress(self, event): + self.events.append(("calibration", event)) + + def fail(self, error): + self.events.append(("failed", error)) + + def complete(self): + self.events.append(("completed",)) + + +def test_staging_bundle_defaults_on_and_no_staging_disables_files( + tmp_path, monkeypatch +): module = _load_builder_module() # The parser defaults staging uploads ON (overridable by env). @@ -13353,20 +13443,51 @@ def namespace(no_staging: bool) -> SimpleNamespace: staging_upload_interval_seconds=60.0, ) - telemetry = module._staging_telemetry( - namespace(no_staging=False), release_root=tmp_path, release_id="rel-1" + progress = module._build_progress( + namespace(no_staging=False), + release_root=tmp_path, + release_id="rel-1", + emitter=_TestEmitter(), ) - assert telemetry is not None - assert telemetry.run_id == "rel-1" - assert telemetry.repo_id is None + assert progress.run_id == "rel-1" + assert progress.staging_bundle is not None + assert progress.repo_id is None - # --no-staging wins even when a staging destination is configured. - assert ( - module._staging_telemetry( - namespace(no_staging=True), release_root=tmp_path, release_id="rel-1" - ) - is None + # --no-staging disables files without disabling hosted progress. + progress = module._build_progress( + namespace(no_staging=True), + release_root=tmp_path, + release_id="rel-1", + emitter=_TestEmitter(), ) + assert progress.staging_bundle is None + assert progress.staging_opt_out_reason == "--no-staging" + + +def test_hosted_progress_survives_staging_bundle_failure() -> None: + module = _load_builder_module() + emitter = _TestEmitter() + + class FailingBundle: + def stage(self, *args, **kwargs): + raise OSError("staging disk unavailable") + + progress = module._BuildProgress( + run_id="rel-1", + emitter=emitter, + staging_bundle=FailingBundle(), + ) + + with pytest.raises(OSError, match="staging disk unavailable"): + progress.stage("target_compilation", batches=4) + + assert emitter.events == [ + ( + "stage", + "target_compilation", + {"status": "running", "message": None, "batches": 4}, + ) + ] def test_blank_staging_repo_id_is_refused_at_parse_time(monkeypatch, capsys) -> None: @@ -13416,7 +13537,7 @@ class Telemetry: def fail(self, error): recorded.append(error) - monkeypatch.setattr(module, "_ACTIVE_TELEMETRY", Telemetry()) + monkeypatch.setattr(module, "_ACTIVE_PROGRESS", Telemetry()) monkeypatch.setattr( module, "_main", @@ -13436,7 +13557,7 @@ class ExplodingTelemetry: def fail(self, error): raise RuntimeError("telemetry itself is broken") - monkeypatch.setattr(module, "_ACTIVE_TELEMETRY", ExplodingTelemetry()) + monkeypatch.setattr(module, "_ACTIVE_PROGRESS", ExplodingTelemetry()) monkeypatch.setattr( module, "_main", @@ -13449,7 +13570,7 @@ def fail(self, error): assert "could not record the staging run as failed" in capsys.readouterr().err -def test_staging_telemetry_clears_any_previous_active_run(tmp_path) -> None: +def test_build_progress_replaces_any_previous_active_run(tmp_path) -> None: module = _load_builder_module() args = SimpleNamespace( no_staging=False, @@ -13459,15 +13580,24 @@ def test_staging_telemetry_clears_any_previous_active_run(tmp_path) -> None: staging_prefix=module.DEFAULT_STAGING_PREFIX, staging_upload_interval_seconds=60.0, ) - module._staging_telemetry(args, release_root=tmp_path, release_id="rel-1") - assert module._ACTIVE_TELEMETRY is not None + first = module._build_progress( + args, + release_root=tmp_path, + release_id="rel-1", + emitter=_TestEmitter(), + ) + assert module._ACTIVE_PROGRESS is first + assert first.staging_bundle is not None args.no_staging = True - assert ( - module._staging_telemetry(args, release_root=tmp_path, release_id="rel-2") - is None + second = module._build_progress( + args, + release_root=tmp_path, + release_id="rel-2", + emitter=_TestEmitter(), ) - assert module._ACTIVE_TELEMETRY is None + assert module._ACTIVE_PROGRESS is second + assert second.staging_bundle is None def test_staging_manifest_block_distinguishes_opt_out_from_delivery() -> None: @@ -13482,6 +13612,8 @@ class Delivered: run_id = "rel-1" repo_id = "policyengine/populace-us-staging" uploads_succeeded = 7 + staging_bundle = object() + staging_opt_out_reason = None assert module._staging_manifest_block(Delivered()) == { "enabled": True, @@ -13494,6 +13626,8 @@ class Undelivered: run_id = "rel-2" repo_id = None uploads_succeeded = 0 + staging_bundle = object() + staging_opt_out_reason = None block = module._staging_manifest_block(Undelivered()) assert block["enabled"] is True @@ -13501,7 +13635,7 @@ class Undelivered: assert block["repo_id"] is None -def test_staging_telemetry_refuses_a_destinationless_namespace(tmp_path) -> None: +def test_staging_bundle_refuses_a_destinationless_namespace(tmp_path) -> None: module = _load_builder_module() args = SimpleNamespace( no_staging=False, @@ -13513,7 +13647,12 @@ def test_staging_telemetry_refuses_a_destinationless_namespace(tmp_path) -> None ) with pytest.raises(ValueError, match="no destination"): - module._staging_telemetry(args, release_root=tmp_path, release_id="rel-1") + module._build_progress( + args, + release_root=tmp_path, + release_id="rel-1", + emitter=_TestEmitter(), + ) # --------------------------------------------------------------------------- @@ -14720,7 +14859,7 @@ def _record_qrf_tail(builder, tmp_path, frame, *, register, allow, failures): allow_concentration=allow, terminal_gate_failures=failures, release_dir=tmp_path, - telemetry=builder._TerminalBatchTelemetry(recorder, failures), + telemetry=builder._TerminalBatchProgress(recorder, failures), ) return register_failures, recorder diff --git a/tools/build_us_fiscal_refresh_release.py b/tools/build_us_fiscal_refresh_release.py index 9c16a01e1..630194a81 100644 --- a/tools/build_us_fiscal_refresh_release.py +++ b/tools/build_us_fiscal_refresh_release.py @@ -43,6 +43,7 @@ import sys import time import tomllib +import uuid from collections.abc import Callable, Iterable, Mapping, Sequence from contextlib import contextmanager, nullcontext from datetime import UTC, datetime @@ -67,7 +68,11 @@ ) from microcosm.build.ledger_artifact import load_ledger_consumer_artifact from microcosm.build.source_runtime import SourceRuntimeConfig, run_source_stage -from microcosm.build.staging import DEFAULT_STAGING_PREFIX, StagingTelemetry +from microcosm.build.staging import DEFAULT_STAGING_PREFIX, StagingRunBundleWriter +from microcosm.build.telemetry_emitter import ( + LocalTelemetryEmitter, + start_local_telemetry_emitter_service, +) from microcosm.build.us_runtime import ( ASEC_2023_WEEKS_UNEMPLOYED_SOURCE_SHA256, CONGRESSIONAL_DISTRICT_VINTAGE_CROSSWALK_SHA256_ATTR, @@ -1773,7 +1778,7 @@ def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: "--staging-dir", type=Path, help=( - "Optional local directory for staging telemetry artifacts. Defaults " + "Optional local directory for staging run files. Defaults " "to /staging/runs/ when --staging-repo-id is set." ), ) @@ -1781,7 +1786,7 @@ def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: "--staging-repo-id", default=_env_default("POPULACE_STAGING_REPO_ID", STAGING_REPO_ID), help=( - "Hugging Face dataset repo to upload staging telemetry to while " + "Hugging Face dataset repo to upload staging run files to while " "the build runs. On by default (uploads are best-effort and never " "fail the build); override with POPULACE_STAGING_REPO_ID or " "disable with --no-staging. An empty POPULACE_STAGING_REPO_ID is " @@ -1810,7 +1815,10 @@ def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: parser.add_argument( "--no-staging", action="store_true", - help="Disable staging telemetry (local staging dir and uploads) for this build.", + help=( + "Disable staging run files and their uploads for this build; hosted " + "telemetry remains active." + ), ) parser.add_argument( "--staging-prefix", @@ -1907,7 +1915,7 @@ def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: # effect of a blank repo id. parser.error( "--staging-repo-id is empty and no --staging-dir is set, so staging " - "telemetry would silently do nothing. Pass --no-staging to skip " + "run files would have no destination. Pass --no-staging to skip " "staging deliberately, or --staging-dir for a local-only run." ) if args.evidence_failure_owners is not None and not args.evidence_release: @@ -5168,7 +5176,12 @@ def _score( parts: dict[PostExportKey, list[tuple[np.ndarray, np.ndarray]]] = { key: [] for key in keys } - for batch_frame in batches: + sweep = _WorkCounter( + stage_id="post_export_scoring", + unit="scoring_batch", + total=len(batches), + ) + for batch, batch_frame in enumerate(batches, start=1): with _automatic_gc_suspended(): simulation = self._construct(batch_frame, reform_system=reform_system) try: @@ -5194,6 +5207,7 @@ def _score( release_engine_simulation(simulation) del simulation, engine _collect_batch_garbage() + sweep.advance(pass_name=label, batch=batch, batches=len(batches)) _collect_family_garbage() return { key: _concatenate_post_export_parts(key_parts) @@ -5423,6 +5437,7 @@ def _reform_household_income_tax( reform_income_tax[household_positions] = batch_income_tax del batch_income_tax, reformed, reformed_dataset, batch_frame _collect_batch_garbage() + _advance_work(pass_name=reform_spec.measure, batch=batch, batches=len(batches)) del reform_system _collect_family_garbage() return reform_income_tax @@ -6783,6 +6798,7 @@ def _materialize_base_simulation_columns( columns[column] = pool_values del batch_columns, batch_frame _collect_batch_garbage() + _advance_work(pass_name="base", batch=batch, batches=len(batches)) _collect_family_garbage() assert column_order is not None @@ -6830,6 +6846,20 @@ def _materialize_target_frame( _assert_no_formula_owned_columns(base_frame) system = CountryTaxBenefitSystem() n_households = base_frame.n("household") + requested_reform_measures = {spec.measure for spec in target_specs} + global _ACTIVE_WORK + passes = 1 + sum( + spec.measure in requested_reform_measures + for spec in US_JCT_TAX_EXPENDITURE_REFORMS + ) + batch_count = len( + tuple(_household_position_batches(n_households, maximum_microsim_batch_size)) + ) + _ACTIVE_WORK = _WorkCounter( + stage_id="target_compilation", + unit="engine_batch", + total=passes * batch_count, + ) # The base simulation uses the JCT reform loop's household partition. base_columns, base_simulation_batching = _materialize_base_simulation_columns( base_frame, @@ -6849,7 +6879,6 @@ def _materialize_target_frame( del base_columns base_income_tax_household = hh["income_tax"].to_numpy(dtype=np.float64) - requested_reform_measures = {spec.measure for spec in target_specs} cache_context = ( dict(target_materialization_cache_context) if target_materialization_cache_context is not None @@ -6888,6 +6917,11 @@ def _materialize_target_frame( if cached is not None: reform_income_tax, cache_digest, cache_path = cached cache_stats["hits"] = int(cache_stats["hits"]) + 1 + _advance_work( + batch_count, + pass_name=reform_spec.measure, + cached=True, + ) cache_entry = { "measure": reform_spec.measure, "neutralized_variable": reform_spec.neutralized_variable, @@ -8020,7 +8054,7 @@ def _record_qrf_tail_concentration_gate( allow_concentration: bool, terminal_gate_failures: list[str], release_dir: Path, - telemetry: _TerminalBatchTelemetry, + telemetry: _TerminalBatchProgress, ) -> list[str]: """Evaluate the terminal QRF tail gate and record everything it measured. @@ -8783,7 +8817,7 @@ def _enforce_ssi_take_up_delivery( *, targets: Mapping[str, float], release_dir: Path, - telemetry: StagingTelemetry | None, + telemetry: _BuildProgress | None, enforcement_fences: Mapping[str, str] | None = None, ) -> tuple[list[str], GateResult]: """Fail the release on an enforced-band delivery miss, via the batch. @@ -11211,6 +11245,25 @@ def _default_release_id( return f"populace-us-2024-{digest}-{commit}-{build_timestamp:%Y%m%dT%H%M%SZ}" +def _telemetry_run_id(args: argparse.Namespace, *, timestamp: datetime) -> str: + """Choose an attempt id before release inputs have been loaded. + + An explicit staging id remains authoritative, and an explicit release id + is already stable enough to identify the attempt. Builds that derive their + release id from the input digest need a separate attempt id so telemetry can + start before downloading or hashing that input. + """ + + if args.staging_run_id: + return str(args.staging_run_id) + if args.release_id: + return str(args.release_id) + instant = timestamp.astimezone(UTC) + return ( + f"populace-us-build-{instant.strftime('%Y%m%dT%H%M%SZ')}-{uuid.uuid4().hex[:8]}" + ) + + def _assert_us_release_id(release_id: str, *, evidence_release: bool = False) -> None: if not release_id.startswith("populace-us-"): raise ValueError( @@ -11404,11 +11457,121 @@ def _assert_exact_k_original_pool_alignment( ) -#: The staging run for the build in flight, so the entry point can mark it -#: failed on the way out. The build body hands its telemetry object down a -#: large call stack; a module-level handle avoids threading a second copy back -#: up purely for the failure path. -_ACTIVE_TELEMETRY: StagingTelemetry | None = None +_ACTIVE_EMITTER: LocalTelemetryEmitter | None = None + + +class _BuildProgress: + """Report progress to an emitter and, independently, a staging bundle.""" + + def __init__( + self, + *, + run_id: str, + emitter: LocalTelemetryEmitter, + staging_bundle: StagingRunBundleWriter | None, + staging_opt_out_reason: str | None = None, + ) -> None: + self.run_id = run_id + self.emitter = emitter + self.staging_bundle = staging_bundle + self.staging_opt_out_reason = staging_opt_out_reason + + @property + def repo_id(self) -> str | None: + if self.staging_bundle is None: + return None + return self.staging_bundle.repo_id + + @property + def uploads_succeeded(self) -> int: + if self.staging_bundle is None: + return 0 + return self.staging_bundle.uploads_succeeded + + def stage( + self, + stage: str, + *, + message: str | None = None, + status: str = "running", + force_upload: bool = False, + **details: Any, + ) -> None: + self.emitter.transition_stage( + stage, + status=status, + message=message, + **details, + ) + if self.staging_bundle is not None: + self.staging_bundle.stage( + stage, + message=message, + status=status, + force_upload=force_upload, + **details, + ) + + def calibration_progress(self, event: dict[str, object]) -> None: + self.emitter.transition_calibration_progress(event) + if self.staging_bundle is not None: + self.staging_bundle.calibration_progress(event) + + def attach_artifact( + self, + name: str, + path: Path | str, + **details: Any, + ) -> None: + if self.staging_bundle is not None: + self.staging_bundle.attach_artifact(name, path, **details) + + def fail(self, error: BaseException) -> None: + self.emitter.fail(error) + if self.staging_bundle is not None: + self.staging_bundle.fail(error) + + def complete(self) -> None: + if self.staging_bundle is not None: + self.staging_bundle.complete() + self.emitter.complete() + + +#: The progress coordinator for the build in flight, so the entry point can +#: report failure without threading another handle back through the call stack. +_ACTIVE_PROGRESS: _BuildProgress | None = None + + +class _WorkCounter: + """Report completed independent work units through the emitter service.""" + + def __init__(self, *, stage_id: str, unit: str, total: int) -> None: + self.stage_id = stage_id + self.unit = unit + self.total = max(0, int(total)) + self.done = 0 + self.started = time.monotonic() + + def advance(self, units: int = 1, **details: object) -> None: + self.done = min(self.total, self.done + max(0, int(units))) + if _ACTIVE_EMITTER is None: + return + _ACTIVE_EMITTER.progress( + self.stage_id, + done=self.done, + total=self.total, + unit=self.unit, + elapsed_seconds=time.monotonic() - self.started, + **details, + ) + + +_ACTIVE_WORK: _WorkCounter | None = None + + +def _advance_work(units: int = 1, **details: object) -> None: + if _ACTIVE_WORK is not None: + _ACTIVE_WORK.advance(units, **details) class _ReleaseDryRun: @@ -11417,7 +11580,8 @@ class _ReleaseDryRun: ``_main`` runs the release as it would build it, up to the point where the staged frame goes to target materialization. It differs in four ways: - * Nothing is written under ``--out`` and staging telemetry stays off. + * Nothing is written under ``--out`` and staging run files stay off; + hosted progress reporting remains active. * The input-mass, degenerate-input and eCPS parity gates take their degraded-mode branch and batch instead of raising, so one report carries them with every register. @@ -11584,7 +11748,7 @@ def _write(self, report: ReleaseDryRunReport) -> int: #: The dry run in flight, so :func:`main` can turn a refusal before the stop -#: point into that run's report. Like :data:`_ACTIVE_TELEMETRY`, a module-level +#: point into that run's report. Like :data:`_ACTIVE_PROGRESS`, a module-level #: handle rather than a second object threaded back up the build's call stack. _ACTIVE_DRY_RUN: _ReleaseDryRun | None = None @@ -12006,15 +12170,20 @@ def zero_support() -> CheckResult: return checks -def _staging_manifest_block(telemetry: StagingTelemetry | None) -> dict[str, object]: +def _staging_manifest_block(telemetry: _BuildProgress | None) -> dict[str, object]: """Record what staging did and distinguish an opt-out from non-delivery. Uploads are best-effort and self-disable after repeated failures, so a configured destination is not evidence that anything reached it. """ - if telemetry is None: - return {"enabled": False, "reason": "--no-staging"} + if telemetry is None or telemetry.staging_bundle is None: + reason = ( + "--no-staging" + if telemetry is None + else telemetry.staging_opt_out_reason or "--no-staging" + ) + return {"enabled": False, "reason": reason} return { "enabled": True, "run_id": telemetry.run_id, @@ -12023,18 +12192,27 @@ def _staging_manifest_block(telemetry: StagingTelemetry | None) -> dict[str, obj } -def _staging_telemetry( +def _build_progress( args: argparse.Namespace, *, release_root: Path, release_id: str, -) -> StagingTelemetry | None: - global _ACTIVE_TELEMETRY + run_id: str | None = None, + emitter: LocalTelemetryEmitter, +) -> _BuildProgress: + global _ACTIVE_PROGRESS # Each call establishes the current run, so a handle from a previous one # can never be marked failed in place of this build's. - _ACTIVE_TELEMETRY = None + _ACTIVE_PROGRESS = None + run_id = run_id or args.staging_run_id or release_id if args.no_staging: - return None + _ACTIVE_PROGRESS = _BuildProgress( + run_id=run_id, + emitter=emitter, + staging_bundle=None, + staging_opt_out_reason="--no-staging", + ) + return _ACTIVE_PROGRESS if not args.staging_dir and not args.staging_repo_id: # The parser rejects this combination, so reaching it means a caller # built the namespace directly. Returning None here would reinstate @@ -12044,9 +12222,8 @@ def _staging_telemetry( "empty and staging_dir is unset. Set no_staging to skip staging, " "or give a staging_dir for a local-only run." ) - run_id = args.staging_run_id or release_id run_dir = args.staging_dir or release_root / "staging" / "runs" / run_id - _ACTIVE_TELEMETRY = StagingTelemetry( + staging_bundle = StagingRunBundleWriter( run_id=run_id, candidate_release_id=release_id, run_dir=run_dir, @@ -12054,10 +12231,15 @@ def _staging_telemetry( path_prefix=args.staging_prefix, upload_interval_seconds=args.staging_upload_interval_seconds, ) - return _ACTIVE_TELEMETRY + _ACTIVE_PROGRESS = _BuildProgress( + run_id=run_id, + emitter=emitter, + staging_bundle=staging_bundle, + ) + return _ACTIVE_PROGRESS -class _TerminalBatchTelemetry: +class _TerminalBatchProgress: """Turn terminal-batch telemetry crashes into release-gate failures. The proxy is deliberately scoped to the post-diagnostics terminal batch. @@ -12068,7 +12250,7 @@ class _TerminalBatchTelemetry: def __init__( self, - telemetry: StagingTelemetry | None, + telemetry: _BuildProgress | None, terminal_gate_failures: list[str], ) -> None: self._telemetry = telemetry @@ -12160,10 +12342,15 @@ def main(argv: Sequence[str] | None = None) -> None: if dry_run is not None and _is_release_refusal(error): # A dry run the release refuses before its stop point: the # refusal is the report's certain failure (exit 1), not a crash. + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.fail( + error, + failure_class="dry_run_refusal", + ) raise SystemExit(dry_run.refused(error)) from error - if _ACTIVE_TELEMETRY is not None: + if _ACTIVE_PROGRESS is not None: try: - _ACTIVE_TELEMETRY.fail(error) + _ACTIVE_PROGRESS.fail(error) except Exception as telemetry_error: # pragma: no cover - defensive # A failing failure-report must not replace the real traceback. print( @@ -12171,9 +12358,15 @@ def main(argv: Sequence[str] | None = None) -> None: f"{type(telemetry_error).__name__}: {telemetry_error}", file=sys.stderr, ) + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.fail(error) raise if dry_run_exit is not None: + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.complete() raise SystemExit(dry_run_exit) + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.complete() def _check_committed_us_ledger_feed_pin( @@ -12215,7 +12408,26 @@ def _check_committed_us_ledger_feed_pin( def _main(argv: Sequence[str] | None = None) -> int | None: + global _ACTIVE_EMITTER, _ACTIVE_PROGRESS + _ACTIVE_EMITTER = None + _ACTIVE_PROGRESS = None args = _parse_args(argv) + build_started = time.perf_counter() + attempt_started_at = datetime.now(UTC) + run_id = _telemetry_run_id(args, timestamp=attempt_started_at) + is_dry_run = args.dry_run_gates_report is not None + _ACTIVE_EMITTER = start_local_telemetry_emitter_service( + run_id=run_id, + country_code="US", + pipeline="us_fiscal_refresh", + candidate_id=args.release_id, + release_id=args.release_id if not is_dry_run else None, + run_kind="dry_run" if is_dry_run else "release", + ) + _ACTIVE_EMITTER.transition_stage( + "preflight", + message="Validating release inputs and configuration.", + ) # --dry-run-gates-report: runs this build up to target materialization and # returns the report's exit code from the stop point below (_ReleaseDryRun). dry_run = _ReleaseDryRun.start(args, argv) @@ -12230,7 +12442,6 @@ def _main(argv: Sequence[str] | None = None) -> int | None: if args.evidence_release else () ) - build_started = time.perf_counter() timing: dict[str, float] = {} if args.release_id: @@ -12514,17 +12725,24 @@ def _main(argv: Sequence[str] | None = None) -> int | None: checkpoint_root.mkdir(parents=True, exist_ok=True) artifact_root.mkdir(parents=True, exist_ok=True) release_dir.mkdir(parents=True, exist_ok=True) - # A dry run writes only its report: no directories under --out, and no - # staging run that a dashboard would show as a release attempt. - telemetry = ( - _staging_telemetry( + # A dry run writes only its report under --out and creates no staging + # artifacts. Hosted telemetry identifies it separately from a release. + if dry_run is None: + telemetry = _build_progress( args, release_root=release_root, release_id=release_id, + run_id=run_id, + emitter=_ACTIVE_EMITTER, ) - if dry_run is None - else None - ) + else: + telemetry = _BuildProgress( + run_id=run_id, + emitter=_ACTIVE_EMITTER, + staging_bundle=None, + staging_opt_out_reason="dry run", + ) + _ACTIVE_PROGRESS = telemetry if telemetry is not None: telemetry.stage( "target_registry", @@ -15048,7 +15266,7 @@ def _main(argv: Sequence[str] | None = None) -> int | None: # An unbuildable post-export plan refuses here, before the export write, # with every other terminal group still evaluated (microcosm#956). terminal_gate_failures.extend(post_export_scoring_plan.terminal_failures()) - terminal_batch_telemetry = _TerminalBatchTelemetry( + terminal_batch_telemetry = _TerminalBatchProgress( telemetry, terminal_gate_failures, ) @@ -15407,7 +15625,7 @@ def _main(argv: Sequence[str] | None = None) -> int | None: # The owned failures ride into the release manifest's known_failures # block instead of aborting the export; the H5 written below carries # the calibrated weights, so the sidecar is not written on this path. - # Failures appended AFTER this point (a _TerminalBatchTelemetry crash + # Failures appended AFTER this point (a _TerminalBatchProgress crash # line, the smoke/take-up/coverage recordings) are owner-checked at # their own append sites and again before the manifest write; if one # is unowned the run dies post-H5 — weights retained in the written diff --git a/tools/generate_staging_contract_fixtures.py b/tools/generate_staging_contract_fixtures.py index a1733a584..71563c8cd 100644 --- a/tools/generate_staging_contract_fixtures.py +++ b/tools/generate_staging_contract_fixtures.py @@ -9,7 +9,7 @@ import tempfile from pathlib import Path -from microcosm.build.staging_v2 import StagingTelemetryV2 +from microcosm.build.staging_v2 import StagingRunBundleWriterV2 ROOT = Path(__file__).resolve().parents[1] DEFAULT_OUTPUT = ( @@ -29,7 +29,7 @@ def __call__(self) -> str: def _completed_spine(root: Path) -> None: - recorder = StagingTelemetryV2( + recorder = StagingRunBundleWriterV2( run_id="uk-spine-v2-fixture", country_code="GB", operation_id="uk_frs_spine", @@ -54,7 +54,7 @@ def _completed_spine(root: Path) -> None: def _calibration(root: Path) -> None: - recorder = StagingTelemetryV2( + recorder = StagingRunBundleWriterV2( run_id="uk-calibration-v2-fixture", country_code="GB", operation_id="uk_national_calibration", @@ -86,7 +86,7 @@ def _calibration(root: Path) -> None: def _failed(root: Path) -> None: - recorder = StagingTelemetryV2( + recorder = StagingRunBundleWriterV2( run_id="uk-failed-v2-fixture", country_code="GB", operation_id="uk_frs_spine", diff --git a/uv.lock b/uv.lock index 37d480654..774fbc6d8 100644 --- a/uv.lock +++ b/uv.lock @@ -1,5 +1,5 @@ version = 1 -revision = 3 +revision = 5 requires-python = ">=3.13" resolution-markers = [ "python_full_version >= '3.14' and sys_platform == 'win32'", @@ -22,6 +22,20 @@ members = [ "microcosm-workspace", ] +[[package]] +name = "alembic" +version = "1.20.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "mako" }, + { name = "sqlalchemy" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ed/aa/02910bdb8e2f1444f6654d5b296cd827d126f82209050ee7b1000f92ac4b/alembic-1.20.0.tar.gz", hash = "sha256:db505480647bc60386c5369402f4a57a506b7539c9e9ef5e270d45cbbe4939bf", size = 2093272, upload-time = "2026-09-11T19:09:11.126Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/3f/27/78a89b55b0904d222183164e079b4ca56208e94eff1d35ad1f1ad5be9b06/alembic-1.20.0-py3-none-any.whl", hash = "sha256:77eb101048d95f982c0353e9233404889dcd7a6fc244c107836c0e2fc9cf7d9d", size = 268719, upload-time = "2026-09-11T19:09:12.88Z" }, +] + [[package]] name = "annotated-doc" version = "0.0.4" @@ -206,7 +220,7 @@ name = "cuda-bindings" version = "13.3.1" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "cuda-pathfinder", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "cuda-pathfinder" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/cc/6e/2394f8163360f8391f8f1b7e72d300a82724edb81a7b7084c799fbd4c91f/cuda_bindings-13.3.1-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9efb21c1ee64981e184b9e0ba5eb3179e5ba3d4b51665a6cb52b8ef3d01a7cbf", size = 5920504, upload-time = "2026-05-29T23:11:56.883Z" }, @@ -235,34 +249,34 @@ wheels = [ [package.optional-dependencies] cudart = [ - { name = "nvidia-cuda-runtime", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cuda-runtime" }, ] cufft = [ - { name = "nvidia-cufft", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cufft" }, ] cufile = [ - { name = "nvidia-cufile", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cufile" }, ] cupti = [ - { name = "nvidia-cuda-cupti", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cuda-cupti" }, ] curand = [ - { name = "nvidia-curand", marker = "sys_platform == 'linux'" }, + { name = "nvidia-curand" }, ] cusolver = [ - { name = "nvidia-cusolver", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cusolver" }, ] cusparse = [ - { name = "nvidia-cusparse", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cusparse" }, ] nvjitlink = [ - { name = "nvidia-nvjitlink", marker = "sys_platform == 'linux'" }, + { name = "nvidia-nvjitlink" }, ] nvrtc = [ - { name = "nvidia-cuda-nvrtc", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cuda-nvrtc" }, ] nvtx = [ - { name = "nvidia-nvtx", marker = "sys_platform == 'linux'" }, + { name = "nvidia-nvtx" }, ] [[package]] @@ -653,6 +667,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/41/45/1a4ed80516f02155c51f51e8cedb3c1902296743db0bbc66608a0db2814f/jsonschema_specifications-2025.9.1-py3-none-any.whl", hash = "sha256:98802fee3a11ee76ecaca44429fda8a41bff98b00a0f2838151b113f210cc6fe", size = 18437, upload-time = "2025-09-08T01:34:57.871Z" }, ] +[[package]] +name = "mako" +version = "1.4.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "markupsafe" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/5a/09/e07c4b5579a79f4b16f8d4f29f6c54514ac787c4ad506b8c4f28a0e6b0bf/mako-1.4.3.tar.gz", hash = "sha256:cd6537fe88d5fec315c55c2f8529bc4ce7a9a352ad7db3eeaa6a66e2dd4ec37a", size = 412799, upload-time = "2026-09-22T20:54:31.509Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/6d/a0/053d6af3e8f871e0073b4a36732d9e65be77a72e5434c31b94f6af78a6bb/mako-1.4.3-py3-none-any.whl", hash = "sha256:723296007c870bfd6b3f0c3230dba7198096e5269297ebf5e4eff9e7ffa39d4f", size = 80164, upload-time = "2026-09-22T20:54:33.128Z" }, +] + [[package]] name = "markdown-it-py" version = "4.2.0" @@ -743,6 +769,7 @@ name = "microcosm-build" version = "0.1.0" source = { editable = "packages/microcosm-build" } dependencies = [ + { name = "alembic" }, { name = "huggingface-hub" }, { name = "jsonschema" }, { name = "microcosm-calibrate" }, @@ -756,6 +783,7 @@ dependencies = [ { name = "pyyaml" }, { name = "referencing" }, { name = "scipy" }, + { name = "sqlalchemy" }, ] [package.optional-dependencies] @@ -781,6 +809,7 @@ dev = [ [package.metadata] requires-dist = [ + { name = "alembic", specifier = ">=1.13.3,<2" }, { name = "h5py", marker = "extra == 'uk'", specifier = ">=3" }, { name = "h5py", marker = "extra == 'us'", specifier = ">=3" }, { name = "huggingface-hub", specifier = ">=0.20" }, @@ -802,6 +831,7 @@ requires-dist = [ { name = "pyyaml", specifier = ">=6" }, { name = "referencing", specifier = ">=0.35,<1" }, { name = "scipy", specifier = ">=1.13" }, + { name = "sqlalchemy", specifier = ">=2,<3" }, { name = "tables", marker = "extra == 'uk'", specifier = ">=3" }, { name = "tables", marker = "extra == 'us'", specifier = ">=3" }, ] @@ -1251,7 +1281,7 @@ name = "nvidia-cublas" version = "13.1.1.3" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cuda-nvrtc", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-cuda-nvrtc" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/a7/a1/0bd24ee8c8d03adac032fd2909426a00c88f8c57961b1277ded97f91119f/nvidia_cublas-13.1.1.3-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:b7a210458267ac818974c53038fbec2e969d5c99f305ab15c72522fa9f001dd5", size = 542848918, upload-time = "2026-04-08T18:46:22.985Z" }, @@ -1290,7 +1320,7 @@ name = "nvidia-cudnn-cu13" version = "9.20.0.48" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cublas", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-cublas" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/56/c5/83384d846b2fd17c44bd499b36c75a45ed4f095fbbb2252294e89cea5c5c/nvidia_cudnn_cu13-9.20.0.48-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:e31454ae00094b0c55319d9d15b6fa2fc50a9e1c0f5c8c80fb75258234e731e1", size = 444574296, upload-time = "2026-03-09T19:28:27.751Z" }, @@ -1302,7 +1332,7 @@ name = "nvidia-cufft" version = "12.0.0.61" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-nvjitlink", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-nvjitlink" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/8b/ae/f417a75c0259e85c1d2f83ca4e960289a5f814ed0cea74d18c353d3e989d/nvidia_cufft-12.0.0.61-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:2708c852ef8cd89d1d2068bdbece0aa188813a0c934db3779b9b1faa8442e5f5", size = 214053554, upload-time = "2025-09-04T08:31:38.196Z" }, @@ -1332,9 +1362,9 @@ name = "nvidia-cusolver" version = "12.0.4.66" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cublas", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, - { name = "nvidia-cusparse", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, - { name = "nvidia-nvjitlink", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-cublas" }, + { name = "nvidia-cusparse" }, + { name = "nvidia-nvjitlink" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/c8/c3/b30c9e935fc01e3da443ec0116ed1b2a009bb867f5324d3f2d7e533e776b/nvidia_cusolver-12.0.4.66-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:02c2457eaa9e39de20f880f4bd8820e6a1cfb9f9a34f820eb12a155aa5bc92d2", size = 223467760, upload-time = "2025-09-04T08:33:04.222Z" }, @@ -1346,7 +1376,7 @@ name = "nvidia-cusparse" version = "12.6.3.3" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-nvjitlink", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-nvjitlink" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/f8/94/5c26f33738ae35276672f12615a64bd008ed5be6d1ebcb23579285d960a9/nvidia_cusparse-12.6.3.3-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:80bcc4662f23f1054ee334a15c72b8940402975e0eab63178fc7e670aa59472c", size = 162155568, upload-time = "2025-09-04T08:33:42.864Z" }, @@ -1477,7 +1507,7 @@ name = "pexpect" version = "4.9.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "ptyprocess", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "ptyprocess" }, ] sdist = { url = "https://files.pythonhosted.org/packages/42/92/cc564bf6381ff43ce1f4d06852fc19a2f11d180f23dc32d9588bee2f149d/pexpect-4.9.0.tar.gz", hash = "sha256:ee7d41123f3c9911050ea2c2dac107568dc43b2d3b0c7557a33212c398ead30f", size = 166450, upload-time = "2023-11-25T09:07:26.339Z" } wheels = [ @@ -2131,6 +2161,68 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/19/a0/c484f69a0ebf88a9b9fd0ac28176f8df66550f46714f1b0c5ec0b360824e/spm_calculator-1.0.0-py3-none-any.whl", hash = "sha256:e354937a5e1a4045d4966ed594a528d8b02866fabaac9bb5672017004b627305", size = 7395384, upload-time = "2026-09-11T15:45:10.492Z" }, ] +[[package]] +name = "sqlalchemy" +version = "2.1.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/02/4b/81d972a46c9f1d978af1795e2abc988855c711453552cb45d75f6e4abbf5/sqlalchemy-2.1.3.tar.gz", hash = "sha256:ade5281df06038c6394f532590d592d3381ee5d11b9e0006a2570aef41145118", size = 10525778, upload-time = "2026-10-02T22:57:00.308Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/1a/71/68c29931fb2c5178241f7d0c63dbdd800ad65685eac70d7f2117145b5e3c/sqlalchemy-2.1.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:260bdd8baa1dc631d76d14b26ad009ebbb665b4469666afef8dcaa1f964016cb", size = 2457479, upload-time = "2026-10-03T02:39:19.741Z" }, + { url = "https://files.pythonhosted.org/packages/94/21/5dbd91816b2cdb367bca3195f300ffeca916762e088d6924fc33a18f50ad/sqlalchemy-2.1.3-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:5fc01a8047d6a4f4e997441994f07af79fdd470fc80669751e32f91153d6b491", size = 4582199, upload-time = "2026-10-02T23:37:44.508Z" }, + { url = "https://files.pythonhosted.org/packages/32/d1/79962ce06d6cd34516def113f5170184fa9401ebddc59fd6cb18f6e919f3/sqlalchemy-2.1.3-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b1b3ff10eba0f9290142375e667351c99d24ee7fcff6e0f85bd8be9d39b50a17", size = 4632923, upload-time = "2026-10-02T23:35:55.318Z" }, + { url = "https://files.pythonhosted.org/packages/3c/94/03e98031dc9a1738c580c3bf01beeafe695f247eb647f4e8dda3fab2454f/sqlalchemy-2.1.3-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:efb1ebf695e80d9a69f5bbff9d812d2491fd168b27d6b42ad4c8a4d3d6faaef1", size = 4300224, upload-time = "2026-10-02T23:46:45.495Z" }, + { url = "https://files.pythonhosted.org/packages/3a/1f/388c87607eb6c7abee4866355eeebb6221b3bcb413a64ce4a5cfc3edd22f/sqlalchemy-2.1.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:a95754928d47b37ea5c4b5180101db2afabef1da0dd5cd0c573b547d5d036ff0", size = 4507170, upload-time = "2026-10-02T23:38:29.697Z" }, + { url = "https://files.pythonhosted.org/packages/57/73/5515a3f555d31748564fbeb0b9e0b6ab7a647bf2ea0cdb6d233f20f710e3/sqlalchemy-2.1.3-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:7606b413730ac6002930faa96f52195ed6969040b78071a3336370661f5e9e09", size = 4301439, upload-time = "2026-10-02T23:46:47.845Z" }, + { url = "https://files.pythonhosted.org/packages/a3/a0/0eae9b96f6f8a09f3e5d388f6fc5d6fe8d5ff2aced2a5c75bea7333d445e/sqlalchemy-2.1.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:76c3d5c7ada0925ff12e8c2395c547f88fcfccc13368ff8e8dd55ab5e28dcdab", size = 4594441, upload-time = "2026-10-03T02:46:16.928Z" }, + { url = "https://files.pythonhosted.org/packages/f5/b1/b6d681015414ce9048c07ef3ac49b693c1b8373de4716ee4bdeff4ec19a0/sqlalchemy-2.1.3-cp313-cp313-win32.whl", hash = "sha256:cb6fc5c856e8f1c0228bf74b636c0a03a6c12dd502ced7aa82ced21dcdc7f547", size = 2371547, upload-time = "2026-10-03T02:49:23.549Z" }, + { url = "https://files.pythonhosted.org/packages/56/ad/35aa4d737016640c6da7573a4bcd9d91b59a87359bc6df075627963a4781/sqlalchemy-2.1.3-cp313-cp313-win_amd64.whl", hash = "sha256:82d4c63d338737d593f1d29c4dae0a124577a9a76186d90b5c0f9ec9cbeeab35", size = 2421558, upload-time = "2026-10-03T02:49:25.448Z" }, + { url = "https://files.pythonhosted.org/packages/fc/42/0dcc7cd92a3f4773376a5488b6ce0d6d75d2d029ad7e7f9e0ecc5fb48fcd/sqlalchemy-2.1.3-cp313-cp313-win_arm64.whl", hash = "sha256:0ab6f42761cc5c48d2f3e8a70810d10630d8222a51a7111cf9b3a48f22c3f148", size = 2382507, upload-time = "2026-10-03T02:42:42.812Z" }, + { url = "https://files.pythonhosted.org/packages/11/55/d626c933b7d034619c57120fafb47c0d44b2553ad2b8f74e904a40c01258/sqlalchemy-2.1.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:660c6e16d254568874d4b77fd593aa5426f55cf40d0acc8b5364a5e61cc1044d", size = 2460626, upload-time = "2026-10-03T02:39:21.751Z" }, + { url = "https://files.pythonhosted.org/packages/35/90/9b276c8eed59694d5c2291588fae9c51f25a5c63efb00ef9a7c6e754a2cd/sqlalchemy-2.1.3-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7277ba31ffb636666ed207d2aaf74de3cd2ee78436da7552bd4eeadf1f67740a", size = 4576390, upload-time = "2026-10-02T23:38:31.943Z" }, + { url = "https://files.pythonhosted.org/packages/60/21/818e89fe73ba4600abf5090e0d2356d92de9ca3a8aa9222d64801e510bd2/sqlalchemy-2.1.3-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b728c406b1b202e8998e8f9adacb56773bfeea67da74252ad3025bcb8f5863f1", size = 4606177, upload-time = "2026-10-03T02:46:19.291Z" }, + { url = "https://files.pythonhosted.org/packages/88/75/ab96bb0e27153d123340d6f1b14fac3713c03b7e3ca3f74f4c0715345775/sqlalchemy-2.1.3-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:85623c57c3696dcbc4fc909bbf335a2f0c27ae36f762d83810067fb49abc040e", size = 4300438, upload-time = "2026-10-02T23:46:50.03Z" }, + { url = "https://files.pythonhosted.org/packages/94/49/c7e0223fe5553ff1c31853c07f158f75245f8d435435755a789249f5121a/sqlalchemy-2.1.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:ac0de6dc326cc99ac31206359e9fc2c13345f667edeb6ef556585fedac4ff698", size = 4506089, upload-time = "2026-10-02T23:38:34.209Z" }, + { url = "https://files.pythonhosted.org/packages/10/83/d6aad5e1a5b3a3e9257f932971b7f2105be6669205d4d9946f5f87d73216/sqlalchemy-2.1.3-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:65e002b794efa1bacbbde9fa973e5340d6c3c29b53e592b863a281802829319d", size = 4301730, upload-time = "2026-10-02T23:46:52.115Z" }, + { url = "https://files.pythonhosted.org/packages/04/54/310b9fd03ba7de6359185a0914bcd0cf45a76c6314e49d1ec814a50ae13e/sqlalchemy-2.1.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:65e4813c78eaa679c9369b289e67fdba924db6be14067700c5ccbda37953af9a", size = 4565315, upload-time = "2026-10-03T02:46:21.64Z" }, + { url = "https://files.pythonhosted.org/packages/d4/8b/f6f7d46bc4e7bdc16fa78ec9a390f01852f2b080b7c75e83db4947b707c6/sqlalchemy-2.1.3-cp314-cp314-win32.whl", hash = "sha256:726987e7f8500614016be056ba5c9715b760ab465295741e2087b29753d72632", size = 2378402, upload-time = "2026-10-03T02:49:27.524Z" }, + { url = "https://files.pythonhosted.org/packages/9a/5f/dabd1ea48dd6c45383526c7f2e453fea15dcb53235edbadab49857fd7bfd/sqlalchemy-2.1.3-cp314-cp314-win_amd64.whl", hash = "sha256:0560458a775917bcd7a47eccf4f9e0b135fd9b72d7b819cb1bd60581de946f9a", size = 2428828, upload-time = "2026-10-03T02:49:29.481Z" }, + { url = "https://files.pythonhosted.org/packages/23/aa/ecba6a6141d24dbd4b002b451e0c410b7c6cb82a8a873f915f89314af830/sqlalchemy-2.1.3-cp314-cp314-win_arm64.whl", hash = "sha256:63c9289ed81991ba4d500611ce201531711430f5d23d7ae9a61a8842639b85ea", size = 2393599, upload-time = "2026-10-03T02:42:44.416Z" }, + { url = "https://files.pythonhosted.org/packages/33/41/f2c5eb9371cc1352a16001642702dfc2d73537995eea5cdeec7738f90a9d/sqlalchemy-2.1.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:08b2f345db8b0742e7814c49f9a210d5f5c8fb9d80d12c8a63bba762fa277477", size = 2500084, upload-time = "2026-10-02T23:35:35.774Z" }, + { url = "https://files.pythonhosted.org/packages/73/f1/fabd748dd8a1700579906649ac05414d65eca5bca44a3ac3c22856ed8d25/sqlalchemy-2.1.3-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:493b632f426579c4b0e97a5610f76c8c3d401cede442d462bf700b537035c91e", size = 4904105, upload-time = "2026-10-03T02:49:06.101Z" }, + { url = "https://files.pythonhosted.org/packages/ad/6b/a7a3d5c77c16386e16eb9dc8ba82e0cc6b03d6ee0ef6620921655df82413/sqlalchemy-2.1.3-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:55e06933580f4faa5a7b1d2c9ad04f365fe01c40465eb982e38fda8758c1e95c", size = 4814161, upload-time = "2026-10-02T23:34:39.253Z" }, + { url = "https://files.pythonhosted.org/packages/63/72/275dac552b869a21774c08a7084bad16bd99f28384b8c95ffcded1a18884/sqlalchemy-2.1.3-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:fdcee9536cb3281b8810787ba651f4e5f22bd687b177de89fba2155b055b07b0", size = 4494471, upload-time = "2026-10-02T23:59:11.642Z" }, + { url = "https://files.pythonhosted.org/packages/e8/3e/41d8071c58a20e2ba9fabc422938a219461f0b32aa537ab69afe06daa0e5/sqlalchemy-2.1.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:d2cb28779dbf54067655030d7ab3a6726c7957c56bde5ff6745ed86b5cebff41", size = 4762070, upload-time = "2026-10-03T02:49:08.052Z" }, + { url = "https://files.pythonhosted.org/packages/75/73/b747233b8ad935467eb1f2aa1c466af69831693cab3495c88842bf1bd3ac/sqlalchemy-2.1.3-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:01a90cac3f8d93fcbda0df119ef56c4d58aad5601f6b7ee5a38bfabbd5d119b8", size = 4494691, upload-time = "2026-10-02T23:59:13.733Z" }, + { url = "https://files.pythonhosted.org/packages/71/aa/ba11b725b9529879aff489e36cdaccc3fd69620fb30b89186792523bc262/sqlalchemy-2.1.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:edbc28049ab8da07387ea2b85ceee747b36b25444ffa4aaba9f6e1461d5b2634", size = 4752692, upload-time = "2026-10-03T02:54:01.04Z" }, + { url = "https://files.pythonhosted.org/packages/b6/b8/86ec625db10e285ed8459e39ab3e56b7630554bc733e64a190e248eeea25/sqlalchemy-2.1.3-cp314-cp314t-win32.whl", hash = "sha256:724f587ced076d89291032b35476d6bdb401cce1dd680c1e3418c01d9bd55ad4", size = 2436746, upload-time = "2026-10-03T02:56:47.297Z" }, + { url = "https://files.pythonhosted.org/packages/18/be/1309b0b3acfcd6e0060c815762012c66d23eece8c9eb8091da5c850a1289/sqlalchemy-2.1.3-cp314-cp314t-win_amd64.whl", hash = "sha256:935085a5b8849d324c22bc3cc09fd07b3669cc22ba5bdd3ec208829e535cdef9", size = 2504160, upload-time = "2026-10-03T02:56:49.705Z" }, + { url = "https://files.pythonhosted.org/packages/9b/b2/ac9067f4d6c43887570df54e6c340ae9c591a8e75a5fd3029a9d7deba163/sqlalchemy-2.1.3-cp314-cp314t-win_arm64.whl", hash = "sha256:2ebfff82cece947fea1ed6337d4f0613b9799f6db621c9729790fdd41edfb5ea", size = 2422266, upload-time = "2026-10-02T23:35:30.802Z" }, + { url = "https://files.pythonhosted.org/packages/a0/e5/54a863bf8e12fa97cd06bb2e012aeca53b5535bc07aabe0192a4ed77ed44/sqlalchemy-2.1.3-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:47a052dcd935a62326216bc5f0ed10ad941c77c3acc9200385e0c03c4f31d5a9", size = 2460525, upload-time = "2026-10-03T02:38:02.872Z" }, + { url = "https://files.pythonhosted.org/packages/f9/63/f6f8a8d8e8f8be456bea70ad2b9af90a60a7f98d349b178b9eaaea86431b/sqlalchemy-2.1.3-cp315-cp315-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1e5916b43a9a8f346026f663583f203d0c43febad504ff2fed504dec6e078f20", size = 4579659, upload-time = "2026-10-02T23:35:35.789Z" }, + { url = "https://files.pythonhosted.org/packages/fa/7f/9f1064ef0f8f15e3c1cb8fa7ad5e4fc7e1e311ad0e1e46bc2e77940585bb/sqlalchemy-2.1.3-cp315-cp315-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b9af78a85af63114e2650f019ddf8fffb871e659e92bffb7d0bce5753f706caf", size = 4615129, upload-time = "2026-10-02T23:38:44.119Z" }, + { url = "https://files.pythonhosted.org/packages/c4/4a/f1a002235cc785d219738e096c7c65f68e9b7ed30c810997236567c50af9/sqlalchemy-2.1.3-cp315-cp315-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:d218698c4f05728a941d6ad7c769b6491f03058e102cf811b7e4900f78fffa07", size = 4316528, upload-time = "2026-10-02T23:39:13.275Z" }, + { url = "https://files.pythonhosted.org/packages/a8/9c/6ee4e5ab55558408c16fe1713db415e45024f42cdb1c573ebceacc09d01c/sqlalchemy-2.1.3-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:80e6ccf111bc3f57530935f5541c6c1da50ac9713c418378840683532a45dd2d", size = 4509727, upload-time = "2026-10-03T02:41:25.314Z" }, + { url = "https://files.pythonhosted.org/packages/64/8c/a6bdf2874518c628fdedf24144194afa88cceb70339e5050459cdc2723cf/sqlalchemy-2.1.3-cp315-cp315-musllinux_1_2_riscv64.whl", hash = "sha256:9a3ce1ff88dfb9f9f4305da07f4fab87720b609e264d134e47ae84a52df130bd", size = 4316490, upload-time = "2026-10-02T23:39:15.342Z" }, + { url = "https://files.pythonhosted.org/packages/10/3b/d859b638495a63ef7c155a8185369fa86ba2b860154d777e46eba922324f/sqlalchemy-2.1.3-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:77a365f7cb81bbd738c948b64e0b683f8f1b68efe51ca2815e8ad4baf6287ed9", size = 4576491, upload-time = "2026-10-02T23:38:45.97Z" }, + { url = "https://files.pythonhosted.org/packages/23/0b/c560177600613df0cfe27e27c346bac956b631cd5236178b81144e783e9e/sqlalchemy-2.1.3-cp315-cp315-win32.whl", hash = "sha256:83757320391b96cde715af0772675941d3e345da9ced3696c030f7f55ffa8eab", size = 2378116, upload-time = "2026-10-02T23:14:57.441Z" }, + { url = "https://files.pythonhosted.org/packages/d3/72/d39f60d0bc873653c0fec76aa088c969a2394ea7737f18d812fbe0e8c0c5/sqlalchemy-2.1.3-cp315-cp315-win_amd64.whl", hash = "sha256:f87533295b84c9e19ca2a72dc26a76e0192f0a2424f8fd575a82e75e7d58f9e1", size = 2428865, upload-time = "2026-10-02T23:14:58.959Z" }, + { url = "https://files.pythonhosted.org/packages/c5/03/953cf93136de44ee9d6cebee14666dcc2c4a4285fe159b2e459326631e49/sqlalchemy-2.1.3-cp315-cp315-win_arm64.whl", hash = "sha256:3e46a27b0c74af74aa6c08c983ca51db88d22dd86c316d72a634e989b7caef58", size = 2393289, upload-time = "2026-10-02T23:08:31.958Z" }, + { url = "https://files.pythonhosted.org/packages/a2/1e/5b5af9c0fe4e6f713d2ab684b88f10fd1cbe55bd34cca4d676fb930c9da3/sqlalchemy-2.1.3-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:06ef75e3e0c72d774b41f659bfc4f63d5e207d6c8418abe2b80843a4a25e2fa6", size = 2496326, upload-time = "2026-10-03T02:48:37.363Z" }, + { url = "https://files.pythonhosted.org/packages/2b/6c/6192d30fe24edc7c9e98c315341b7342ba7168d1729867c5d87e56285658/sqlalchemy-2.1.3-cp315-cp315t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:70e05ae07dd9be29d4723bd0c880df622387eb51efb46a7b8efd49295248ffd7", size = 4880284, upload-time = "2026-10-03T02:49:10.221Z" }, + { url = "https://files.pythonhosted.org/packages/40/11/a758ae0116ab663f3520970cd870da5616cbb09c369b843239cce50696d7/sqlalchemy-2.1.3-cp315-cp315t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:10fdff11782842696172489bd4861ae63c212e62ee7c4fc61fdda90961abd1f3", size = 4830599, upload-time = "2026-10-03T02:54:03.353Z" }, + { url = "https://files.pythonhosted.org/packages/8a/e8/faa500255fd90f421ef9e918d522aa4f02212b53859c35dadbafb3532bb2/sqlalchemy-2.1.3-cp315-cp315t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:edd3e6690f350c89c616e78e9b58833c3d8925607ac5faf3530f38f053f58e4d", size = 4481272, upload-time = "2026-10-02T23:59:15.592Z" }, + { url = "https://files.pythonhosted.org/packages/3c/4d/26fc9adb3fbc1d5fb8b92043a1639fd111761b8c52a2ce61df8b00ff405d/sqlalchemy-2.1.3-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:e3c165df26eca3d2612f60ea91af02648cb1bd492eadfa64dba15c9a3a0fb09c", size = 4737553, upload-time = "2026-10-03T02:49:12.413Z" }, + { url = "https://files.pythonhosted.org/packages/08/41/3639fc1d9fd7ea201ed4941215c14b87ce3c3b5f5f0190eb93321d673e48/sqlalchemy-2.1.3-cp315-cp315t-musllinux_1_2_riscv64.whl", hash = "sha256:70c073d28d043766eb1725010792f302f59b14636ffef478793045471efe92dd", size = 4479761, upload-time = "2026-10-02T23:59:17.572Z" }, + { url = "https://files.pythonhosted.org/packages/47/c8/541035b19b34dd581fb24b6fcf065db6fb4f210b9fbf93e1b2bf5f8cd5c0/sqlalchemy-2.1.3-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:bf5f63229512f14e7bffc9a0041386318f592e76bbeaf9b750b39056094d91f3", size = 4757307, upload-time = "2026-10-03T02:54:05.524Z" }, + { url = "https://files.pythonhosted.org/packages/40/a3/bb51886c02a60f1b7fb5461a25d6a3e57ce0557b916032dc3c7a72fae226/sqlalchemy-2.1.3-cp315-cp315t-win32.whl", hash = "sha256:697942762c40e31eb0fa39bf0e58e1ac1b5b1ff9022da62b74f7cdc53096a482", size = 2433533, upload-time = "2026-10-03T02:56:52.298Z" }, + { url = "https://files.pythonhosted.org/packages/e5/8e/1739b378a2b40bfc5541355c9d944b96dffcbe86aa1ed3aff31ad13f5d9b/sqlalchemy-2.1.3-cp315-cp315t-win_amd64.whl", hash = "sha256:a9c33c0d081a9d2c94229b9ae1834cb2ef81dd81253df4763e32111287b64c31", size = 2499211, upload-time = "2026-10-03T02:56:54.801Z" }, + { url = "https://files.pythonhosted.org/packages/5d/9a/ce4ea58551258203381017bc4724369cc8948ac5a46212671266b13424c4/sqlalchemy-2.1.3-cp315-cp315t-win_arm64.whl", hash = "sha256:c36ff11196dacf50312a2cc51a798bb41611331c174871f065be651a838a5361", size = 2418086, upload-time = "2026-10-03T02:44:11.611Z" }, + { url = "https://files.pythonhosted.org/packages/e5/12/23d8fb7ed90e5b674006ace415ede1c60cdfc38ace50968b59ef42c22754/sqlalchemy-2.1.3-py3-none-any.whl", hash = "sha256:6f0cf4debd86cb5623a50310784bd005c215761ca7f618d0d4f46193d033675a", size = 2052603, upload-time = "2026-10-03T02:34:19.115Z" }, +] + [[package]] name = "stack-data" version = "0.6.3"