From f1579cfd0e11aa66dc5356f64603ec3723b2e5d6 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:59:38 +0400 Subject: [PATCH 1/8] Add always-on build telemetry emitter --- .../always-on-telemetry-emitter.added.md | 1 + .../src/microcosm/build/staging.py | 16 + .../src/microcosm/build/staging_v2.py | 40 +- .../src/microcosm/build/telemetry_emitter.py | 523 +++++++++++++ .../build/telemetry_emitter_service.py | 703 ++++++++++++++++++ .../build/uk_runtime/full_build_cli.py | 67 +- .../build/uk_runtime/rowwise_staging.py | 73 +- .../microcosm/build/uk_runtime/spine_build.py | 74 +- .../shared/test_telemetry_emitter.py | 354 +++++++++ tools/build_us_fiscal_refresh_release.py | 231 +++++- 10 files changed, 2040 insertions(+), 42 deletions(-) create mode 100644 changelog.d/always-on-telemetry-emitter.added.md create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py create mode 100644 packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py diff --git a/changelog.d/always-on-telemetry-emitter.added.md b/changelog.d/always-on-telemetry-emitter.added.md new file mode 100644 index 000000000..b5f123813 --- /dev/null +++ b/changelog.d/always-on-telemetry-emitter.added.md @@ -0,0 +1 @@ +Microcosm builds now start a local telemetry emitter service that reports authenticated live progress, process-tree CPU and memory, heartbeats, US engine-batch progress, and UK graph-node progress without making collector network requests in the build process. diff --git a/packages/microcosm-build/src/microcosm/build/staging.py b/packages/microcosm-build/src/microcosm/build/staging.py index d7ebd59a3..aaf75633b 100644 --- a/packages/microcosm-build/src/microcosm/build/staging.py +++ b/packages/microcosm-build/src/microcosm/build/staging.py @@ -24,6 +24,7 @@ BestEffortUploadSession, HuggingFaceDatasetStorage, ) +from microcosm.build.telemetry_emitter import LocalTelemetryEmitter STAGING_SCHEMA_VERSION = 1 LATEST_STAGING_POINTER = "latest_staging.json" @@ -81,6 +82,7 @@ class StagingTelemetry: api: Any = None upload_interval_seconds: float = 30.0 started_at: str = field(default_factory=_now) + emitter: LocalTelemetryEmitter | None = None def __post_init__(self) -> None: self.run_dir = Path(self.run_dir) @@ -313,6 +315,13 @@ def stage( "details": details, } ) + if self.emitter is not None: + self.emitter.transition_stage( + stage, + status=status, + message=message, + **details, + ) self._maybe_upload(force=force_upload) def calibration_progress(self, event: dict[str, object]) -> None: @@ -339,6 +348,8 @@ def calibration_progress(self, event: dict[str, object]) -> None: ) self._write_progress() self._write_calibration_progress() + if self.emitter is not None: + self.emitter.transition_calibration_progress(event) self._maybe_upload() def attach_artifact( @@ -363,6 +374,7 @@ def attach_artifact( self._maybe_upload(force=force_upload) def fail(self, error: BaseException) -> None: + failed_during = str(self._progress.get("stage") or "unknown") self.stage( "failed", status="failed", @@ -371,6 +383,8 @@ def fail(self, error: BaseException) -> None: error_type=type(error).__name__, traceback=traceback.format_exc(), ) + if self.emitter is not None: + self.emitter.fail(error, failed_during=failed_during) def complete(self) -> None: self.stage( @@ -379,3 +393,5 @@ def complete(self) -> None: message="Staging run completed.", force_upload=True, ) + if self.emitter is not None: + self.emitter.complete() diff --git a/packages/microcosm-build/src/microcosm/build/staging_v2.py b/packages/microcosm-build/src/microcosm/build/staging_v2.py index 1a9ec3b34..46197a4c6 100644 --- a/packages/microcosm-build/src/microcosm/build/staging_v2.py +++ b/packages/microcosm-build/src/microcosm/build/staging_v2.py @@ -25,6 +25,7 @@ BestEffortUploadSession, HuggingFaceDatasetStorage, ) +from microcosm.build.telemetry_emitter import LocalTelemetryEmitter STAGING_CONTRACT_VERSION = 2 DEFAULT_STAGING_PREFIX = "runs" @@ -728,6 +729,7 @@ def __init__( monotonic: Callable[[], float] = time.monotonic, content_policy: StagingContentPolicy | None = None, sleep: Callable[[float], None] = time.sleep, + emitter: LocalTelemetryEmitter | None = None, ) -> None: self.run_id = _safe_identifier(run_id, label="run_id") self.candidate_id = _safe_identifier(candidate_id, label="candidate_id") @@ -765,6 +767,7 @@ def __init__( self._monotonic = monotonic self._sleep = sleep self._content_policy = content_policy or StagingContentPolicy() + self.emitter = emitter self._transport = ( HuggingFaceDatasetStorage(self.repo_id, api=api) if self.delivery_mode == "local_and_remote" and self.repo_id @@ -970,7 +973,22 @@ def fail( timestamp=self.updated_at, ) self._persist_bundle() - self._terminal_upload() + try: + self._terminal_upload() + finally: + if self.emitter is not None: + self.emitter.emit( + event_type="run", + stage_id="failed", + status="failed", + message=self.message, + details={ + "error_type": error_type, + "failure_class": error_code.lower(), + "failed_during": stage, + }, + ) + self.emitter.close() def complete(self, *, message: str = "Staging run completed.") -> None: self._require_running("complete the run") @@ -987,7 +1005,11 @@ def complete(self, *, message: str = "Staging run completed.") -> None: timestamp=self.updated_at, ) self._persist_bundle() - self._terminal_upload() + try: + self._terminal_upload() + finally: + if self.emitter is not None: + self.emitter.close() def verify_remote(self) -> None: if self.status == "running": @@ -1065,6 +1087,20 @@ def _append_event( } self._content_policy.validate_payload(event["details"]) self._events.append(validate_v2_document(event)) + if self.emitter is not None: + collector_status = { + "completed": "completed", + "failed": "failed", + "progress": "progress", + "started": "started", + }.get(status, "progress") + self.emitter.emit( + event_type=("calibration" if event_type == "calibration" else "stage"), + stage_id=stage_id, + status=collector_status, + message=message, + details=details, + ) def _run_fields(self) -> dict[str, Any]: return { diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py new file mode 100644 index 000000000..5b0227f5a --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py @@ -0,0 +1,523 @@ +"""Best-effort client for the local telemetry emitter service. + +The build process only writes small messages to a private local socket. A +separate process owns retry, authentication, resource sampling, and network +I/O so collector availability cannot delay or fail a dataset build. +""" + +from __future__ import annotations + +import json +import os +import re +import socket +import subprocess +import sys +import tempfile +import time +import uuid +from collections.abc import Mapping +from dataclasses import dataclass, field +from datetime import UTC, datetime +from importlib import metadata +from pathlib import Path +from platform import platform +from typing import Any, Literal +from urllib.parse import urlsplit + +DEFAULT_COLLECTOR_URL = "https://microcosm-telemetry-389282473430.us-central1.run.app" +COLLECTOR_URL_ENV = "MICROCOSM_TELEMETRY_COLLECTOR_URL" +_SENSITIVE_KEY_PARTS = ( + "authorization", + "credential", + "password", + "secret", + "token", + "traceback", +) +_SECRET_TEXT = re.compile( + r"(?i)(?:bearer\s+[^\s]+|(?:token|secret|password|credential)\s*[:=]\s*[^\s]+|hf_[A-Za-z0-9_-]{8,})" +) + + +def _now() -> str: + return datetime.now(UTC).isoformat() + + +def _cache_dir() -> Path: + configured = os.environ.get("XDG_CACHE_HOME", "").strip() + root = Path(configured).expanduser() if configured else Path.home() / ".cache" + return root / "microcosm" / "telemetry" + + +def _collector_url(value: str) -> str: + parsed = urlsplit(value) + local = parsed.hostname in {"localhost", "127.0.0.1", "::1"} + if not parsed.hostname or ( + parsed.scheme != "https" and not (local and parsed.scheme == "http") + ): + raise ValueError("collector URL must use HTTPS except on localhost") + if parsed.username or parsed.password or parsed.query or parsed.fragment: + raise ValueError("collector URL must not contain credentials or query data") + return value.rstrip("/") + + +def _safe_text(value: str, *, limit: int = 2_000) -> str: + return _SECRET_TEXT.sub("[redacted]", value)[:limit] + + +def _safe_json(value: Any, *, depth: int = 0) -> Any: + """Return JSON-compatible telemetry without retaining model objects.""" + + if depth >= 6: + return "[maximum depth]" + if isinstance(value, Mapping): + result = {} + for key, item in list(value.items())[:200]: + name = str(key) + if any(part in name.lower() for part in _SENSITIVE_KEY_PARTS): + result[name] = "[redacted]" + else: + result[name] = _safe_json(item, depth=depth + 1) + return result + if isinstance(value, (list, tuple)): + return [_safe_json(item, depth=depth + 1) for item in value[:200]] + if isinstance(value, Path): + return str(value) + if isinstance(value, float): + return value if value == value and abs(value) != float("inf") else None + if isinstance(value, str): + return _safe_text(value) + if isinstance(value, (int, bool)) or value is None: + return value + item = getattr(value, "item", None) + if callable(item): + return _safe_json(item()) + return _safe_text(str(value)) + + +def _safe_details(value: Mapping[str, Any]) -> dict[str, Any]: + sanitized = _safe_json(value) + assert isinstance(sanitized, dict) + if len(json.dumps(sanitized, separators=(",", ":")).encode()) <= 8_192: + return sanitized + compact: dict[str, Any] = {"telemetry_details_truncated": True} + for key, item in sanitized.items(): + if not (isinstance(item, (str, int, float, bool)) or item is None): + continue + candidate = {**compact, key: item} + if len(json.dumps(candidate, separators=(",", ":")).encode()) > 8_192: + break + compact[key] = item + return compact + + +def _run_identity() -> dict[str, Any]: + try: + commit = subprocess.run( + ["git", "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + timeout=1, + ).stdout.strip() + except (OSError, subprocess.SubprocessError): + commit = None + try: + memory_bytes = int(os.sysconf("SC_PHYS_PAGES")) * int( + os.sysconf("SC_PAGE_SIZE") + ) + except (AttributeError, OSError, ValueError): + memory_bytes = None + versions = {} + for distribution in ( + "microcosm-build", + "microcosm-graph", + "policyengine-us", + "policyengine-uk", + ): + try: + versions[distribution] = metadata.version(distribution) + except metadata.PackageNotFoundError: + continue + return { + "git_commit": commit, + "host": { + "platform": platform(), + "cpu_count": os.cpu_count(), + "memory_bytes": memory_bytes, + }, + "runtime": versions, + } + + +@dataclass(frozen=True) +class TelemetryRun: + """Identity registered with the hosted collector for one build.""" + + run_id: str + country_code: str + pipeline: str + candidate_id: str | None = None + release_id: str | None = None + run_kind: str = "build" + producer_id: str = field(default_factory=lambda: uuid.uuid4().hex) + + def as_registration(self) -> dict[str, Any]: + return { + "run_id": self.run_id, + "producer_id": self.producer_id, + "country_code": self.country_code, + "pipeline": self.pipeline, + "candidate_id": self.candidate_id, + "release_id": self.release_id, + "run_kind": self.run_kind, + } + + +class LocalTelemetryEmitter: + """Non-blocking handle owned by the instrumented build process.""" + + def __init__( + self, + *, + run: TelemetryRun, + process: subprocess.Popen[bytes] | None, + socket_path: Path | None, + runtime_dir: Path | None, + send_timeout_seconds: float = 0.2, + ) -> None: + self.run = run + self._process = process + self._socket_path = socket_path + self._runtime_dir = runtime_dir + self._send_timeout_seconds = send_timeout_seconds + self._closed = False + self._warned = False + self._transition_stage: str | None = None + + @classmethod + def start( + cls, + *, + run_id: str, + country_code: str, + pipeline: str, + candidate_id: str | None = None, + release_id: str | None = None, + run_kind: str = "build", + collector_url: str | None = None, + heartbeat_seconds: float = 60.0, + startup_timeout_seconds: float = 3.0, + spool_path: Path | str | None = None, + ) -> LocalTelemetryEmitter: + """Start the service, returning a harmless disabled handle on failure.""" + + run = TelemetryRun( + run_id=run_id, + country_code=country_code, + pipeline=pipeline, + candidate_id=candidate_id, + release_id=release_id, + run_kind=run_kind, + ) + raw_url = ( + collector_url + or os.environ.get(COLLECTOR_URL_ENV, "").strip() + or DEFAULT_COLLECTOR_URL + ) + try: + url = _collector_url(raw_url) + except ValueError as error: + print( + f"warning: Microcosm telemetry is unavailable: {error}.", + file=sys.stderr, + ) + return cls(run=run, process=None, socket_path=None, runtime_dir=None) + if not hasattr(socket, "AF_UNIX"): + print( + "warning: Microcosm telemetry is unavailable because this " + "platform has no local Unix sockets.", + file=sys.stderr, + ) + return cls(run=run, process=None, socket_path=None, runtime_dir=None) + + # macOS limits AF_UNIX paths to roughly 100 bytes. Its default + # temporary directory is already long, so use the short system alias. + temporary_root = Path("/tmp") if Path("/tmp").is_dir() else None + runtime_dir = Path( + tempfile.mkdtemp(prefix="microcosm-telemetry-", dir=temporary_root) + ) + runtime_dir.chmod(0o700) + socket_path = runtime_dir / "emitter.sock" + queue_path = Path(spool_path) if spool_path else _cache_dir() / "events.sqlite3" + command = [ + sys.executable, + "-m", + "microcosm.build.telemetry_emitter_service", + "--socket", + str(socket_path), + "--spool", + str(queue_path), + "--collector-url", + url, + "--registration-json", + json.dumps(run.as_registration(), separators=(",", ":")), + "--parent-pid", + str(os.getpid()), + "--heartbeat-seconds", + str(heartbeat_seconds), + ] + try: + process = subprocess.Popen( + command, + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + start_new_session=True, + ) + deadline = time.monotonic() + startup_timeout_seconds + while time.monotonic() < deadline: + if socket_path.exists() and _service_ready(socket_path): + emitter = cls( + run=run, + process=process, + socket_path=socket_path, + runtime_dir=runtime_dir, + ) + emitter.emit( + event_type="run", + stage_id="created", + status="started", + message="Microcosm build started.", + details={"identity": _run_identity()}, + ) + return emitter + if process.poll() is not None: + break + time.sleep(0.02) + except Exception as error: + print( + "warning: the local telemetry emitter service could not start: " + f"{type(error).__name__}: {error}", + file=sys.stderr, + ) + else: + print( + "warning: the local telemetry emitter service did not become ready; " + "the build will continue without hosted telemetry.", + file=sys.stderr, + ) + try: + socket_path.unlink(missing_ok=True) + runtime_dir.rmdir() + except OSError: + pass + return cls(run=run, process=None, socket_path=None, runtime_dir=runtime_dir) + + @property + def available(self) -> bool: + return self._socket_path is not None and not self._closed + + def emit( + self, + *, + event_type: Literal["run", "stage", "progress", "calibration", "heartbeat"], + status: Literal["started", "progress", "completed", "failed"], + stage_id: str | None = None, + message: str | None = None, + details: Mapping[str, Any] | None = None, + ) -> None: + """Queue one event locally, never raising into the build.""" + + if not self.available: + return + self._send( + { + "action": "event", + "event": { + "timestamp": _now(), + "event_type": event_type, + "stage_id": stage_id, + "status": status, + "message": _safe_text(message, limit=500) if message else None, + "details": _safe_details(details or {}), + }, + } + ) + + def stage( + self, + stage_id: str, + *, + status: Literal["started", "progress", "completed", "failed"] = "started", + message: str | None = None, + **details: Any, + ) -> None: + self.emit( + event_type="stage", + stage_id=stage_id, + status=status, + message=message, + details=details, + ) + + def transition_stage( + self, + stage_id: str, + *, + status: str = "running", + message: str | None = None, + **details: Any, + ) -> None: + """Translate a sequential stage update into explicit lifecycle events.""" + + collector_status = { + "failed": "failed", + "passed": "completed", + "running": "started", + }.get(status, "progress") + if collector_status == "started": + if self._transition_stage == stage_id: + self.stage( + stage_id, + status="progress", + message=message, + **details, + ) + return + self._close_transition_stage() + self._transition_stage = stage_id + elif self._transition_stage == stage_id: + self._transition_stage = None + else: + self._close_transition_stage() + self.stage( + stage_id, + status=collector_status, + message=message, + **details, + ) + + def progress( + self, + stage_id: str, + *, + done: int, + total: int, + unit: str | None = None, + **details: Any, + ) -> None: + self.emit( + event_type="progress", + stage_id=stage_id, + status="progress", + details={"done": done, "total": total, "unit": unit, **details}, + ) + + def calibration_progress(self, event: Mapping[str, Any]) -> None: + if event.get("kind") != "calibration_epoch": + return + self.emit( + event_type="calibration", + stage_id="calibrating", + status="progress", + details=event, + ) + + def transition_calibration_progress(self, event: Mapping[str, Any]) -> None: + """Enter the sequential calibration stage, then report one epoch.""" + + if event.get("kind") != "calibration_epoch": + return + if self._transition_stage != "calibrating": + self.transition_stage("calibrating") + self.calibration_progress(event) + + def fail( + self, + error: BaseException, + *, + failed_during: str | None = None, + failure_class: str = "build_failure", + ) -> None: + failed_stage = failed_during or self._transition_stage + self._close_transition_stage() + self.emit( + event_type="run", + stage_id="failed", + status="failed", + message=str(error)[:500], + details={ + "error_type": type(error).__name__, + "failure_class": failure_class, + "failed_during": failed_stage, + }, + ) + self.close() + + def complete(self) -> None: + self._close_transition_stage() + self.emit( + event_type="run", + stage_id="complete", + status="completed", + message="Microcosm build completed.", + ) + self.close() + + def close(self) -> None: + """Ask the service to flush in the background, then release the handle.""" + + if self._closed: + return + self._send({"action": "close"}) + self._closed = True + + def _close_transition_stage(self) -> None: + if self._transition_stage is None: + return + stage_id = self._transition_stage + self._transition_stage = None + self.stage(stage_id, status="completed") + + def _send(self, payload: Mapping[str, Any]) -> None: + if self._socket_path is None: + return + try: + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as client: + client.settimeout(self._send_timeout_seconds) + client.connect(str(self._socket_path)) + client.sendall( + json.dumps(payload, separators=(",", ":"), allow_nan=False).encode() + + b"\n" + ) + acknowledgement = client.recv(16) + if acknowledgement != b"ok\n": + raise OSError("invalid acknowledgement") + except Exception as error: + if not self._warned: + print( + "warning: the local telemetry emitter service could not queue " + f"an update ({type(error).__name__}); the build will continue.", + file=sys.stderr, + ) + self._warned = True + + +def start_local_telemetry_emitter_service( + **run: Any, +) -> LocalTelemetryEmitter: + """Named construction seam for build entrypoints and tests.""" + + return LocalTelemetryEmitter.start(**run) + + +def _service_ready(socket_path: Path) -> bool: + try: + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as client: + client.settimeout(0.1) + client.connect(str(socket_path)) + client.sendall(b'{"action":"ping"}\n') + return client.recv(16) == b"ok\n" + except OSError: + return False diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py new file mode 100644 index 000000000..1889121f8 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py @@ -0,0 +1,703 @@ +"""Standalone local service that delivers Microcosm telemetry to a collector.""" + +from __future__ import annotations + +import argparse +import json +import os +import socket +import sqlite3 +import sys +import threading +import time +import urllib.error +import urllib.request +import uuid +from collections.abc import Mapping +from datetime import UTC, datetime, timedelta +from pathlib import Path +from typing import Any + +from huggingface_hub import get_token + +try: + import psutil +except ModuleNotFoundError: # Base installs use the standard-library fallback. + psutil = None + +SCHEMA_VERSION = 1 +RETENTION_DAYS = 7 +MAX_QUEUED_BYTES = 100 * 1024 * 1024 +BATCH_SIZE = 100 + + +def _now() -> str: + return datetime.now(UTC).isoformat() + + +class _NoRedirectHandler(urllib.request.HTTPRedirectHandler): + """Keep bearer credentials on the explicitly configured origin.""" + + def redirect_request(self, req, fp, code, msg, headers, newurl): + return None + + +class EventSpool: + """Small SQLite queue shared by successive emitter service processes.""" + + def __init__(self, path: Path | str) -> None: + self.path = Path(path) + self.path.parent.mkdir(parents=True, exist_ok=True) + try: + self.path.parent.chmod(0o700) + except OSError: + pass + self._connection = sqlite3.connect( + self.path, timeout=5, check_same_thread=False + ) + self._connection.row_factory = sqlite3.Row + self._lock = threading.RLock() + self._last_prune_at = 0.0 + with self._lock, self._connection: + self._connection.execute("PRAGMA journal_mode=WAL") + self._connection.execute("PRAGMA synchronous=NORMAL") + self._connection.executescript( + """ + CREATE TABLE IF NOT EXISTS telemetry_runs ( + run_id TEXT NOT NULL, + producer_id TEXT NOT NULL, + registration_json TEXT NOT NULL, + next_sequence INTEGER NOT NULL DEFAULT 1, + updated_at TEXT NOT NULL, + PRIMARY KEY(run_id, producer_id) + ); + CREATE TABLE IF NOT EXISTS telemetry_events ( + event_id TEXT PRIMARY KEY, + run_id TEXT NOT NULL, + producer_id TEXT NOT NULL, + sequence INTEGER NOT NULL, + payload_json TEXT NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(run_id, producer_id, sequence), + FOREIGN KEY(run_id, producer_id) + REFERENCES telemetry_runs(run_id, producer_id) + ); + CREATE INDEX IF NOT EXISTS telemetry_events_run_sequence + ON telemetry_events(run_id, producer_id, sequence); + """ + ) + self.prune() + + def register(self, registration: Mapping[str, Any]) -> None: + run_id = str(registration["run_id"]) + producer_id = str(registration["producer_id"]) + encoded = json.dumps(registration, separators=(",", ":"), sort_keys=True) + with self._lock, self._connection: + existing = self._connection.execute( + """ + SELECT registration_json FROM telemetry_runs + WHERE run_id = ? AND producer_id = ? + """, + (run_id, producer_id), + ).fetchone() + if existing is not None: + old = json.loads(existing["registration_json"]) + if old.get("producer_id") != registration.get("producer_id"): + raise ValueError(f"run_id {run_id!r} already has another producer") + self._connection.execute( + """ + INSERT INTO telemetry_runs ( + run_id, producer_id, registration_json, next_sequence, updated_at + ) VALUES (?, ?, ?, 1, ?) + ON CONFLICT(run_id, producer_id) DO UPDATE SET + registration_json = excluded.registration_json, + updated_at = excluded.updated_at + """, + (run_id, producer_id, encoded, _now()), + ) + + def append( + self, + registration: Mapping[str, Any], + event: Mapping[str, Any], + *, + resources: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + run_id = str(registration["run_id"]) + producer_id = str(registration["producer_id"]) + with self._lock, self._connection: + row = self._connection.execute( + """ + SELECT next_sequence FROM telemetry_runs + WHERE run_id = ? AND producer_id = ? + """, + (run_id, producer_id), + ).fetchone() + if row is None: + raise KeyError(run_id) + sequence = int(row["next_sequence"]) + event_id = uuid.uuid4().hex + payload = { + "schema_version": SCHEMA_VERSION, + "event_id": event_id, + "run_id": run_id, + "producer_id": producer_id, + "sequence": sequence, + "timestamp": event.get("timestamp") or _now(), + "event_type": event["event_type"], + "stage_id": event.get("stage_id"), + "status": event["status"], + "message": event.get("message"), + "details": event.get("details") or {}, + "resources": resources, + } + encoded = json.dumps(payload, separators=(",", ":"), allow_nan=False) + self._connection.execute( + """ + INSERT INTO telemetry_events ( + event_id, run_id, producer_id, sequence, + payload_json, created_at + ) VALUES (?, ?, ?, ?, ?, ?) + """, + (event_id, run_id, producer_id, sequence, encoded, _now()), + ) + self._connection.execute( + """ + UPDATE telemetry_runs + SET next_sequence = ?, updated_at = ? + WHERE run_id = ? AND producer_id = ? + """, + (sequence + 1, _now(), run_id, producer_id), + ) + if time.monotonic() - self._last_prune_at >= 60: + self.prune() + return payload + + def pending_runs(self) -> list[dict[str, Any]]: + with self._lock: + rows = self._connection.execute( + """ + SELECT DISTINCT r.registration_json + FROM telemetry_runs r + JOIN telemetry_events e + ON e.run_id = r.run_id AND e.producer_id = r.producer_id + ORDER BY r.updated_at + """ + ).fetchall() + return [json.loads(row["registration_json"]) for row in rows] + + def batch( + self, run_id: str, producer_id: str, limit: int = BATCH_SIZE + ) -> list[dict[str, Any]]: + with self._lock: + rows = self._connection.execute( + """ + SELECT payload_json FROM telemetry_events + WHERE run_id = ? AND producer_id = ? + ORDER BY sequence LIMIT ? + """, + (run_id, producer_id, limit), + ).fetchall() + return [json.loads(row["payload_json"]) for row in rows] + + def acknowledge(self, event_ids: list[str]) -> None: + if not event_ids: + return + placeholders = ",".join("?" for _ in event_ids) + with self._lock, self._connection: + self._connection.execute( + f"DELETE FROM telemetry_events WHERE event_id IN ({placeholders})", + event_ids, + ) + + def has_pending(self) -> bool: + with self._lock: + row = self._connection.execute( + "SELECT 1 FROM telemetry_events LIMIT 1" + ).fetchone() + return row is not None + + def prune(self) -> None: + cutoff = (datetime.now(UTC) - timedelta(days=RETENTION_DAYS)).isoformat() + with self._lock, self._connection: + self._connection.execute( + "DELETE FROM telemetry_events WHERE created_at < ?", (cutoff,) + ) + size_row = self._connection.execute( + "SELECT COALESCE(SUM(LENGTH(payload_json)), 0) AS bytes " + "FROM telemetry_events" + ).fetchone() + excess = int(size_row["bytes"]) - MAX_QUEUED_BYTES + if excess > 0: + rows = self._connection.execute( + """ + SELECT event_id, LENGTH(payload_json) AS bytes + FROM telemetry_events ORDER BY created_at, sequence + """ + ).fetchall() + removed = 0 + ids: list[str] = [] + for row in rows: + ids.append(row["event_id"]) + removed += int(row["bytes"]) + if removed >= excess: + break + if ids: + placeholders = ",".join("?" for _ in ids) + self._connection.execute( + f"DELETE FROM telemetry_events " + f"WHERE event_id IN ({placeholders})", + ids, + ) + self._connection.execute( + """ + DELETE FROM telemetry_runs + WHERE updated_at < ? AND NOT EXISTS ( + SELECT 1 FROM telemetry_events e + WHERE e.run_id = telemetry_runs.run_id + AND e.producer_id = telemetry_runs.producer_id + ) + """, + (cutoff,), + ) + self._last_prune_at = time.monotonic() + + +def _http_post( + url: str, + payload: Mapping[str, Any], + bearer_token: str, + *, + timeout: float = 5.0, +) -> tuple[int, dict[str, Any]]: + request = urllib.request.Request( + url, + data=json.dumps(payload, separators=(",", ":")).encode(), + headers={ + "Authorization": f"Bearer {bearer_token}", + "Content-Type": "application/json", + "User-Agent": "microcosm-telemetry-emitter/1", + }, + method="POST", + ) + try: + opener = urllib.request.build_opener(_NoRedirectHandler) + with opener.open(request, timeout=timeout) as response: + body = response.read(1_048_577) + if len(body) > 1_048_576: + raise OSError("collector response exceeds 1 MiB") + try: + payload = json.loads(body) if body else {} + except (json.JSONDecodeError, UnicodeDecodeError): + payload = {} + return response.status, payload + except urllib.error.HTTPError as error: + body = error.read(1_048_577) + if len(body) > 1_048_576: + return error.code, {} + try: + detail = json.loads(body) if body else {} + except (json.JSONDecodeError, UnicodeDecodeError): + detail = {} + return error.code, detail + + +def _huggingface_token() -> str | None: + return ( + os.environ.get("HF_TOKEN", "").strip() + or os.environ.get("HUGGINGFACE_TOKEN", "").strip() + or get_token() + ) + + +class CollectorDelivery: + """Authenticate queued runs and deliver idempotent event batches.""" + + def __init__(self, collector_url: str, spool: EventSpool) -> None: + self.collector_url = collector_url.rstrip("/") + self.spool = spool + self._tokens: dict[str, tuple[str, float]] = {} + self._denied: set[str] = set() + self._warned_no_token = False + self._warned_denied: set[str] = set() + self._next_attempt_at = 0.0 + self._retry_seconds = 1.0 + + def flush_once(self) -> bool: + if time.monotonic() < self._next_attempt_at: + return False + made_progress = False + for registration in self.spool.pending_runs(): + run_id = registration["run_id"] + registration_key = f"{run_id}:{registration['producer_id']}" + if registration_key in self._denied: + continue + token = self._run_token(registration) + if token is None: + continue + events = self.spool.batch(run_id, registration["producer_id"]) + if not events: + continue + try: + status, _ = _http_post( + f"{self.collector_url}/v1/runs/{run_id}/events", + {"events": events}, + token, + ) + except (OSError, TimeoutError): + self._defer_retry() + continue + if status == 202: + self.spool.acknowledge([event["event_id"] for event in events]) + made_progress = True + self._retry_seconds = 1.0 + elif status == 401: + self._tokens.pop(registration_key, None) + elif status == 403: + self._denied.add(registration_key) + self._warn_denied(registration_key) + else: + self._defer_retry() + return made_progress + + def _run_token(self, registration: Mapping[str, Any]) -> str | None: + run_id = str(registration["run_id"]) + registration_key = f"{run_id}:{registration['producer_id']}" + cached = self._tokens.get(registration_key) + if cached is not None and cached[1] > time.monotonic() + 30: + return cached[0] + hf_token = _huggingface_token() + if not hf_token: + if not self._warned_no_token: + print( + "Microcosm telemetry is local-only: no ambient Hugging Face " + "credential was found. The dataset build will continue.", + file=sys.stderr, + flush=True, + ) + self._warned_no_token = True + self._next_attempt_at = time.monotonic() + 60.0 + return None + try: + status, response = _http_post( + f"{self.collector_url}/v1/auth/huggingface/exchange", + registration, + hf_token, + ) + except (OSError, TimeoutError): + self._defer_retry() + return None + if status == 200 and isinstance(response.get("access_token"), str): + expires_in = max(60, int(response.get("expires_in", 900))) + token = response["access_token"] + self._tokens[registration_key] = (token, time.monotonic() + expires_in) + return token + if status in {401, 403}: + self._denied.add(registration_key) + self._warn_denied(registration_key) + else: + self._defer_retry() + return None + + def _defer_retry(self) -> None: + self._next_attempt_at = time.monotonic() + self._retry_seconds + self._retry_seconds = min(60.0, self._retry_seconds * 2) + + def _warn_denied(self, run_id: str) -> None: + if run_id in self._warned_denied: + return + print( + "Microcosm telemetry is local-only for this run: the ambient " + "Hugging Face credential was not accepted as a PolicyEngine " + "organization member. The dataset build will continue.", + file=sys.stderr, + flush=True, + ) + self._warned_denied.add(run_id) + + +class ProcessTreeSampler: + """Collect cumulative CPU and resident memory for a build process tree.""" + + def __init__(self, parent_pid: int) -> None: + self.parent_pid = parent_pid + self._peak_rss = 0 + self._parent_create_time = self._create_time() + + def _create_time(self) -> float | str | None: + if psutil is not None: + try: + return psutil.Process(self.parent_pid).create_time() + except (psutil.Error, OSError): + return None + stat_path = Path(f"/proc/{self.parent_pid}/stat") + try: + return stat_path.read_text().rsplit(")", 1)[1].split()[19] + except (OSError, IndexError): + return None + + def parent_alive(self) -> bool: + if psutil is not None: + try: + process = psutil.Process(self.parent_pid) + return ( + self._parent_create_time is not None + and process.create_time() == self._parent_create_time + and process.is_running() + and process.status() != psutil.STATUS_ZOMBIE + ) + except (psutil.Error, OSError): + return False + try: + os.kill(self.parent_pid, 0) + except OSError: + return False + current = self._create_time() + # /proc supplies a process-start tick that protects against PID reuse. + # On platforms without it, existence is the best standard-library test. + return self._parent_create_time is None or current == self._parent_create_time + + def sample(self) -> dict[str, Any]: + if psutil is None: + return self._fallback_sample() + processes = [] + try: + parent = psutil.Process(self.parent_pid) + processes = [parent, *parent.children(recursive=True)] + except (psutil.Error, OSError): + pass + user = 0.0 + system = 0.0 + rss = 0 + for process in processes: + try: + cpu = process.cpu_times() + user += float(cpu.user) + system += float(cpu.system) + rss += int(process.memory_info().rss) + except (psutil.Error, OSError): + continue + self._peak_rss = max(self._peak_rss, rss) + return { + "cpu_user_seconds": user, + "cpu_system_seconds": system, + "rss_bytes": rss, + "peak_rss_bytes": self._peak_rss, + } + + def _fallback_sample(self) -> dict[str, Any]: + """Sample Linux process trees when the country engine is not installed.""" + + process_rows: dict[int, tuple[int, float, float, int]] = {} + try: + clock_ticks = float(os.sysconf("SC_CLK_TCK")) + page_size = int(os.sysconf("SC_PAGE_SIZE")) + except (AttributeError, OSError, ValueError): + return { + "cpu_user_seconds": 0.0, + "cpu_system_seconds": 0.0, + "rss_bytes": 0, + "peak_rss_bytes": self._peak_rss, + } + for stat_path in Path("/proc").glob("[0-9]*/stat"): + try: + pid = int(stat_path.parent.name) + fields = stat_path.read_text().rsplit(")", 1)[1].split() + process_rows[pid] = ( + int(fields[1]), + int(fields[11]) / clock_ticks, + int(fields[12]) / clock_ticks, + int(fields[21]) * page_size, + ) + except (OSError, ValueError, IndexError): + continue + selected = {self.parent_pid} + changed = True + while changed: + changed = False + for pid, (parent, *_rest) in process_rows.items(): + if parent in selected and pid not in selected: + selected.add(pid) + changed = True + user = sum(process_rows[pid][1] for pid in selected if pid in process_rows) + system = sum(process_rows[pid][2] for pid in selected if pid in process_rows) + rss = sum(process_rows[pid][3] for pid in selected if pid in process_rows) + self._peak_rss = max(self._peak_rss, rss) + return { + "cpu_user_seconds": user, + "cpu_system_seconds": system, + "rss_bytes": rss, + "peak_rss_bytes": self._peak_rss, + } + + +class EmitterService: + """Private local socket server with a concurrent delivery worker.""" + + def __init__( + self, + *, + socket_path: Path, + registration: Mapping[str, Any], + spool: EventSpool, + delivery: CollectorDelivery, + sampler: ProcessTreeSampler, + heartbeat_seconds: float, + drain_seconds: float = 15.0, + ) -> None: + self.socket_path = socket_path + self.registration = dict(registration) + self.spool = spool + self.delivery = delivery + self.sampler = sampler + self.heartbeat_seconds = max(1.0, heartbeat_seconds) + self.drain_seconds = max(0.0, drain_seconds) + self._stop = threading.Event() + self._closed_by_client = False + self._last_stage = "created" + + def run(self) -> None: + self.spool.register(self.registration) + self.socket_path.parent.mkdir(parents=True, exist_ok=True) + try: + self.socket_path.unlink(missing_ok=True) + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as server: + server.bind(str(self.socket_path)) + os.chmod(self.socket_path, 0o600) + server.listen(16) + server.settimeout(0.5) + worker = threading.Thread(target=self._worker, daemon=True) + worker.start() + while not self._stop.is_set(): + try: + connection, _ = server.accept() + except TimeoutError: + continue + with connection: + connection.settimeout(0.25) + try: + message = self._read_message(connection) + self._handle(message) + response = b"ok\n" + except Exception: + response = b"error\n" + try: + connection.sendall(response) + except OSError: + pass + worker.join(timeout=self.drain_seconds + 1) + finally: + self.socket_path.unlink(missing_ok=True) + try: + self.socket_path.parent.rmdir() + except OSError: + pass + + @staticmethod + def _read_message(connection: socket.socket) -> dict[str, Any]: + chunks = bytearray() + while len(chunks) <= 1_048_576: + data = connection.recv(65536) + if not data: + break + chunks.extend(data) + if b"\n" in data: + break + if len(chunks) > 1_048_576: + raise ValueError("local telemetry message exceeds 1 MiB") + return json.loads(bytes(chunks).split(b"\n", 1)[0]) + + def _handle(self, message: Mapping[str, Any]) -> None: + action = message.get("action") + if action == "event": + event = message.get("event") + if not isinstance(event, Mapping): + raise ValueError("event must be an object") + stage_id = event.get("stage_id") + if isinstance(stage_id, str) and stage_id not in {"complete", "failed"}: + self._last_stage = stage_id + self.spool.append( + self.registration, + event, + resources=self.sampler.sample(), + ) + elif action == "close": + self._closed_by_client = True + self._stop.set() + elif action == "ping": + return + else: + raise ValueError("unsupported local telemetry action") + + def _worker(self) -> None: + next_heartbeat = time.monotonic() + self.heartbeat_seconds + while not self._stop.wait(1.0): + now = time.monotonic() + if now >= next_heartbeat: + self.spool.append( + self.registration, + { + "timestamp": _now(), + "event_type": "heartbeat", + "stage_id": self._last_stage, + "status": "progress", + "message": None, + "details": {}, + }, + resources=self.sampler.sample(), + ) + next_heartbeat = now + self.heartbeat_seconds + self.delivery.flush_once() + if not self.sampler.parent_alive(): + self.spool.append( + self.registration, + { + "timestamp": _now(), + "event_type": "run", + "stage_id": "failed", + "status": "failed", + "message": "The build process exited without reporting completion.", + "details": { + "failure_class": "unexpected_process_exit", + "failed_during": self._last_stage, + }, + }, + resources=self.sampler.sample(), + ) + self._stop.set() + break + deadline = time.monotonic() + self.drain_seconds + while self.spool.has_pending() and time.monotonic() < deadline: + if not self.delivery.flush_once(): + self._stop.wait(0.5) + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser() + parser.add_argument("--socket", type=Path, required=True) + parser.add_argument("--spool", type=Path, required=True) + parser.add_argument("--collector-url", required=True) + parser.add_argument("--registration-json", required=True) + parser.add_argument("--parent-pid", type=int, required=True) + parser.add_argument("--heartbeat-seconds", type=float, default=60.0) + return parser + + +def main(argv: list[str] | None = None) -> int: + args = _parser().parse_args(argv) + registration = json.loads(args.registration_json) + spool = EventSpool(args.spool) + service = EmitterService( + socket_path=args.socket, + registration=registration, + spool=spool, + delivery=CollectorDelivery(args.collector_url, spool), + sampler=ProcessTreeSampler(args.parent_pid), + heartbeat_seconds=args.heartbeat_seconds, + ) + service.run() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py index ff684bde8..ef5abdfd9 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py @@ -174,10 +174,12 @@ from .rowwise_staging import ( STAGED_DATASET_PHASES, STAGING_UPLOAD_INTERVAL_SECONDS, + calibration_progress, create_staging_telemetry, fail_staging_telemetry, finalize_staging_telemetry, gate_statuses, + graph_progress, preflight_staged_dataset, replace_manifest, stage, @@ -212,6 +214,34 @@ ) +def _run_graph_with_progress(compiled, *, telemetry, **kwargs): + """Run a graph and emit one lightweight update per completed node.""" + + completed = 0 + previous = time.monotonic() + total = len(compiled.order) + + def observe(node_id, _population) -> None: + nonlocal completed, previous + now = time.monotonic() + completed += 1 + graph_progress( + telemetry, + node_id=node_id, + done=completed, + total=total, + elapsed_seconds=now - previous, + ) + previous = now + + return run_graph( + compiled, + _population_observer=observe, + _population_observer_detach=False, + **kwargs, + ) + + def _target_geographies(value: str) -> tuple[str, ...] | None: if value == "all": return None @@ -571,12 +601,12 @@ def _solve_observer(args: argparse.Namespace, telemetry): from .solve_progress import uk_solve_progress_callback sinks = [uk_solve_progress_callback(_stderr_progress)] - if telemetry is not None: - sinks.append( - thinned_epochs( - telemetry.calibration_progress, every=staging_epoch_every(args) - ) + sinks.append( + thinned_epochs( + lambda event: calibration_progress(telemetry, event), + every=staging_epoch_every(args), ) + ) def observer(event: dict[str, object]) -> None: for sink in sinks: @@ -1283,8 +1313,9 @@ def _execute_full_build( ): if endpoint not in node_ids: continue - checkpoint = run_graph( + checkpoint = _run_graph_with_progress( compile_graph(_through(graph, endpoint)), + telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1295,8 +1326,9 @@ def _execute_full_build( # Persist preflight outcomes before any solver can reject them. stage(telemetry, "target_compilation", "started") preflight_graph = _through(graph, "uk.full.gates.preflight") - preflight = run_graph( + preflight = _run_graph_with_progress( compile_graph(preflight_graph), + telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1331,8 +1363,9 @@ def _execute_full_build( epoch_every=staging_epoch_every(args), resumed_from_checkpoint=args.resume_size_checkpoint is not None, ) - manifest = run_graph( + manifest = _run_graph_with_progress( compile_graph(_through(graph, "uk.full.gates.calibrated")), + telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1408,8 +1441,9 @@ def _execute_full_build( if not enforcement["artifact_permitted"]: return 1 stage(telemetry, "output_bundle", "started") - manifest = run_graph( + manifest = _run_graph_with_progress( compile_graph(graph), + telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1488,8 +1522,9 @@ def _execute_full_build( spine_provenance=prepared.spine_provenance, comparison_sources=tuple(comparisons), ) - final = run_graph( + final = _run_graph_with_progress( compile_graph(graph), + telemetry=telemetry, sources={ **sources, "exported_dataset": dataset, @@ -1874,8 +1909,9 @@ def _execute_national_build( resume = args.resume # 1. The bound checkpoint: its provenance artifact is the solve's admission. stage(telemetry, "input_loading", "started") - checkpoint = run_graph( + checkpoint = _run_graph_with_progress( compile_graph(_through(graph, "uk.full.spine_checkpoint")), + telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1895,8 +1931,9 @@ def _execute_national_build( ) # 2. The register: compiled inside the graph, frozen beside the outputs. stage(telemetry, "target_compilation", "started") - targets = run_graph( + targets = _run_graph_with_progress( compile_graph(_through(graph, NATIONAL_TARGETS_NODE)), + telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1948,8 +1985,9 @@ def _execute_national_build( epochs=int(args.epochs), epoch_every=staging_epoch_every(args), ) - manifest = run_graph( + manifest = _run_graph_with_progress( compile_graph(_through(graph, NATIONAL_GATES_NODE)), + telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -2101,8 +2139,9 @@ def _execute_national_build( size_bytes=paths["dataset"].stat().st_size, ) continued = add_uk_national_readback(graph, population=national.population) - final = run_graph( + final = _run_graph_with_progress( compile_graph(continued), + telemetry=telemetry, sources={**sources, "exported_dataset": paths["dataset"]}, store=store, kernels=kernels, diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py index ccb031c11..e57fa57d0 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py @@ -35,6 +35,10 @@ StagingTelemetryV2, disabled_staging_delivery, ) +from microcosm.build.telemetry_emitter import ( + LocalTelemetryEmitter, + start_local_telemetry_emitter_service, +) from microcosm.build.uk_runtime.rowwise_cli import ( BUDGET_ITERS, MANIFEST_FILENAME, @@ -50,11 +54,13 @@ "STAGING_MAX_EPOCH_ROWS", "STAGING_UPLOAD_INTERVAL_SECONDS", "add_staging_artifact", + "calibration_progress", "create_staging_telemetry", "fail_staging_telemetry", "finalize_staging_telemetry", "fit_summary", "gate_statuses", + "graph_progress", "preflight_staged_dataset", "publish_staged_files", "replace_manifest", @@ -92,6 +98,7 @@ _STAGING_EPOCH_EVERY = STAGING_EPOCH_EVERY _STAGING_MAX_EPOCH_ROWS = STAGING_MAX_EPOCH_ROWS _STAGED_DATASET_PHASES = STAGED_DATASET_PHASES +_ACTIVE_EMITTER: LocalTelemetryEmitter | None = None def _hub_api() -> Any: @@ -186,13 +193,22 @@ def _require_write_credential(storage: HuggingFaceDatasetStorage, *, hint: str) def create_staging_telemetry( args: argparse.Namespace, *, build_id: str ) -> StagingTelemetryV2 | None: + global _ACTIVE_EMITTER + posture = posture_of(args) + run_id = args.staging_run_id or build_id + _ACTIVE_EMITTER = start_local_telemetry_emitter_service( + run_id=run_id, + country_code="GB", + pipeline=posture.pipeline, + candidate_id=args.staging_candidate_id or build_id, + run_kind="calibration", + ) if args.no_staging: return None local_only = bool(args.staging_local_only) out_dir = args.out.expanduser().resolve() - posture = posture_of(args) return StagingTelemetryV2( - run_id=args.staging_run_id or build_id, + run_id=run_id, country_code="GB", operation_id=posture.staging_operation_id, pipeline_id=posture.pipeline, @@ -204,6 +220,7 @@ def create_staging_telemetry( repo_id=None if local_only else args.staging_repo_id, upload_interval_seconds=args.staging_upload_interval_seconds, api=None if local_only else _hub_api(), + emitter=_ACTIVE_EMITTER, ) @@ -225,6 +242,12 @@ def stage( """ if telemetry is None: + if _ACTIVE_EMITTER is not None: + _ACTIVE_EMITTER.stage( + stage_id, + status=event_status, + **details, + ) return try: telemetry.stage(stage_id, event_status=event_status, **details) @@ -241,6 +264,44 @@ def stage( _stage = stage +def calibration_progress( + telemetry: StagingTelemetryV2 | None, + event: Mapping[str, Any], +) -> None: + """Forward calibration progress with or without staging artifacts.""" + + if telemetry is not None: + telemetry.calibration_progress(event) + elif _ACTIVE_EMITTER is not None: + _ACTIVE_EMITTER.transition_calibration_progress(event) + + +def graph_progress( + telemetry: StagingTelemetryV2 | None, + *, + node_id: str, + done: int, + total: int, + elapsed_seconds: float, +) -> None: + """Send graph work progress only through the local emitter service.""" + + emitter = telemetry.emitter if telemetry is not None else _ACTIVE_EMITTER + if emitter is None: + return + emitter.emit( + event_type="progress", + stage_id=node_id, + status="completed", + details={ + "done": done, + "total": total, + "unit": "graph_nodes", + "elapsed_seconds": elapsed_seconds, + }, + ) + + def stage_sample( telemetry: StagingTelemetryV2 | None, *, sample_fraction: float ) -> None: @@ -334,7 +395,11 @@ def gate_statuses(gate_report: Mapping[str, Any]) -> dict[str, str]: def fail_staging_telemetry( telemetry: StagingTelemetryV2 | None, error: BaseException ) -> None: - if telemetry is None or telemetry.status != "running": + if telemetry is None: + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.fail(error) + return + if telemetry.status != "running": return try: telemetry.fail(error) @@ -350,6 +415,8 @@ def finalize_staging_telemetry( args: argparse.Namespace, telemetry: StagingTelemetryV2 | None ) -> None: if telemetry is None: + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.complete() return try: telemetry.complete(message="UK rowwise candidate staging run completed.") diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py index 523a1f226..22ce17ea1 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py @@ -57,6 +57,10 @@ StagingTelemetryV2, disabled_staging_delivery, ) +from microcosm.build.telemetry_emitter import ( + LocalTelemetryEmitter, + start_local_telemetry_emitter_service, +) from microcosm.build.uk_runtime.age_tail import UKAgeTailStageTransform from microcosm.build.uk_runtime.battery_bindings import UK_GATE_REGISTRY from microcosm.build.uk_runtime.calibration_run import ( @@ -162,6 +166,7 @@ from microcosm.graph import ContentStore, compile_graph, run_graph _PIPELINE = "uk-frs-spine" +_ACTIVE_EMITTER: LocalTelemetryEmitter | None = None _REPOSITORY = next( ( parent @@ -1026,6 +1031,49 @@ def observe(observation: StageObservation) -> None: return observe +def _emitter_stage_observer(emitter: LocalTelemetryEmitter) -> StageObserver: + """Report graph stage observations when file staging is disabled.""" + + def observe(observation: StageObservation) -> None: + emitter.stage( + observation.stage_id, + status=observation.status, + elapsed_seconds=observation.elapsed_seconds, + entity_row_counts=dict(observation.entity_row_counts), + produced_column_count=observation.produced_column_count, + ) + + return observe + + +def _graph_progress_observer(emitter: LocalTelemetryEmitter | None, *, total: int): + """Report graph-node completion without copying the observed population.""" + + completed = 0 + previous = time.monotonic() + + def observe(node_id, _population) -> None: + nonlocal completed, previous + if emitter is None: + return + now = time.monotonic() + completed += 1 + emitter.emit( + event_type="progress", + stage_id=node_id, + status="completed", + details={ + "done": completed, + "total": total, + "unit": "graph_nodes", + "elapsed_seconds": now - previous, + }, + ) + previous = now + + return observe + + class _GraphSourceTransform: """Build a file-reading stage from only the node's declared source paths.""" @@ -1205,12 +1253,21 @@ def _exception_chain_contains(error: BaseException, text: str) -> bool: def _create_staging_telemetry( args: argparse.Namespace, *, state: AttemptState ) -> StagingTelemetryV2 | None: + global _ACTIVE_EMITTER + run_id = args.staging_run_id or state.build_id + _ACTIVE_EMITTER = start_local_telemetry_emitter_service( + run_id=run_id, + country_code="GB", + pipeline=_PIPELINE, + candidate_id=args.staging_candidate_id or state.build_id, + run_kind="smoke" if args.smoke else "spine", + ) if args.no_staging: return None local_dir = args.staging_dir or args.spine_h5.parent / "staging" local_only = args.staging_local_only return StagingTelemetryV2( - run_id=args.staging_run_id or state.build_id, + run_id=run_id, country_code="GB", operation_id="uk_frs_spine", pipeline_id=_PIPELINE, @@ -1221,6 +1278,7 @@ def _create_staging_telemetry( delivery_mode="local_only" if local_only else "local_and_remote", repo_id=None if local_only else args.staging_repo_id, upload_interval_seconds=args.staging_upload_interval_seconds, + emitter=_ACTIVE_EMITTER, ) @@ -1718,7 +1776,11 @@ def main(argv: list[str] | None = None) -> int: prepared = prepare_uk_spine_execution( args, observer=( - _staging_stage_observer(telemetry) if telemetry is not None else None + _staging_stage_observer(telemetry) + if telemetry is not None + else _emitter_stage_observer(_ACTIVE_EMITTER) + if _ACTIVE_EMITTER is not None + else None ), ) spec = prepared.spec @@ -1786,6 +1848,10 @@ def main(argv: list[str] | None = None) -> int: kernels=prepared.kernels, resume="auto", decisions=(), + _population_observer=_graph_progress_observer( + _ACTIVE_EMITTER, total=len(compiled_graph.order) + ), + _population_observer_detach=False, ) graph_manifest.save(checkpoint_root / "spine.graph.json") final_version = compiled_graph.versions[stage_names[-1]] @@ -1947,6 +2013,8 @@ def main(argv: list[str] | None = None) -> int: telemetry.validate_local_bundle() sidecar["staging_delivery"] = telemetry.delivery_summary atomic_write_json(sidecar_path, sidecar) + elif _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.complete() state.artifact_location = local_artifact_reference( output, repository_hint=_REPOSITORY, @@ -1999,6 +2067,8 @@ def main(argv: list[str] | None = None) -> int: telemetry.validate_local_bundle() except Exception: pass + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.fail(error) if _is_sampled(args) and _exception_chain_contains( error, _RUNG_NAMED_EDGE_SIGNATURE ): diff --git a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py new file mode 100644 index 000000000..96c7af4f9 --- /dev/null +++ b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py @@ -0,0 +1,354 @@ +import json +import os +import socket +import tempfile +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +import microcosm.build.telemetry_emitter_service as service_module +from microcosm.build.telemetry_emitter import ( + LocalTelemetryEmitter, + TelemetryRun, + _collector_url, + _safe_details, + _safe_json, + _safe_text, +) +from microcosm.build.telemetry_emitter_service import ( + CollectorDelivery, + EmitterService, + EventSpool, +) + + +def _registration(run_id: str = "run-a") -> dict[str, object]: + return TelemetryRun( + run_id=run_id, + country_code="US", + pipeline="us_fiscal_refresh", + candidate_id="candidate-a", + producer_id="producer-a", + ).as_registration() + + +def _event(stage_id: str = "compile_targets") -> dict[str, object]: + return { + "timestamp": "2026-10-02T10:00:00+00:00", + "event_type": "stage", + "stage_id": stage_id, + "status": "started", + "message": "Compiling targets.", + "details": {"batches": 12}, + } + + +def test_collector_url_requires_encrypted_remote_transport() -> None: + assert _collector_url( + "https://microcosm-telemetry-389282473430.us-central1.run.app/" + ) == ("https://microcosm-telemetry-389282473430.us-central1.run.app") + assert _collector_url("http://127.0.0.1:8080") == "http://127.0.0.1:8080" + try: + _collector_url("http://telemetry.example") + except ValueError as error: + assert "HTTPS" in str(error) + else: + raise AssertionError("remote HTTP collector URL was accepted") + + +def test_token_bearing_http_post_does_not_follow_redirects() -> None: + paths: list[str] = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + paths.append(self.path) + self.send_response(307) + self.send_header("Location", "/credential-leak") + self.end_headers() + + def log_message(self, format, *args): + return + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + server_thread = threading.Thread(target=server.serve_forever) + server_thread.start() + try: + status, _ = service_module._http_post( + f"http://127.0.0.1:{server.server_port}/exchange", + {"run_id": "run-a"}, + "hf-private-token", + ) + finally: + server.shutdown() + server_thread.join(timeout=2) + server.server_close() + + assert status == 307 + assert paths == ["/exchange"] + + +def test_outbound_payload_redacts_credentials_and_tracebacks() -> None: + assert _safe_text("Bearer hf_abcdefghijk") == "[redacted]" + assert _safe_json( + { + "HF_TOKEN": "hf_abcdefghijk", + "traceback": "private stack", + "message": "credential=top-secret", + } + ) == { + "HF_TOKEN": "[redacted]", + "traceback": "[redacted]", + "message": "[redacted]", + } + bounded = _safe_details( + {"done": 2, **{f"large_{index}": "x" * 2_000 for index in range(10)}} + ) + assert bounded["done"] == 2 + assert bounded["telemetry_details_truncated"] is True + assert len(json.dumps(bounded).encode()) <= 8_192 + + +def test_event_spool_assigns_stable_sequences_and_acknowledges(tmp_path) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + spool.register(registration) + + first = spool.append(registration, _event()) + second = spool.append(registration, _event("calibrate")) + + assert first["sequence"] == 1 + assert second["sequence"] == 2 + assert first["producer_id"] == "producer-a" + assert [event["sequence"] for event in spool.batch("run-a", "producer-a")] == [ + 1, + 2, + ] + + spool.acknowledge([first["event_id"]]) + assert [event["sequence"] for event in spool.batch("run-a", "producer-a")] == [2] + + +def test_sequential_stage_updates_close_the_previous_stage() -> None: + emitter = LocalTelemetryEmitter( + run=TelemetryRun( + run_id="run-a", + country_code="US", + pipeline="us_fiscal_refresh", + ), + process=None, + socket_path=Path("/unused"), + runtime_dir=None, + ) + messages: list[dict[str, object]] = [] + emitter._send = messages.append + + emitter.transition_stage("load", message="Loading.") + emitter.transition_stage("compile", message="Compiling.") + emitter.transition_stage("compile", status="passed", batches=4) + + events = [message["event"] for message in messages] + assert [(event["stage_id"], event["status"]) for event in events] == [ + ("load", "started"), + ("load", "completed"), + ("compile", "started"), + ("compile", "completed"), + ] + + +def test_event_spool_keeps_repeated_run_producers_separate(tmp_path) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + first = _registration() + second = {**first, "producer_id": "producer-b"} + spool.register(first) + spool.register(second) + + spool.append(first, _event()) + spool.append(second, _event()) + + assert spool.batch("run-a", "producer-a")[0]["sequence"] == 1 + assert spool.batch("run-a", "producer-b")[0]["sequence"] == 1 + assert len(spool.pending_runs()) == 2 + + +def test_collector_delivery_exchanges_hf_token_then_flushes( + tmp_path, monkeypatch +) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + spool.register(registration) + queued = spool.append(registration, _event()) + requests: list[tuple[str, dict[str, object], str]] = [] + + monkeypatch.setattr(service_module, "_huggingface_token", lambda: "hf-secret") + + def fake_post(url, payload, token, *, timeout=5.0): + requests.append((url, payload, token)) + if url.endswith("/v1/auth/huggingface/exchange"): + return 200, {"access_token": "run-token", "expires_in": 900} + return 202, {"accepted": 1, "duplicates": 0} + + monkeypatch.setattr(service_module, "_http_post", fake_post) + + assert CollectorDelivery("https://collector.example", spool).flush_once() + assert requests[0][2] == "hf-secret" + assert requests[1][2] == "run-token" + assert requests[1][1] == {"events": [queued]} + assert not spool.has_pending() + + +def test_non_org_credential_keeps_event_local(tmp_path, monkeypatch, capsys) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + spool.register(registration) + spool.append(registration, _event()) + monkeypatch.setattr(service_module, "_huggingface_token", lambda: "hf-outsider") + monkeypatch.setattr( + service_module, + "_http_post", + lambda *args, **kwargs: (403, {"detail": "not a member"}), + ) + delivery = CollectorDelivery("https://collector.example", spool) + + assert not delivery.flush_once() + assert not delivery.flush_once() + assert spool.has_pending() + assert capsys.readouterr().err.count("local-only for this run") == 1 + + +def test_resource_sampler_has_a_base_install_fallback(monkeypatch) -> None: + monkeypatch.setattr(service_module, "psutil", None) + sampler = service_module.ProcessTreeSampler(os.getpid()) + + assert sampler.parent_alive() + sample = sampler.sample() + assert sample.keys() == { + "cpu_user_seconds", + "cpu_system_seconds", + "rss_bytes", + "peak_rss_bytes", + } + assert all(value >= 0 for value in sample.values()) + + +class _FakeSampler: + def sample(self): + return { + "cpu_user_seconds": 1.0, + "cpu_system_seconds": 0.5, + "rss_bytes": 100, + "peak_rss_bytes": 120, + } + + def parent_alive(self): + return True + + +class _FakeDelivery: + def flush_once(self): + return False + + +def test_local_socket_acknowledges_after_durable_queue(tmp_path) -> None: + socket_path = ( + Path(tempfile.mkdtemp(prefix="microcosm-test-", dir="/tmp")) / "e.sock" + ) + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + emitter = EmitterService( + socket_path=socket_path, + registration=registration, + spool=spool, + delivery=_FakeDelivery(), + sampler=_FakeSampler(), + heartbeat_seconds=60, + drain_seconds=0, + ) + thread = threading.Thread(target=emitter.run) + thread.start() + deadline = time.monotonic() + 2 + while not socket_path.exists() and time.monotonic() < deadline: + time.sleep(0.01) + + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as client: + client.connect(str(socket_path)) + client.sendall( + json.dumps({"action": "event", "event": _event()}).encode() + b"\n" + ) + assert client.recv(16) == b"ok\n" + + assert spool.batch("run-a", "producer-a")[0]["resources"]["rss_bytes"] == 100 + + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as client: + client.connect(str(socket_path)) + client.sendall(b'{"action":"close"}\n') + assert client.recv(16) == b"ok\n" + thread.join(timeout=2) + assert not thread.is_alive() + + +def test_subprocess_exchanges_ambient_token_and_delivers_events( + tmp_path, monkeypatch +) -> None: + requests: list[tuple[str, str, dict[str, object]]] = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + length = int(self.headers.get("Content-Length", "0")) + payload = json.loads(self.rfile.read(length)) + requests.append((self.path, self.headers["Authorization"], payload)) + if self.path.endswith("/v1/auth/huggingface/exchange"): + response = {"access_token": "run-token", "expires_in": 900} + status = 200 + else: + response = { + "accepted": len(payload["events"]), + "duplicates": 0, + } + status = 202 + body = json.dumps(response).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def log_message(self, format, *args): + return + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + server_thread = threading.Thread(target=server.serve_forever) + server_thread.start() + monkeypatch.setenv("HF_TOKEN", "hf-ambient-test-token") + emitter = LocalTelemetryEmitter.start( + run_id="subprocess-run", + country_code="US", + pipeline="test-pipeline", + collector_url=f"http://127.0.0.1:{server.server_port}", + spool_path=tmp_path / "subprocess.sqlite3", + heartbeat_seconds=60, + ) + assert emitter.available + emitter.stage("compile", message="Started.") + emitter.complete() + assert emitter._process is not None + emitter._process.wait(timeout=10) + server.shutdown() + server_thread.join(timeout=2) + server.server_close() + + assert requests[0][0] == "/v1/auth/huggingface/exchange" + assert requests[0][1] == "Bearer hf-ambient-test-token" + ingestion = [request for request in requests if request[0].endswith("/events")] + assert ingestion + events = [ + event + for _, authorization, payload in ingestion + for event in payload["events"] + if authorization == "Bearer run-token" + ] + assert [event["sequence"] for event in events] == list(range(1, len(events) + 1)) + identity = events[0]["details"]["identity"] + assert identity["host"]["cpu_count"] is not None + assert "runtime" in identity + assert events[-1]["status"] == "completed" diff --git a/tools/build_us_fiscal_refresh_release.py b/tools/build_us_fiscal_refresh_release.py index 9c16a01e1..f3380b5b9 100644 --- a/tools/build_us_fiscal_refresh_release.py +++ b/tools/build_us_fiscal_refresh_release.py @@ -48,7 +48,7 @@ from datetime import UTC, datetime from pathlib import Path from types import MethodType -from typing import Any +from typing import Any, Protocol import numpy as np import pandas as pd @@ -68,6 +68,10 @@ from microcosm.build.ledger_artifact import load_ledger_consumer_artifact from microcosm.build.source_runtime import SourceRuntimeConfig, run_source_stage from microcosm.build.staging import DEFAULT_STAGING_PREFIX, StagingTelemetry +from microcosm.build.telemetry_emitter import ( + LocalTelemetryEmitter, + start_local_telemetry_emitter_service, +) from microcosm.build.us_runtime import ( ASEC_2023_WEEKS_UNEMPLOYED_SOURCE_SHA256, CONGRESSIONAL_DISTRICT_VINTAGE_CROSSWALK_SHA256_ATTR, @@ -5168,7 +5172,12 @@ def _score( parts: dict[PostExportKey, list[tuple[np.ndarray, np.ndarray]]] = { key: [] for key in keys } - for batch_frame in batches: + sweep = _WorkCounter( + stage_id="post_export_scoring", + unit="scoring_batch", + total=len(batches), + ) + for batch, batch_frame in enumerate(batches, start=1): with _automatic_gc_suspended(): simulation = self._construct(batch_frame, reform_system=reform_system) try: @@ -5194,6 +5203,7 @@ def _score( release_engine_simulation(simulation) del simulation, engine _collect_batch_garbage() + sweep.advance(pass_name=label, batch=batch, batches=len(batches)) _collect_family_garbage() return { key: _concatenate_post_export_parts(key_parts) @@ -5423,6 +5433,7 @@ def _reform_household_income_tax( reform_income_tax[household_positions] = batch_income_tax del batch_income_tax, reformed, reformed_dataset, batch_frame _collect_batch_garbage() + _advance_work(pass_name=reform_spec.measure, batch=batch, batches=len(batches)) del reform_system _collect_family_garbage() return reform_income_tax @@ -6783,6 +6794,7 @@ def _materialize_base_simulation_columns( columns[column] = pool_values del batch_columns, batch_frame _collect_batch_garbage() + _advance_work(pass_name="base", batch=batch, batches=len(batches)) _collect_family_garbage() assert column_order is not None @@ -6830,6 +6842,20 @@ def _materialize_target_frame( _assert_no_formula_owned_columns(base_frame) system = CountryTaxBenefitSystem() n_households = base_frame.n("household") + requested_reform_measures = {spec.measure for spec in target_specs} + global _ACTIVE_WORK + passes = 1 + sum( + spec.measure in requested_reform_measures + for spec in US_JCT_TAX_EXPENDITURE_REFORMS + ) + batch_count = len( + tuple(_household_position_batches(n_households, maximum_microsim_batch_size)) + ) + _ACTIVE_WORK = _WorkCounter( + stage_id="target_compilation", + unit="engine_batch", + total=passes * batch_count, + ) # The base simulation uses the JCT reform loop's household partition. base_columns, base_simulation_batching = _materialize_base_simulation_columns( base_frame, @@ -6849,7 +6875,6 @@ def _materialize_target_frame( del base_columns base_income_tax_household = hh["income_tax"].to_numpy(dtype=np.float64) - requested_reform_measures = {spec.measure for spec in target_specs} cache_context = ( dict(target_materialization_cache_context) if target_materialization_cache_context is not None @@ -6888,6 +6913,11 @@ def _materialize_target_frame( if cached is not None: reform_income_tax, cache_digest, cache_path = cached cache_stats["hits"] = int(cache_stats["hits"]) + 1 + _advance_work( + batch_count, + pass_name=reform_spec.measure, + cached=True, + ) cache_entry = { "measure": reform_spec.measure, "neutralized_variable": reform_spec.neutralized_variable, @@ -8783,7 +8813,7 @@ def _enforce_ssi_take_up_delivery( *, targets: Mapping[str, float], release_dir: Path, - telemetry: StagingTelemetry | None, + telemetry: _BuildTelemetry | None, enforcement_fences: Mapping[str, str] | None = None, ) -> tuple[list[str], GateResult]: """Fail the release on an enforced-band delivery miss, via the batch. @@ -11404,11 +11434,127 @@ def _assert_exact_k_original_pool_alignment( ) -#: The staging run for the build in flight, so the entry point can mark it -#: failed on the way out. The build body hands its telemetry object down a -#: large call stack; a module-level handle avoids threading a second copy back -#: up purely for the failure path. -_ACTIVE_TELEMETRY: StagingTelemetry | None = None +_ACTIVE_EMITTER: LocalTelemetryEmitter | None = None + + +class _BuildTelemetry(Protocol): + """Telemetry operations used by the release build body.""" + + run_id: str + repo_id: str | None + uploads_succeeded: int + + def stage( + self, + stage: str, + *, + message: str | None = None, + status: str = "running", + force_upload: bool = False, + **details: Any, + ) -> None: ... + + def calibration_progress(self, event: dict[str, object]) -> None: ... + + def attach_artifact( + self, + name: str, + path: Path | str, + **details: Any, + ) -> None: ... + + def fail(self, error: BaseException) -> None: ... + + def complete(self) -> None: ... + + +class _EmitterOnlyTelemetry: + """Expose the build telemetry interface without writing staging files.""" + + def __init__( + self, + *, + run_id: str, + emitter: LocalTelemetryEmitter, + reason: str, + ) -> None: + self.run_id = run_id + self.repo_id: str | None = None + self.uploads_succeeded = 0 + self.reason = reason + self.emitter = emitter + + def stage( + self, + stage: str, + *, + message: str | None = None, + status: str = "running", + force_upload: bool = False, + **details: Any, + ) -> None: + del force_upload + self.emitter.transition_stage( + stage, + status=status, + message=message, + **details, + ) + + def calibration_progress(self, event: dict[str, object]) -> None: + self.emitter.transition_calibration_progress(event) + + def attach_artifact( + self, + name: str, + path: Path | str, + **details: Any, + ) -> None: + del name, path, details + + def fail(self, error: BaseException) -> None: + self.emitter.fail(error) + + def complete(self) -> None: + self.emitter.complete() + + +#: The telemetry interface for the build in flight, so the entry point can +#: report failure on the way out without threading another handle through the +#: release call stack. +_ACTIVE_TELEMETRY: _BuildTelemetry | None = None + + +class _WorkCounter: + """Report completed independent work units through the emitter service.""" + + def __init__(self, *, stage_id: str, unit: str, total: int) -> None: + self.stage_id = stage_id + self.unit = unit + self.total = max(0, int(total)) + self.done = 0 + self.started = time.monotonic() + + def advance(self, units: int = 1, **details: object) -> None: + self.done = min(self.total, self.done + max(0, int(units))) + if _ACTIVE_EMITTER is None: + return + _ACTIVE_EMITTER.progress( + self.stage_id, + done=self.done, + total=self.total, + unit=self.unit, + elapsed_seconds=time.monotonic() - self.started, + **details, + ) + + +_ACTIVE_WORK: _WorkCounter | None = None + + +def _advance_work(units: int = 1, **details: object) -> None: + if _ACTIVE_WORK is not None: + _ACTIVE_WORK.advance(units, **details) class _ReleaseDryRun: @@ -12006,7 +12152,7 @@ def zero_support() -> CheckResult: return checks -def _staging_manifest_block(telemetry: StagingTelemetry | None) -> dict[str, object]: +def _staging_manifest_block(telemetry: _BuildTelemetry | None) -> dict[str, object]: """Record what staging did and distinguish an opt-out from non-delivery. Uploads are best-effort and self-disable after repeated failures, so a @@ -12015,6 +12161,8 @@ def _staging_manifest_block(telemetry: StagingTelemetry | None) -> dict[str, obj if telemetry is None: return {"enabled": False, "reason": "--no-staging"} + if isinstance(telemetry, _EmitterOnlyTelemetry): + return {"enabled": False, "reason": telemetry.reason} return { "enabled": True, "run_id": telemetry.run_id, @@ -12028,13 +12176,22 @@ def _staging_telemetry( *, release_root: Path, release_id: str, -) -> StagingTelemetry | None: + emitter: LocalTelemetryEmitter | None = None, +) -> _BuildTelemetry | None: global _ACTIVE_TELEMETRY # Each call establishes the current run, so a handle from a previous one # can never be marked failed in place of this build's. _ACTIVE_TELEMETRY = None + run_id = args.staging_run_id or release_id if args.no_staging: - return None + if emitter is None: + return None + _ACTIVE_TELEMETRY = _EmitterOnlyTelemetry( + run_id=run_id, + emitter=emitter, + reason="--no-staging", + ) + return _ACTIVE_TELEMETRY if not args.staging_dir and not args.staging_repo_id: # The parser rejects this combination, so reaching it means a caller # built the namespace directly. Returning None here would reinstate @@ -12044,7 +12201,6 @@ def _staging_telemetry( "empty and staging_dir is unset. Set no_staging to skip staging, " "or give a staging_dir for a local-only run." ) - run_id = args.staging_run_id or release_id run_dir = args.staging_dir or release_root / "staging" / "runs" / run_id _ACTIVE_TELEMETRY = StagingTelemetry( run_id=run_id, @@ -12053,6 +12209,7 @@ def _staging_telemetry( repo_id=args.staging_repo_id, path_prefix=args.staging_prefix, upload_interval_seconds=args.staging_upload_interval_seconds, + emitter=emitter, ) return _ACTIVE_TELEMETRY @@ -12068,7 +12225,7 @@ class _TerminalBatchTelemetry: def __init__( self, - telemetry: StagingTelemetry | None, + telemetry: _BuildTelemetry | None, terminal_gate_failures: list[str], ) -> None: self._telemetry = telemetry @@ -12160,6 +12317,11 @@ def main(argv: Sequence[str] | None = None) -> None: if dry_run is not None and _is_release_refusal(error): # A dry run the release refuses before its stop point: the # refusal is the report's certain failure (exit 1), not a crash. + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.fail( + error, + failure_class="dry_run_refusal", + ) raise SystemExit(dry_run.refused(error)) from error if _ACTIVE_TELEMETRY is not None: try: @@ -12171,9 +12333,15 @@ def main(argv: Sequence[str] | None = None) -> None: f"{type(telemetry_error).__name__}: {telemetry_error}", file=sys.stderr, ) + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.fail(error) raise if dry_run_exit is not None: + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.complete() raise SystemExit(dry_run_exit) + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.complete() def _check_committed_us_ledger_feed_pin( @@ -12215,6 +12383,9 @@ def _check_committed_us_ledger_feed_pin( def _main(argv: Sequence[str] | None = None) -> int | None: + global _ACTIVE_EMITTER, _ACTIVE_TELEMETRY + _ACTIVE_EMITTER = None + _ACTIVE_TELEMETRY = None args = _parse_args(argv) # --dry-run-gates-report: runs this build up to target materialization and # returns the report's exit code from the stop point below (_ReleaseDryRun). @@ -12319,6 +12490,19 @@ def _main(argv: Sequence[str] | None = None) -> int | None: build_timestamp=build_timestamp, ) _assert_us_release_id(release_id, evidence_release=args.evidence_release) + run_id = args.staging_run_id or release_id + _ACTIVE_EMITTER = start_local_telemetry_emitter_service( + run_id=run_id, + country_code="US", + pipeline="us_fiscal_refresh", + candidate_id=release_id, + release_id=release_id if dry_run is None else None, + run_kind="release" if dry_run is None else "dry_run", + ) + _ACTIVE_EMITTER.transition_stage( + "preflight", + message="Validating release inputs and configuration.", + ) if args.exact_k is not None: _assert_exact_k_release_id(release_id, args.exact_k) # The immutable release id is the dataset's exact-count name. Keep the @@ -12514,17 +12698,22 @@ def _main(argv: Sequence[str] | None = None) -> int | None: checkpoint_root.mkdir(parents=True, exist_ok=True) artifact_root.mkdir(parents=True, exist_ok=True) release_dir.mkdir(parents=True, exist_ok=True) - # A dry run writes only its report: no directories under --out, and no - # staging run that a dashboard would show as a release attempt. - telemetry = ( - _staging_telemetry( + # A dry run writes only its report under --out and creates no staging + # artifacts. Hosted telemetry identifies it separately from a release. + if dry_run is None: + telemetry = _staging_telemetry( args, release_root=release_root, release_id=release_id, + emitter=_ACTIVE_EMITTER, ) - if dry_run is None - else None - ) + else: + telemetry = _EmitterOnlyTelemetry( + run_id=run_id, + emitter=_ACTIVE_EMITTER, + reason="dry run", + ) + _ACTIVE_TELEMETRY = telemetry if telemetry is not None: telemetry.stage( "target_registry", From 36a905e53c9af65af3477f45af1388635b2501ae Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:07:20 +0400 Subject: [PATCH 2/8] Harden telemetry process accounting and startup --- .../src/microcosm/build/telemetry_emitter.py | 17 ++++ .../build/telemetry_emitter_service.py | 9 +- .../shared/test_telemetry_emitter.py | 91 +++++++++++++++++++ 3 files changed, 115 insertions(+), 2 deletions(-) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py index 5b0227f5a..89b4a53cb 100644 --- a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py @@ -268,6 +268,7 @@ def start( "--heartbeat-seconds", str(heartbeat_seconds), ] + process: subprocess.Popen[bytes] | None = None try: process = subprocess.Popen( command, @@ -307,6 +308,8 @@ def start( "the build will continue without hosted telemetry.", file=sys.stderr, ) + if process is not None: + _terminate_process(process) try: socket_path.unlink(missing_ok=True) runtime_dir.rmdir() @@ -521,3 +524,17 @@ def _service_ready(socket_path: Path) -> bool: return client.recv(16) == b"ok\n" except OSError: return False + + +def _terminate_process(process: subprocess.Popen[bytes]) -> None: + """Terminate and reap an emitter service that did not become usable.""" + + try: + if process.poll() is None: + process.terminate() + process.wait(timeout=1.0) + except subprocess.TimeoutExpired: + process.kill() + process.wait(timeout=1.0) + except OSError: + pass diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py index 1889121f8..07acf7086 100644 --- a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py @@ -472,8 +472,10 @@ def sample(self) -> dict[str, Any]: for process in processes: try: cpu = process.cpu_times() - user += float(cpu.user) - system += float(cpu.system) + user += float(cpu.user) + float(getattr(cpu, "children_user", 0.0)) + system += float(cpu.system) + float( + getattr(cpu, "children_system", 0.0) + ) rss += int(process.memory_info().rss) except (psutil.Error, OSError): continue @@ -633,6 +635,9 @@ def _worker(self) -> None: next_heartbeat = time.monotonic() + self.heartbeat_seconds while not self._stop.wait(1.0): now = time.monotonic() + # Sample every worker iteration so short-lived build children are + # much less likely to disappear between stage and heartbeat events. + self.sampler.sample() if now >= next_heartbeat: self.spool.append( self.registration, diff --git a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py index 96c7af4f9..07a485674 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py @@ -6,6 +6,7 @@ import time from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from pathlib import Path +from types import SimpleNamespace import microcosm.build.telemetry_emitter_service as service_module from microcosm.build.telemetry_emitter import ( @@ -65,6 +66,7 @@ def do_POST(self): paths.append(self.path) self.send_response(307) self.send_header("Location", "/credential-leak") + self.send_header("Content-Length", "0") self.end_headers() def log_message(self, format, *args): @@ -231,6 +233,95 @@ def test_resource_sampler_has_a_base_install_fallback(monkeypatch) -> None: assert all(value >= 0 for value in sample.values()) +def test_resource_sampler_keeps_reaped_child_cpu_monotonic(monkeypatch) -> None: + state = {"child_alive": True} + + class FakeProcess: + def __init__(self, pid): + self.pid = pid + + def create_time(self): + return float(self.pid) + + def children(self, recursive=False): + assert recursive is True + if self.pid == 10 and state["child_alive"]: + return [FakeProcess(11)] + return [] + + def cpu_times(self): + if self.pid == 10: + return SimpleNamespace( + user=10.0, + system=2.0, + children_user=100.0 if not state["child_alive"] else 0.0, + children_system=20.0 if not state["child_alive"] else 0.0, + ) + return SimpleNamespace( + user=80.0, + system=15.0, + children_user=0.0, + children_system=0.0, + ) + + def memory_info(self): + return SimpleNamespace(rss=100) + + monkeypatch.setattr( + service_module, + "psutil", + SimpleNamespace(Process=FakeProcess, Error=OSError, STATUS_ZOMBIE="zombie"), + ) + sampler = service_module.ProcessTreeSampler(10) + + before = sampler.sample() + state["child_alive"] = False + after = sampler.sample() + + assert before["cpu_user_seconds"] == 90.0 + assert before["cpu_system_seconds"] == 17.0 + assert after["cpu_user_seconds"] == 110.0 + assert after["cpu_system_seconds"] == 22.0 + + +def test_emitter_startup_timeout_terminates_and_reaps_service( + tmp_path, monkeypatch +) -> None: + class FakeProcess: + def __init__(self): + self.terminated = False + self.waited = False + + def poll(self): + return None + + def terminate(self): + self.terminated = True + + def wait(self, timeout=None): + self.waited = True + return 0 + + process = FakeProcess() + monkeypatch.setattr( + "microcosm.build.telemetry_emitter.subprocess.Popen", + lambda *args, **kwargs: process, + ) + + emitter = LocalTelemetryEmitter.start( + run_id="startup-timeout", + country_code="US", + pipeline="test-pipeline", + collector_url="https://collector.example", + spool_path=tmp_path / "events.sqlite3", + startup_timeout_seconds=0, + ) + + assert not emitter.available + assert process.terminated + assert process.waited + + class _FakeSampler: def sample(self): return { From 70570de738ccd544c87ee4c309e9e70a773aa76d Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:07:24 +0400 Subject: [PATCH 3/8] Start hosted telemetry before build preparation --- .../build/uk_runtime/full_build_cli.py | 120 ++++++++++++------ .../build/uk_runtime/rowwise_staging.py | 41 ++++-- .../microcosm/build/uk_runtime/spine_build.py | 50 ++++++-- .../tests/engine_free/uk/test_uk_frs_spine.py | 50 ++++++++ .../engine_free/uk/test_uk_full_build_cli.py | 74 +++++++++++ .../us/test_us_fiscal_refresh_builder.py | 67 ++++++++++ tools/build_us_fiscal_refresh_release.py | 54 +++++--- 7 files changed, 382 insertions(+), 74 deletions(-) diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py index ef5abdfd9..a3202cfa6 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py @@ -187,6 +187,7 @@ stage_sample, staging_delivery, staging_epoch_every, + start_telemetry_emitter, thinned_epochs, ) from .size_checkpoint import uk_size_checkpoint_identity @@ -2348,26 +2349,26 @@ def _close_national_attempt( def _national_main(args: argparse.Namespace) -> int: """The national role's envelope: the seam's attempt id and pipeline, the graph build.""" posture = posture_of(args) - national_role.require_bound_input(args) - if args.dry_run: - return national_role.national_dry_run( - args, - operation_inventory=lambda: prepare_national_build( - args - ).national.operation_inventory(), - ) - # Argument refusals cost nothing, the occupied output directory included; - # the credential check reaches the Hub, so it runs last, still before any - # input is read. - refuse_occupied_national_output(args) - preflight_staged_dataset(args) started_at = time.perf_counter() started_ts = datetime.now(UTC) - digest = preflight_digest(posture.pipeline) + build_id = new_uk_calibration_attempt_id(timestamp=started_ts) + emitter = start_telemetry_emitter( + args, + build_id=build_id, + run_kind="dry_run" if args.dry_run else "calibration", + ) + emitter.transition_stage( + "preflight", + message="Validating UK national build inputs and configuration.", + ) + try: + digest = preflight_digest(posture.pipeline) + except BaseException as error: + fail_staging_telemetry(None, error) + raise state = AttemptState( - # The attempt id is minted before telemetry opens so the staging run - # id and the Logbook row agree, as the seam minted it. - build_id=new_uk_calibration_attempt_id(timestamp=started_ts), + # The staging run id and Logbook row use the same attempt id. + build_id=build_id, identity_digest=digest, input_pins_digest=digest, phases_reached=["attempt_started"], @@ -2378,8 +2379,31 @@ def _national_main(args: argparse.Namespace) -> int: } }, ) - predecessor = resolve_predecessor(args.logbook_prev_row_digest) - telemetry = create_staging_telemetry(args, build_id=state.build_id) + if args.dry_run: + try: + status = national_role.national_dry_run( + args, + operation_inventory=lambda: prepare_national_build( + args + ).national.operation_inventory(), + ) + except BaseException as error: + fail_staging_telemetry(None, error) + raise + finalize_staging_telemetry(args, None) + return status + + try: + national_role.require_bound_input(args) + # These checks must precede input loading. The emitter is already + # running so a refusal is visible as a failed attempt. + refuse_occupied_national_output(args) + preflight_staged_dataset(args) + predecessor = resolve_predecessor(args.logbook_prev_row_digest) + telemetry = create_staging_telemetry(args, build_id=state.build_id) + except BaseException as error: + fail_staging_telemetry(None, error) + raise attempt = { "state": state, "started_at": started_at, @@ -2493,22 +2517,29 @@ def main(argv: list[str] | None = None) -> int: return _national_main(args) if args.candidate_clone_counts is not None and not args.dry_run: raise ValueError("--candidate-clone-counts is valid only with --dry-run.") - if args.dry_run: - # Dry runs plan without solving or writing and record no Logbook - # row on any path, so they need no chain configuration. - return _dry_run(args) - # Argument refusals above cost nothing; the credential check reaches the - # Hub, so it runs last, still before any input is read. - preflight_staged_dataset(args) started_at = time.perf_counter() started_ts = datetime.now(UTC) - digest = preflight_digest(posture.pipeline) + build_id = new_candidate_build_id( + seed=args.seed, + timestamp=started_ts, + rung=UK_SAMPLE_RUNG_TOKENS[args.sample_fraction], + ) + emitter = start_telemetry_emitter( + args, + build_id=build_id, + run_kind="dry_run" if args.dry_run else "calibration", + ) + emitter.transition_stage( + "preflight", + message="Validating UK dense build inputs and configuration.", + ) + try: + digest = preflight_digest(posture.pipeline) + except BaseException as error: + fail_staging_telemetry(None, error) + raise state = AttemptState( - build_id=new_candidate_build_id( - seed=args.seed, - timestamp=started_ts, - rung=UK_SAMPLE_RUNG_TOKENS[args.sample_fraction], - ), + build_id=build_id, identity_digest=digest, input_pins_digest=digest, phases_reached=["attempt_started"], @@ -2519,11 +2550,26 @@ def main(argv: list[str] | None = None) -> int: } }, ) - # Logbook chain configuration is validated before any terminal work: a - # malformed or conflicting head refuses the run with no row and no side - # effects. - predecessor = resolve_predecessor(args.logbook_prev_row_digest) - telemetry = create_staging_telemetry(args, build_id=state.build_id) + if args.dry_run: + # Dry runs plan without solving or writing and record no Logbook row, + # but their hosted telemetry still has a complete lifecycle. + try: + status = _dry_run(args) + except BaseException as error: + fail_staging_telemetry(None, error) + raise + finalize_staging_telemetry(args, None) + return status + + try: + # The credential and chain checks still precede input loading, while + # the emitter records their failures. + preflight_staged_dataset(args) + predecessor = resolve_predecessor(args.logbook_prev_row_digest) + telemetry = create_staging_telemetry(args, build_id=state.build_id) + except BaseException as error: + fail_staging_telemetry(None, error) + raise attempt = { "state": state, "started_at": started_at, diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py index e57fa57d0..b8894cd45 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py @@ -64,6 +64,7 @@ "preflight_staged_dataset", "publish_staged_files", "replace_manifest", + "start_telemetry_emitter", "stage", "stage_dataset", "staged_dataset_mode", @@ -193,16 +194,9 @@ def _require_write_credential(storage: HuggingFaceDatasetStorage, *, hint: str) def create_staging_telemetry( args: argparse.Namespace, *, build_id: str ) -> StagingTelemetryV2 | None: - global _ACTIVE_EMITTER posture = posture_of(args) run_id = args.staging_run_id or build_id - _ACTIVE_EMITTER = start_local_telemetry_emitter_service( - run_id=run_id, - country_code="GB", - pipeline=posture.pipeline, - candidate_id=args.staging_candidate_id or build_id, - run_kind="calibration", - ) + emitter = start_telemetry_emitter(args, build_id=build_id) if args.no_staging: return None local_only = bool(args.staging_local_only) @@ -220,8 +214,37 @@ def create_staging_telemetry( repo_id=None if local_only else args.staging_repo_id, upload_interval_seconds=args.staging_upload_interval_seconds, api=None if local_only else _hub_api(), - emitter=_ACTIVE_EMITTER, + emitter=emitter, + ) + + +def start_telemetry_emitter( + args: argparse.Namespace, + *, + build_id: str, + run_kind: str = "calibration", +) -> LocalTelemetryEmitter: + """Start or reuse the process-local emitter for one UK build attempt.""" + + global _ACTIVE_EMITTER + posture = posture_of(args) + run_id = args.staging_run_id or build_id + if ( + _ACTIVE_EMITTER is not None + and _ACTIVE_EMITTER.available + and _ACTIVE_EMITTER.run.run_id == run_id + ): + return _ACTIVE_EMITTER + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.close() + _ACTIVE_EMITTER = start_local_telemetry_emitter_service( + run_id=run_id, + country_code="GB", + pipeline=posture.pipeline, + candidate_id=args.staging_candidate_id or build_id, + run_kind=run_kind, ) + return _ACTIVE_EMITTER _create_staging_telemetry = create_staging_telemetry diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py index 22ce17ea1..87c8a5f21 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py @@ -1251,17 +1251,14 @@ def _exception_chain_contains(error: BaseException, text: str) -> bool: def _create_staging_telemetry( - args: argparse.Namespace, *, state: AttemptState + args: argparse.Namespace, + *, + state: AttemptState, + emitter: LocalTelemetryEmitter | None = None, ) -> StagingTelemetryV2 | None: global _ACTIVE_EMITTER run_id = args.staging_run_id or state.build_id - _ACTIVE_EMITTER = start_local_telemetry_emitter_service( - run_id=run_id, - country_code="GB", - pipeline=_PIPELINE, - candidate_id=args.staging_candidate_id or state.build_id, - run_kind="smoke" if args.smoke else "spine", - ) + _ACTIVE_EMITTER = emitter or _start_telemetry_emitter(args, state=state) if args.no_staging: return None local_dir = args.staging_dir or args.spine_h5.parent / "staging" @@ -1282,6 +1279,29 @@ def _create_staging_telemetry( ) +def _start_telemetry_emitter( + args: argparse.Namespace, *, state: AttemptState +) -> LocalTelemetryEmitter: + global _ACTIVE_EMITTER + run_id = args.staging_run_id or state.build_id + if ( + _ACTIVE_EMITTER is not None + and _ACTIVE_EMITTER.available + and _ACTIVE_EMITTER.run.run_id == run_id + ): + return _ACTIVE_EMITTER + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.close() + _ACTIVE_EMITTER = start_local_telemetry_emitter_service( + run_id=run_id, + country_code="GB", + pipeline=_PIPELINE, + candidate_id=args.staging_candidate_id or state.build_id, + run_kind="smoke" if args.smoke else "spine", + ) + return _ACTIVE_EMITTER + + def _telemetry_sample( args: argparse.Namespace, sampling: Mapping[str, object] | None ) -> dict[str, object] | None: @@ -1724,12 +1744,11 @@ def main(argv: list[str] | None = None) -> int: rung = UK_SAMPLE_RUNG_TOKENS[args.sample_fraction] started_at = time.perf_counter() started_ts = datetime.now(UTC) - predecessor = resolve_predecessor(args.logbook_prev_row_digest) - digest = preflight_digest(_PIPELINE) + predecessor = None state = AttemptState( build_id=_new_build_id(started_ts), - identity_digest=digest, - input_pins_digest=digest, + identity_digest="unresolved-preflight-digest", + input_pins_digest="unresolved-preflight-digest", phases_reached=["attempt_started"], gate_verdicts={ "pipeline": { @@ -1742,7 +1761,12 @@ def main(argv: list[str] | None = None) -> int: spool_dir = args.spine_h5.parent / "logbook-spool" telemetry: StagingTelemetryV2 | None = None spine_battery: GateBatteryRun | None = None + emitter = _start_telemetry_emitter(args, state=state) try: + predecessor = resolve_predecessor(args.logbook_prev_row_digest) + digest = preflight_digest(_PIPELINE) + state.identity_digest = digest + state.input_pins_digest = digest _validate_args(args) # A crash between the H5 write and the sidecar writes must never # leave a stale sidecar beside a fresh H5 (adversarial-review @@ -1760,7 +1784,7 @@ def main(argv: list[str] | None = None) -> int: stale_outputs.append(args.emit_nonzero_shares) for stale in stale_outputs: stale.unlink(missing_ok=True) - telemetry = _create_staging_telemetry(args, state=state) + telemetry = _create_staging_telemetry(args, state=state, emitter=emitter) if telemetry is not None: initial_sample = _telemetry_sample(args, None) if initial_sample is not None: diff --git a/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py b/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py index ce94fbd8d..cf5e58c33 100644 --- a/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py +++ b/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py @@ -1588,6 +1588,56 @@ def test_driver_refuses_missing_spi_tab(tmp_path: Path) -> None: tool._validate_args(args) +def test_driver_reports_validation_failure_through_early_emitter( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + tool = _load_tool() + raw_dir = tmp_path / "raw" + raw_dir.mkdir() + hmrc_ods = tmp_path / "Collated_Tables_3_1_to_3_11_2324.ods" + hmrc_ods.write_text("synthetic\n", encoding="utf-8") + events = [] + + class FakeEmitter: + available = True + + def __init__(self, run_id: str) -> None: + self.run = SimpleNamespace(run_id=run_id) + + def close(self) -> None: + events.append(("close",)) + + def fail(self, error: BaseException) -> None: + events.append(("failed", str(error))) + + def start_emitter(**run): + events.append(("started", run["run_id"])) + return FakeEmitter(run["run_id"]) + + monkeypatch.setattr(tool, "start_local_telemetry_emitter_service", start_emitter) + monkeypatch.setattr(tool, "preflight_digest", lambda pipeline: "0" * 64) + monkeypatch.delenv("POPULACE_LOGBOOK_PREV_ROW_DIGEST", raising=False) + + status = tool.main( + [ + "--frs-raw-dir", + str(raw_dir), + "--spine-h5", + str(tmp_path / "spine.h5"), + "--spi-tab", + str(tmp_path / "missing-put2223uk.tab"), + "--hmrc-ods", + str(hmrc_ods), + "--no-staging", + ] + ) + + assert status == 1 + assert events[0][0] == "started" + assert events[1][0] == "failed" + assert "--spi-tab must be an existing file" in events[1][1] + + def test_driver_refuses_misnamed_spi_tab(tmp_path: Path) -> None: tool = _load_tool() raw_dir = tmp_path / "raw" diff --git a/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py b/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py index 96a19b1e0..9bf43e60c 100644 --- a/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py +++ b/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py @@ -500,6 +500,80 @@ def test_dry_run_has_no_files_or_kernel_execution(tmp_path, monkeypatch, capsys) assert not args.out.exists() +def test_dense_dry_run_starts_and_finishes_hosted_telemetry(tmp_path, monkeypatch): + args = arguments(tmp_path, "--dry-run") + events = [] + + class FakeEmitter: + def transition_stage(self, stage_id, **details): + events.append(("stage", stage_id, details)) + + monkeypatch.setattr(cli, "parse_args", lambda argv: args) + monkeypatch.setattr( + cli, + "start_telemetry_emitter", + lambda requested, *, build_id, run_kind: ( + events.append(("start", build_id, run_kind)) or FakeEmitter() + ), + ) + monkeypatch.setattr( + cli, "_dry_run", lambda requested: events.append(("plan",)) or 0 + ) + monkeypatch.setattr( + cli, + "finalize_staging_telemetry", + lambda requested, telemetry: events.append(("complete", telemetry)), + ) + + assert cli.main([]) == 0 + assert [event[0] for event in events] == ["start", "stage", "plan", "complete"] + assert events[0][2] == "dry_run" + + +def test_dense_preflight_failure_is_reported_by_early_emitter(tmp_path, monkeypatch): + args = arguments(tmp_path) + events = [] + + class FakeEmitter: + def transition_stage(self, stage_id, **details): + events.append(("stage", stage_id, details)) + + monkeypatch.setattr(cli, "parse_args", lambda argv: args) + monkeypatch.setattr( + cli, + "start_telemetry_emitter", + lambda requested, *, build_id, run_kind: ( + events.append(("start", build_id, run_kind)) or FakeEmitter() + ), + ) + + def refuse_preflight(requested): + events.append(("preflight",)) + raise RuntimeError("staging credential unavailable") + + monkeypatch.setattr(cli, "preflight_staged_dataset", refuse_preflight) + monkeypatch.setattr( + cli, + "fail_staging_telemetry", + lambda telemetry, error: events.append(("failed", telemetry, str(error))), + ) + monkeypatch.setattr( + cli, + "create_staging_telemetry", + lambda *args, **kwargs: pytest.fail("preflight failure reached staging setup"), + ) + + with pytest.raises(RuntimeError, match="staging credential unavailable"): + cli.main([]) + assert [event[0] for event in events] == [ + "start", + "stage", + "preflight", + "failed", + ] + assert events[-1][1] is None + + def test_rejected_output_inside_source_never_writes_failure_sidecar( tmp_path, monkeypatch ): diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py b/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py index fe38277f8..33624246c 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py @@ -6,6 +6,7 @@ import os import sys from dataclasses import replace +from datetime import UTC, datetime from pathlib import Path from types import SimpleNamespace @@ -94,6 +95,72 @@ def test_release_and_fiscal_scorer_signatures_have_no_membership_switches() -> N } +def test_telemetry_attempt_id_is_available_before_release_inputs_are_loaded() -> None: + builder = _load_builder_module() + timestamp = datetime(2026, 10, 5, 12, 30, tzinfo=UTC) + + assert ( + builder._telemetry_run_id( + SimpleNamespace(staging_run_id="staged-attempt", release_id="release"), + timestamp=timestamp, + ) + == "staged-attempt" + ) + assert ( + builder._telemetry_run_id( + SimpleNamespace(staging_run_id=None, release_id="release"), + timestamp=timestamp, + ) + == "release" + ) + generated = builder._telemetry_run_id( + SimpleNamespace(staging_run_id=None, release_id=None), + timestamp=timestamp, + ) + assert generated.startswith("populace-us-build-20261005T123000Z-") + assert len(generated.rsplit("-", 1)[-1]) == 8 + + +def test_us_emitter_starts_before_dirty_worktree_refusal(monkeypatch) -> None: + builder = _load_builder_module() + calls = [] + + class FakeEmitter: + available = True + + def transition_stage(self, stage_id, **details): + calls.append(("stage", stage_id, details)) + + def fail(self, error, **details): + calls.append(("failed", type(error).__name__, details)) + + monkeypatch.setattr( + builder, + "_parse_args", + lambda argv: SimpleNamespace( + staging_run_id="observed-run", + release_id="release-id", + dry_run_gates_report=None, + ), + ) + monkeypatch.setattr(builder._ReleaseDryRun, "start", lambda args, argv: None) + monkeypatch.setattr(builder, "_git_dirty", lambda: True) + + def start_emitter(**run): + calls.append(("started", run)) + return FakeEmitter() + + monkeypatch.setattr(builder, "start_local_telemetry_emitter_service", start_emitter) + + with pytest.raises(SystemExit, match="dirty git worktree"): + builder.main([]) + + assert calls[0][0] == "started" + assert calls[0][1]["run_id"] == "observed-run" + assert calls[1][0:2] == ("stage", "preflight") + assert calls[2][0:2] == ("failed", "SystemExit") + + @pytest.mark.parametrize( "removed_option", ( diff --git a/tools/build_us_fiscal_refresh_release.py b/tools/build_us_fiscal_refresh_release.py index f3380b5b9..6729b34c8 100644 --- a/tools/build_us_fiscal_refresh_release.py +++ b/tools/build_us_fiscal_refresh_release.py @@ -43,6 +43,7 @@ import sys import time import tomllib +import uuid from collections.abc import Callable, Iterable, Mapping, Sequence from contextlib import contextmanager, nullcontext from datetime import UTC, datetime @@ -11241,6 +11242,25 @@ def _default_release_id( return f"populace-us-2024-{digest}-{commit}-{build_timestamp:%Y%m%dT%H%M%SZ}" +def _telemetry_run_id(args: argparse.Namespace, *, timestamp: datetime) -> str: + """Choose an attempt id before release inputs have been loaded. + + An explicit staging id remains authoritative, and an explicit release id + is already stable enough to identify the attempt. Builds that derive their + release id from the input digest need a separate attempt id so telemetry can + start before downloading or hashing that input. + """ + + if args.staging_run_id: + return str(args.staging_run_id) + if args.release_id: + return str(args.release_id) + instant = timestamp.astimezone(UTC) + return ( + f"populace-us-build-{instant.strftime('%Y%m%dT%H%M%SZ')}-{uuid.uuid4().hex[:8]}" + ) + + def _assert_us_release_id(release_id: str, *, evidence_release: bool = False) -> None: if not release_id.startswith("populace-us-"): raise ValueError( @@ -12176,13 +12196,14 @@ def _staging_telemetry( *, release_root: Path, release_id: str, + run_id: str | None = None, emitter: LocalTelemetryEmitter | None = None, ) -> _BuildTelemetry | None: global _ACTIVE_TELEMETRY # Each call establishes the current run, so a handle from a previous one # can never be marked failed in place of this build's. _ACTIVE_TELEMETRY = None - run_id = args.staging_run_id or release_id + run_id = run_id or args.staging_run_id or release_id if args.no_staging: if emitter is None: return None @@ -12387,6 +12408,22 @@ def _main(argv: Sequence[str] | None = None) -> int | None: _ACTIVE_EMITTER = None _ACTIVE_TELEMETRY = None args = _parse_args(argv) + build_started = time.perf_counter() + attempt_started_at = datetime.now(UTC) + run_id = _telemetry_run_id(args, timestamp=attempt_started_at) + is_dry_run = args.dry_run_gates_report is not None + _ACTIVE_EMITTER = start_local_telemetry_emitter_service( + run_id=run_id, + country_code="US", + pipeline="us_fiscal_refresh", + candidate_id=args.release_id, + release_id=args.release_id if not is_dry_run else None, + run_kind="dry_run" if is_dry_run else "release", + ) + _ACTIVE_EMITTER.transition_stage( + "preflight", + message="Validating release inputs and configuration.", + ) # --dry-run-gates-report: runs this build up to target materialization and # returns the report's exit code from the stop point below (_ReleaseDryRun). dry_run = _ReleaseDryRun.start(args, argv) @@ -12401,7 +12438,6 @@ def _main(argv: Sequence[str] | None = None) -> int | None: if args.evidence_release else () ) - build_started = time.perf_counter() timing: dict[str, float] = {} if args.release_id: @@ -12490,19 +12526,6 @@ def _main(argv: Sequence[str] | None = None) -> int | None: build_timestamp=build_timestamp, ) _assert_us_release_id(release_id, evidence_release=args.evidence_release) - run_id = args.staging_run_id or release_id - _ACTIVE_EMITTER = start_local_telemetry_emitter_service( - run_id=run_id, - country_code="US", - pipeline="us_fiscal_refresh", - candidate_id=release_id, - release_id=release_id if dry_run is None else None, - run_kind="release" if dry_run is None else "dry_run", - ) - _ACTIVE_EMITTER.transition_stage( - "preflight", - message="Validating release inputs and configuration.", - ) if args.exact_k is not None: _assert_exact_k_release_id(release_id, args.exact_k) # The immutable release id is the dataset's exact-count name. Keep the @@ -12705,6 +12728,7 @@ def _main(argv: Sequence[str] | None = None) -> int | None: args, release_root=release_root, release_id=release_id, + run_id=run_id, emitter=_ACTIVE_EMITTER, ) else: From fe96eb71ad4e14bf9667750f3cb89a6e42cad62e Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Mon, 5 Oct 2026 22:58:22 +0400 Subject: [PATCH 4/8] Make telemetry authentication automatic and fail closed --- .../always-on-telemetry-emitter.added.md | 2 +- .../src/microcosm/build/telemetry_emitter.py | 34 +--- .../build/telemetry_emitter_service.py | 188 +++++++++++++++--- .../shared/test_telemetry_emitter.py | 172 +++++++++++++--- 4 files changed, 313 insertions(+), 83 deletions(-) diff --git a/changelog.d/always-on-telemetry-emitter.added.md b/changelog.d/always-on-telemetry-emitter.added.md index b5f123813..f7b2c8cc1 100644 --- a/changelog.d/always-on-telemetry-emitter.added.md +++ b/changelog.d/always-on-telemetry-emitter.added.md @@ -1 +1 @@ -Microcosm builds now start a local telemetry emitter service that reports authenticated live progress, process-tree CPU and memory, heartbeats, US engine-batch progress, and UK graph-node progress without making collector network requests in the build process. +Microcosm builds now start a local telemetry emitter service that reports authenticated live progress, process-tree CPU and memory, heartbeats, US engine-batch progress, and UK graph-node progress without making collector network requests in the build process. PolicyEngine members use their existing Hugging Face login automatically, while builds without an accepted organization credential remain local-only. diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py index 89b4a53cb..20f639563 100644 --- a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py @@ -23,10 +23,7 @@ from pathlib import Path from platform import platform from typing import Any, Literal -from urllib.parse import urlsplit -DEFAULT_COLLECTOR_URL = "https://microcosm-telemetry-389282473430.us-central1.run.app" -COLLECTOR_URL_ENV = "MICROCOSM_TELEMETRY_COLLECTOR_URL" _SENSITIVE_KEY_PARTS = ( "authorization", "credential", @@ -50,18 +47,6 @@ def _cache_dir() -> Path: return root / "microcosm" / "telemetry" -def _collector_url(value: str) -> str: - parsed = urlsplit(value) - local = parsed.hostname in {"localhost", "127.0.0.1", "::1"} - if not parsed.hostname or ( - parsed.scheme != "https" and not (local and parsed.scheme == "http") - ): - raise ValueError("collector URL must use HTTPS except on localhost") - if parsed.username or parsed.password or parsed.query or parsed.fragment: - raise ValueError("collector URL must not contain credentials or query data") - return value.rstrip("/") - - def _safe_text(value: str, *, limit: int = 2_000) -> str: return _SECRET_TEXT.sub("[redacted]", value)[:limit] @@ -206,7 +191,7 @@ def start( candidate_id: str | None = None, release_id: str | None = None, run_kind: str = "build", - collector_url: str | None = None, + development_collector_url: str | None = None, heartbeat_seconds: float = 60.0, startup_timeout_seconds: float = 3.0, spool_path: Path | str | None = None, @@ -221,19 +206,6 @@ def start( release_id=release_id, run_kind=run_kind, ) - raw_url = ( - collector_url - or os.environ.get(COLLECTOR_URL_ENV, "").strip() - or DEFAULT_COLLECTOR_URL - ) - try: - url = _collector_url(raw_url) - except ValueError as error: - print( - f"warning: Microcosm telemetry is unavailable: {error}.", - file=sys.stderr, - ) - return cls(run=run, process=None, socket_path=None, runtime_dir=None) if not hasattr(socket, "AF_UNIX"): print( "warning: Microcosm telemetry is unavailable because this " @@ -259,8 +231,6 @@ def start( str(socket_path), "--spool", str(queue_path), - "--collector-url", - url, "--registration-json", json.dumps(run.as_registration(), separators=(",", ":")), "--parent-pid", @@ -268,6 +238,8 @@ def start( "--heartbeat-seconds", str(heartbeat_seconds), ] + if development_collector_url is not None: + command.extend(["--development-collector-url", development_collector_url]) process: subprocess.Popen[bytes] | None = None try: process = subprocess.Popen( diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py index 07acf7086..05be6b9d5 100644 --- a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py @@ -17,6 +17,7 @@ from datetime import UTC, datetime, timedelta from pathlib import Path from typing import Any +from urllib.parse import urlsplit from huggingface_hub import get_token @@ -29,6 +30,9 @@ RETENTION_DAYS = 7 MAX_QUEUED_BYTES = 100 * 1024 * 1024 BATCH_SIZE = 100 +PRODUCTION_COLLECTOR_URL = ( + "https://microcosm-telemetry-389282473430.us-central1.run.app" +) def _now() -> str: @@ -42,6 +46,34 @@ def redirect_request(self, req, fp, code, msg, headers, newurl): return None +def _collector_origin(value: str, *, allow_loopback_http: bool = False) -> str: + parsed = urlsplit(value) + loopback = parsed.hostname in {"localhost", "127.0.0.1", "::1"} + valid_scheme = parsed.scheme == "https" or ( + allow_loopback_http and loopback and parsed.scheme == "http" + ) + if not parsed.hostname or not valid_scheme: + raise ValueError("collector URL must be an HTTPS origin") + if ( + parsed.username + or parsed.password + or parsed.query + or parsed.fragment + or parsed.path not in {"", "/"} + ): + raise ValueError( + "collector URL must be an origin without credentials or path data" + ) + return value.rstrip("/") + + +def _development_collector_url(value: str) -> str: + parsed = urlsplit(value) + if parsed.hostname not in {"localhost", "127.0.0.1", "::1"}: + raise ValueError("development collector URL must use a loopback address") + return _collector_origin(value, allow_loopback_http=True) + + class EventSpool: """Small SQLite queue shared by successive emitter service processes.""" @@ -68,6 +100,8 @@ def __init__(self, path: Path | str) -> None: producer_id TEXT NOT NULL, registration_json TEXT NOT NULL, next_sequence INTEGER NOT NULL DEFAULT 1, + upload_state TEXT NOT NULL DEFAULT 'pending', + local_only_reason TEXT, updated_at TEXT NOT NULL, PRIMARY KEY(run_id, producer_id) ); @@ -86,6 +120,29 @@ def __init__(self, path: Path | str) -> None: ON telemetry_events(run_id, producer_id, sequence); """ ) + columns = { + row["name"] + for row in self._connection.execute( + "PRAGMA table_info(telemetry_runs)" + ).fetchall() + } + if "upload_state" not in columns: + self._connection.execute( + "ALTER TABLE telemetry_runs ADD COLUMN " + "upload_state TEXT NOT NULL DEFAULT 'pending'" + ) + self._connection.execute( + "UPDATE telemetry_runs SET upload_state = 'local_only'" + ) + if "local_only_reason" not in columns: + self._connection.execute( + "ALTER TABLE telemetry_runs ADD COLUMN local_only_reason TEXT" + ) + self._connection.execute( + "UPDATE telemetry_runs SET local_only_reason = " + "'created_before_upload_eligibility' " + "WHERE upload_state = 'local_only'" + ) self.prune() def register(self, registration: Mapping[str, Any]) -> None: @@ -181,11 +238,44 @@ def pending_runs(self) -> list[dict[str, Any]]: FROM telemetry_runs r JOIN telemetry_events e ON e.run_id = r.run_id AND e.producer_id = r.producer_id + WHERE r.upload_state = 'pending' ORDER BY r.updated_at """ ).fetchall() return [json.loads(row["registration_json"]) for row in rows] + def make_local_only( + self, + run_id: str, + producer_id: str, + reason: str, + ) -> None: + """Permanently exclude one producer's queued events from upload.""" + + with self._lock, self._connection: + self._connection.execute( + """ + UPDATE telemetry_runs + SET upload_state = 'local_only', local_only_reason = ?, updated_at = ? + WHERE run_id = ? AND producer_id = ? + """, + (reason, _now(), run_id, producer_id), + ) + + def has_deliverable(self) -> bool: + with self._lock: + row = self._connection.execute( + """ + SELECT 1 + FROM telemetry_events e + JOIN telemetry_runs r + ON r.run_id = e.run_id AND r.producer_id = e.producer_id + WHERE r.upload_state = 'pending' + LIMIT 1 + """ + ).fetchone() + return row is not None + def batch( self, run_id: str, producer_id: str, limit: int = BATCH_SIZE ) -> list[dict[str, Any]]: @@ -313,11 +403,20 @@ def _huggingface_token() -> str | None: class CollectorDelivery: """Authenticate queued runs and deliver idempotent event batches.""" - def __init__(self, collector_url: str, spool: EventSpool) -> None: - self.collector_url = collector_url.rstrip("/") + def __init__( + self, + spool: EventSpool, + *, + development_collector_url: str | None = None, + ) -> None: + self.collector_url = ( + _development_collector_url(development_collector_url) + if development_collector_url is not None + else _collector_origin(PRODUCTION_COLLECTOR_URL) + ) self.spool = spool - self._tokens: dict[str, tuple[str, float]] = {} - self._denied: set[str] = set() + self._session_token: tuple[str, float] | None = None + self._registered: set[str] = set() self._warned_no_token = False self._warned_denied: set[str] = set() self._next_attempt_at = 0.0 @@ -328,13 +427,14 @@ def flush_once(self) -> bool: return False made_progress = False for registration in self.spool.pending_runs(): - run_id = registration["run_id"] + run_id = str(registration["run_id"]) registration_key = f"{run_id}:{registration['producer_id']}" - if registration_key in self._denied: - continue - token = self._run_token(registration) + token = self._collector_token(registration) if token is None: continue + if registration_key not in self._registered: + if not self._register(registration, token): + continue events = self.spool.batch(run_id, registration["producer_id"]) if not events: continue @@ -352,20 +452,19 @@ def flush_once(self) -> bool: made_progress = True self._retry_seconds = 1.0 elif status == 401: - self._tokens.pop(registration_key, None) + self._session_token = None elif status == 403: - self._denied.add(registration_key) - self._warn_denied(registration_key) + self._make_local_only(registration, "collector_authorization_rejected") else: self._defer_retry() return made_progress - def _run_token(self, registration: Mapping[str, Any]) -> str | None: - run_id = str(registration["run_id"]) - registration_key = f"{run_id}:{registration['producer_id']}" - cached = self._tokens.get(registration_key) - if cached is not None and cached[1] > time.monotonic() + 30: - return cached[0] + def _collector_token(self, registration: Mapping[str, Any]) -> str | None: + if ( + self._session_token is not None + and self._session_token[1] > time.monotonic() + 30 + ): + return self._session_token[0] hf_token = _huggingface_token() if not hf_token: if not self._warned_no_token: @@ -376,29 +475,63 @@ def _run_token(self, registration: Mapping[str, Any]) -> str | None: flush=True, ) self._warned_no_token = True - self._next_attempt_at = time.monotonic() + 60.0 + self._make_local_only(registration, "missing_huggingface_credential") return None try: status, response = _http_post( f"{self.collector_url}/v1/auth/huggingface/exchange", - registration, + {}, hf_token, ) except (OSError, TimeoutError): self._defer_retry() return None if status == 200 and isinstance(response.get("access_token"), str): - expires_in = max(60, int(response.get("expires_in", 900))) + expires_in = max(60, int(response.get("expires_in", 3600))) token = response["access_token"] - self._tokens[registration_key] = (token, time.monotonic() + expires_in) + self._session_token = (token, time.monotonic() + expires_in) return token if status in {401, 403}: - self._denied.add(registration_key) - self._warn_denied(registration_key) + self._make_local_only(registration, "huggingface_credential_rejected") else: self._defer_retry() return None + def _register(self, registration: Mapping[str, Any], token: str) -> bool: + registration_key = f"{registration['run_id']}:{registration['producer_id']}" + try: + status, _ = _http_post( + f"{self.collector_url}/v1/runs", + registration, + token, + ) + except (OSError, TimeoutError): + self._defer_retry() + return False + if status == 201: + self._registered.add(registration_key) + self._retry_seconds = 1.0 + return True + if status == 401: + self._session_token = None + elif status in {403, 409}: + self._make_local_only(registration, "run_registration_rejected") + else: + self._defer_retry() + return False + + def _make_local_only( + self, + registration: Mapping[str, Any], + reason: str, + ) -> None: + run_id = str(registration["run_id"]) + producer_id = str(registration["producer_id"]) + registration_key = f"{run_id}:{producer_id}" + self.spool.make_local_only(run_id, producer_id, reason) + if reason != "missing_huggingface_credential": + self._warn_denied(registration_key) + def _defer_retry(self) -> None: self._next_attempt_at = time.monotonic() + self._retry_seconds self._retry_seconds = min(60.0, self._retry_seconds * 2) @@ -672,7 +805,7 @@ def _worker(self) -> None: self._stop.set() break deadline = time.monotonic() + self.drain_seconds - while self.spool.has_pending() and time.monotonic() < deadline: + while self.spool.has_deliverable() and time.monotonic() < deadline: if not self.delivery.flush_once(): self._stop.wait(0.5) @@ -681,7 +814,7 @@ def _parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser() parser.add_argument("--socket", type=Path, required=True) parser.add_argument("--spool", type=Path, required=True) - parser.add_argument("--collector-url", required=True) + parser.add_argument("--development-collector-url") parser.add_argument("--registration-json", required=True) parser.add_argument("--parent-pid", type=int, required=True) parser.add_argument("--heartbeat-seconds", type=float, default=60.0) @@ -696,7 +829,10 @@ def main(argv: list[str] | None = None) -> int: socket_path=args.socket, registration=registration, spool=spool, - delivery=CollectorDelivery(args.collector_url, spool), + delivery=CollectorDelivery( + spool, + development_collector_url=args.development_collector_url, + ), sampler=ProcessTreeSampler(args.parent_pid), heartbeat_seconds=args.heartbeat_seconds, ) diff --git a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py index 07a485674..52b4d094c 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py @@ -1,6 +1,7 @@ import json import os import socket +import sqlite3 import tempfile import threading import time @@ -12,7 +13,6 @@ from microcosm.build.telemetry_emitter import ( LocalTelemetryEmitter, TelemetryRun, - _collector_url, _safe_details, _safe_json, _safe_text, @@ -45,17 +45,17 @@ def _event(stage_id: str = "compile_targets") -> dict[str, object]: } -def test_collector_url_requires_encrypted_remote_transport() -> None: - assert _collector_url( - "https://microcosm-telemetry-389282473430.us-central1.run.app/" - ) == ("https://microcosm-telemetry-389282473430.us-central1.run.app") - assert _collector_url("http://127.0.0.1:8080") == "http://127.0.0.1:8080" - try: - _collector_url("http://telemetry.example") - except ValueError as error: - assert "HTTPS" in str(error) - else: - raise AssertionError("remote HTTP collector URL was accepted") +def test_development_collector_must_be_on_loopback() -> None: + assert service_module._development_collector_url("http://127.0.0.1:8080") == ( + "http://127.0.0.1:8080" + ) + for value in ("https://collector.example", "http://192.0.2.1:8080"): + try: + service_module._development_collector_url(value) + except ValueError as error: + assert "loopback" in str(error) + else: + raise AssertionError("non-loopback development collector was accepted") def test_token_bearing_http_post_does_not_follow_redirects() -> None: @@ -173,6 +173,59 @@ def test_event_spool_keeps_repeated_run_producers_separate(tmp_path) -> None: assert len(spool.pending_runs()) == 2 +def test_pre_eligibility_spool_is_not_uploaded_after_upgrade(tmp_path) -> None: + path = tmp_path / "events.sqlite3" + registration = _registration() + connection = sqlite3.connect(path) + connection.executescript( + """ + CREATE TABLE telemetry_runs ( + run_id TEXT NOT NULL, + producer_id TEXT NOT NULL, + registration_json TEXT NOT NULL, + next_sequence INTEGER NOT NULL DEFAULT 1, + updated_at TEXT NOT NULL, + PRIMARY KEY(run_id, producer_id) + ); + CREATE TABLE telemetry_events ( + event_id TEXT PRIMARY KEY, + run_id TEXT NOT NULL, + producer_id TEXT NOT NULL, + sequence INTEGER NOT NULL, + payload_json TEXT NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(run_id, producer_id, sequence) + ); + """ + ) + connection.execute( + "INSERT INTO telemetry_runs VALUES (?, ?, ?, 2, ?)", + ( + "run-a", + "producer-a", + json.dumps(registration), + "2026-10-02T10:00:00+00:00", + ), + ) + connection.execute( + "INSERT INTO telemetry_events VALUES (?, ?, ?, 1, ?, ?)", + ( + "old-event", + "run-a", + "producer-a", + json.dumps({"event_id": "old-event"}), + "2026-10-02T10:00:00+00:00", + ), + ) + connection.commit() + connection.close() + + spool = EventSpool(path) + + assert spool.has_pending() + assert spool.pending_runs() == [] + + def test_collector_delivery_exchanges_hf_token_then_flushes( tmp_path, monkeypatch ) -> None: @@ -187,15 +240,27 @@ def test_collector_delivery_exchanges_hf_token_then_flushes( def fake_post(url, payload, token, *, timeout=5.0): requests.append((url, payload, token)) if url.endswith("/v1/auth/huggingface/exchange"): - return 200, {"access_token": "run-token", "expires_in": 900} + return 200, {"access_token": "collector-token", "expires_in": 3600} + if url.endswith("/v1/runs"): + return 201, {"registered": True} return 202, {"accepted": 1, "duplicates": 0} monkeypatch.setattr(service_module, "_http_post", fake_post) - assert CollectorDelivery("https://collector.example", spool).flush_once() + delivery = CollectorDelivery( + spool, + development_collector_url="http://127.0.0.1:8080", + ) + assert delivery.flush_once() assert requests[0][2] == "hf-secret" - assert requests[1][2] == "run-token" - assert requests[1][1] == {"events": [queued]} + assert requests[0][1] == {} + assert requests[1] == ( + "http://127.0.0.1:8080/v1/runs", + registration, + "collector-token", + ) + assert requests[2][2] == "collector-token" + assert requests[2][1] == {"events": [queued]} assert not spool.has_pending() @@ -204,20 +269,71 @@ def test_non_org_credential_keeps_event_local(tmp_path, monkeypatch, capsys) -> registration = _registration() spool.register(registration) spool.append(registration, _event()) + requests: list[dict[str, object]] = [] monkeypatch.setattr(service_module, "_huggingface_token", lambda: "hf-outsider") - monkeypatch.setattr( - service_module, - "_http_post", - lambda *args, **kwargs: (403, {"detail": "not a member"}), + + def reject(url, payload, token, *, timeout=5.0): + requests.append(payload) + return 403, {"detail": "not a member"} + + monkeypatch.setattr(service_module, "_http_post", reject) + delivery = CollectorDelivery( + spool, + development_collector_url="http://127.0.0.1:8080", ) - delivery = CollectorDelivery("https://collector.example", spool) assert not delivery.flush_once() assert not delivery.flush_once() assert spool.has_pending() + assert spool.pending_runs() == [] + assert requests == [{}] assert capsys.readouterr().err.count("local-only for this run") == 1 +def test_identity_provider_outage_keeps_events_eligible_for_retry( + tmp_path, monkeypatch +) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + spool.register(registration) + spool.append(registration, _event()) + monkeypatch.setattr(service_module, "_huggingface_token", lambda: "hf-member") + monkeypatch.setattr( + service_module, + "_http_post", + lambda *args, **kwargs: (503, {"detail": "temporarily unavailable"}), + ) + + assert not CollectorDelivery(spool).flush_once() + assert spool.pending_runs() == [registration] + + +def test_missing_credential_never_contacts_collector_and_stays_local_only( + tmp_path, monkeypatch +) -> None: + spool_path = tmp_path / "events.sqlite3" + spool = EventSpool(spool_path) + registration = _registration() + spool.register(registration) + spool.append(registration, _event()) + monkeypatch.setattr(service_module, "_huggingface_token", lambda: None) + monkeypatch.setattr( + service_module, + "_http_post", + lambda *args, **kwargs: (_ for _ in ()).throw( + AssertionError("network request") + ), + ) + + delivery = CollectorDelivery(spool) + assert not delivery.flush_once() + assert spool.pending_runs() == [] + + reopened = EventSpool(spool_path) + monkeypatch.setattr(service_module, "_huggingface_token", lambda: "hf-later") + assert reopened.pending_runs() == [] + + def test_resource_sampler_has_a_base_install_fallback(monkeypatch) -> None: monkeypatch.setattr(service_module, "psutil", None) sampler = service_module.ProcessTreeSampler(os.getpid()) @@ -312,7 +428,7 @@ def wait(self, timeout=None): run_id="startup-timeout", country_code="US", pipeline="test-pipeline", - collector_url="https://collector.example", + development_collector_url="http://127.0.0.1:8080", spool_path=tmp_path / "events.sqlite3", startup_timeout_seconds=0, ) @@ -389,8 +505,11 @@ def do_POST(self): payload = json.loads(self.rfile.read(length)) requests.append((self.path, self.headers["Authorization"], payload)) if self.path.endswith("/v1/auth/huggingface/exchange"): - response = {"access_token": "run-token", "expires_in": 900} + response = {"access_token": "collector-token", "expires_in": 3600} status = 200 + elif self.path == "/v1/runs": + response = {"registered": True} + status = 201 else: response = { "accepted": len(payload["events"]), @@ -415,7 +534,7 @@ def log_message(self, format, *args): run_id="subprocess-run", country_code="US", pipeline="test-pipeline", - collector_url=f"http://127.0.0.1:{server.server_port}", + development_collector_url=f"http://127.0.0.1:{server.server_port}", spool_path=tmp_path / "subprocess.sqlite3", heartbeat_seconds=60, ) @@ -430,13 +549,16 @@ def log_message(self, format, *args): assert requests[0][0] == "/v1/auth/huggingface/exchange" assert requests[0][1] == "Bearer hf-ambient-test-token" + assert requests[0][2] == {} + assert requests[1][0] == "/v1/runs" + assert requests[1][1] == "Bearer collector-token" ingestion = [request for request in requests if request[0].endswith("/events")] assert ingestion events = [ event for _, authorization, payload in ingestion for event in payload["events"] - if authorization == "Bearer run-token" + if authorization == "Bearer collector-token" ] assert [event["sequence"] for event in events] == list(range(1, len(events) + 1)) identity = events[0]["details"]["identity"] From 28dd14331d541bc6c05e2554216895986dedf2cb Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Mon, 5 Oct 2026 23:02:49 +0400 Subject: [PATCH 5/8] Test the fixed production telemetry destination --- .../engine_free/shared/test_telemetry_emitter.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py index 52b4d094c..a1ba74b76 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py @@ -58,6 +58,19 @@ def test_development_collector_must_be_on_loopback() -> None: raise AssertionError("non-loopback development collector was accepted") +def test_production_collector_cannot_be_replaced_by_environment( + tmp_path, monkeypatch +) -> None: + monkeypatch.setenv( + "MICROCOSM_TELEMETRY_COLLECTOR_URL", + "https://untrusted.example", + ) + + delivery = CollectorDelivery(EventSpool(tmp_path / "events.sqlite3")) + + assert delivery.collector_url == service_module.PRODUCTION_COLLECTOR_URL + + def test_token_bearing_http_post_does_not_follow_redirects() -> None: paths: list[str] = [] From 1b96d01c99e4279de5d753824ff4bb66590182ca Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Wed, 7 Oct 2026 18:50:08 +0400 Subject: [PATCH 6/8] Separate hosted telemetry from staging run bundles --- README.md | 34 +-- .../always-on-telemetry-emitter.added.md | 2 +- .../src/microcosm/build/__init__.py | 4 +- .../src/microcosm/build/staging.py | 20 +- .../src/microcosm/build/staging_cli.py | 21 +- .../src/microcosm/build/staging_v2.py | 47 +--- .../src/microcosm/build/telemetry_emitter.py | 3 + .../build/uk_runtime/full_build_cli.py | 71 +++--- .../build/uk_runtime/national_role.py | 4 +- .../build/uk_runtime/rowwise_staging.py | 184 ++++++++-------- .../microcosm/build/uk_runtime/spine_build.py | 208 +++++++++--------- .../tests/engine_free/shared/test_staging.py | 29 ++- .../engine_free/shared/test_staging_v2.py | 11 +- .../shared/test_telemetry_emitter.py | 2 +- .../tests/engine_free/uk/test_uk_frs_spine.py | 27 ++- .../engine_free/uk/test_uk_full_build_cli.py | 46 +++- .../us/test_us_exact_k_ladder_launcher.py | 20 +- .../us/test_us_fiscal_refresh_builder.py | 122 +++++++--- tools/build_us_fiscal_refresh_release.py | 169 +++++++------- tools/generate_staging_contract_fixtures.py | 8 +- 20 files changed, 568 insertions(+), 464 deletions(-) diff --git a/README.md b/README.md index e584a6a02..2539cf930 100644 --- a/README.md +++ b/README.md @@ -44,16 +44,23 @@ until you restart the kernel, load it by path with `load_country_spec(Path(...))` (always re-read), or call `microcosm.build.country_spec._load_packaged_country_spec.cache_clear()`. -## Staging build telemetry - -US fiscal refresh builds emit pre-release staging telemetry **by default**: -progress JSON is uploaded to `policyengine/populace-us-staging` while the build -runs (best-effort — a missing token or failed upload never fails the build), so -every candidate shows up on the staging dashboard before it is published. -Disable with `--no-staging`, or point elsewhere with `--staging-repo-id` / -`POPULACE_STAGING_REPO_ID`. An *empty* `POPULACE_STAGING_REPO_ID` is ignored -rather than read as off, and staging with no destination at all is an argparse -error — `--no-staging` is the only way a build produces no telemetry. +## Build progress and staging run files + +Supported US and UK build commands always start the local telemetry emitter +service. It reports live progress to the hosted collector when the operator's +existing Hugging Face login is accepted; otherwise it retains the events +locally and the build continues. This live event path is independent of the +staging run files described below. + +US fiscal refresh builds also write pre-release staging run files **by +default**. Progress JSON is uploaded to `policyengine/populace-us-staging` +while the build runs (best-effort — a missing token or failed upload never +fails the build), so every candidate shows up on the staging dashboard before +it is published. Disable these files with `--no-staging`, or point them +elsewhere with `--staging-repo-id` / `POPULACE_STAGING_REPO_ID`. An *empty* +`POPULACE_STAGING_REPO_ID` is ignored rather than read as off, and staging with +no destination at all is an argparse error. `--no-staging` does not disable +the local telemetry emitter service. The build manifest records what staging did: the run id, the destination, and how many files actually reached it, or an explicit `enabled: false` for a @@ -75,8 +82,8 @@ The UK commands (`tools/build_uk_frs_spine.py`, a shim over the package's `uk_runtime.spine_build`, and `microcosm-build-uk` / `tools/build_uk_full.py`, whose `--release-role` builds either the national or the dense line; `tools/build_uk_rowwise_candidate.py` is a stub over the same driver) stage -version 2 telemetry to `policyengine/populace-uk-staging` under the same -switch. The build command also **stages the finished dataset bundle** it built, +version 2 staging run files to `policyengine/populace-uk-staging` under the +same switch. The build command also **stages the finished dataset bundle** it built, national, dense or exact-count, under `staged//` in the private `policyengine/populace-uk-private` repository so the team can inspect it without publishing it: `releases/` and `latest.json` are untouched, the @@ -203,7 +210,8 @@ weights, calling the release tool's own gate functions (none is re-implemented), and exits `1` on any certain failure, `2` on AT-RISK only, and `0` when clean. An argparse error also exits `2` but writes no report, so the wrapper returns `64` whenever no report was written. It writes only its -report: nothing under `--out`, no staging telemetry, no receipts. A base or +report: nothing under `--out`, no staging run files, no receipts. The local +telemetry emitter service still reports dry-run progress. A base or donor that the config does not name locally is still downloaded, into the same caches the release uses. A refusal before the stop point becomes the report's certain failure. A crash while grading is reported as the dry run's own error, diff --git a/changelog.d/always-on-telemetry-emitter.added.md b/changelog.d/always-on-telemetry-emitter.added.md index f7b2c8cc1..392d39d3b 100644 --- a/changelog.d/always-on-telemetry-emitter.added.md +++ b/changelog.d/always-on-telemetry-emitter.added.md @@ -1 +1 @@ -Microcosm builds now start a local telemetry emitter service that reports authenticated live progress, process-tree CPU and memory, heartbeats, US engine-batch progress, and UK graph-node progress without making collector network requests in the build process. PolicyEngine members use their existing Hugging Face login automatically, while builds without an accepted organization credential remain local-only. +Microcosm builds now start a local telemetry emitter service that reports authenticated live progress, process-tree CPU and memory, heartbeats, US engine-batch progress, and UK graph-node progress without making collector network requests in the build process. PolicyEngine members use their existing Hugging Face login automatically, while builds without an accepted organization credential remain local-only. The staging JSON writers are now named as run-bundle writers and operate independently from the hosted event emitter. diff --git a/packages/microcosm-build/src/microcosm/build/__init__.py b/packages/microcosm-build/src/microcosm/build/__init__.py index 1a883d3db..b3cb80eaa 100644 --- a/packages/microcosm-build/src/microcosm/build/__init__.py +++ b/packages/microcosm-build/src/microcosm/build/__init__.py @@ -190,7 +190,7 @@ def _assert_frame_compatible(version: str, required: tuple[int, int]) -> None: LATEST_STAGING_POINTER, RUNS_INDEX, STAGING_SCHEMA_VERSION, - StagingTelemetry, + StagingRunBundleWriter, ) __version__ = "0.1.0" @@ -233,7 +233,7 @@ def _assert_frame_compatible(version: str, required: tuple[int, int]) -> None: "LATEST_STAGING_POINTER", "RUNS_INDEX", "STAGING_SCHEMA_VERSION", - "StagingTelemetry", + "StagingRunBundleWriter", "TargetCoverageRequirement", "TargetFitRequirement", "ACCEPTED_CONSUMER_ARTIFACT_SCHEMA_VERSIONS", diff --git a/packages/microcosm-build/src/microcosm/build/staging.py b/packages/microcosm-build/src/microcosm/build/staging.py index aaf75633b..58146c159 100644 --- a/packages/microcosm-build/src/microcosm/build/staging.py +++ b/packages/microcosm-build/src/microcosm/build/staging.py @@ -24,7 +24,6 @@ BestEffortUploadSession, HuggingFaceDatasetStorage, ) -from microcosm.build.telemetry_emitter import LocalTelemetryEmitter STAGING_SCHEMA_VERSION = 1 LATEST_STAGING_POINTER = "latest_staging.json" @@ -59,8 +58,8 @@ def _write_json(path: Path, payload: dict[str, Any]) -> None: @dataclass -class StagingTelemetry: - """Write and optionally upload build-run telemetry. +class StagingRunBundleWriter: + """Write and optionally upload a build's staging run bundle. Args: run_id: Stable id for this build attempt. @@ -82,7 +81,6 @@ class StagingTelemetry: api: Any = None upload_interval_seconds: float = 30.0 started_at: str = field(default_factory=_now) - emitter: LocalTelemetryEmitter | None = None def __post_init__(self) -> None: self.run_dir = Path(self.run_dir) @@ -315,13 +313,6 @@ def stage( "details": details, } ) - if self.emitter is not None: - self.emitter.transition_stage( - stage, - status=status, - message=message, - **details, - ) self._maybe_upload(force=force_upload) def calibration_progress(self, event: dict[str, object]) -> None: @@ -348,8 +339,6 @@ def calibration_progress(self, event: dict[str, object]) -> None: ) self._write_progress() self._write_calibration_progress() - if self.emitter is not None: - self.emitter.transition_calibration_progress(event) self._maybe_upload() def attach_artifact( @@ -374,7 +363,6 @@ def attach_artifact( self._maybe_upload(force=force_upload) def fail(self, error: BaseException) -> None: - failed_during = str(self._progress.get("stage") or "unknown") self.stage( "failed", status="failed", @@ -383,8 +371,6 @@ def fail(self, error: BaseException) -> None: error_type=type(error).__name__, traceback=traceback.format_exc(), ) - if self.emitter is not None: - self.emitter.fail(error, failed_during=failed_during) def complete(self) -> None: self.stage( @@ -393,5 +379,3 @@ def complete(self) -> None: message="Staging run completed.", force_upload=True, ) - if self.emitter is not None: - self.emitter.complete() diff --git a/packages/microcosm-build/src/microcosm/build/staging_cli.py b/packages/microcosm-build/src/microcosm/build/staging_cli.py index 633d0160b..f4e6983a1 100644 --- a/packages/microcosm-build/src/microcosm/build/staging_cli.py +++ b/packages/microcosm-build/src/microcosm/build/staging_cli.py @@ -1,4 +1,4 @@ -"""Country-neutral command-line options for staging telemetry.""" +"""Country-neutral command-line options for staging run bundles.""" from __future__ import annotations @@ -27,9 +27,9 @@ def add_staging_arguments( ``default_upload_interval_seconds`` lets a long-running command choose a slower best-effort upload cadence: the Hub allows about 128 commits per - hour per repository and every telemetry cycle is up to eight single-file - commits, so a multi-hour solve at the 30-second default exhausts the - budget and loses uploads (the UK rowwise driver runs at 300). + hour per repository and each run-bundle upload cycle performs up to eight + single-file commits, so a multi-hour solve at the 30-second default + exhausts the budget and loses uploads (the UK rowwise driver runs at 300). """ parser.add_argument( @@ -42,7 +42,7 @@ def add_staging_arguments( default=repository.repo_id(os.environ), help=( "Access-controlled Hugging Face dataset repository for best-effort " - "telemetry delivery." + "staging run-bundle delivery." ), ) parser.add_argument( @@ -68,7 +68,10 @@ def add_staging_arguments( mode.add_argument( "--no-staging", action="store_true", - help="Deliberately disable staging telemetry for this build.", + help=( + "Disable the staging run bundle for this build; hosted telemetry " + "remains active." + ), ) parser.add_argument( "--staging-read-back", @@ -107,8 +110,8 @@ def add_staged_dataset_arguments( The finished bundle follows the staging mode switch: ``--no-staging`` keeps nothing, ``--staging-local-only`` keeps the bundle and its sidecars on disk, and the default uploads it to ``repository`` under - ``staged//``. ``--no-staged-dataset`` runs telemetry alone: the - bundle is neither inventoried nor uploaded. + ``staged//``. ``--no-staged-dataset`` retains only the staging run + files: the dataset bundle is neither inventoried nor uploaded. """ parser.add_argument( @@ -125,7 +128,7 @@ def add_staged_dataset_arguments( "--no-staged-dataset", action="store_true", help=( - "Run staging telemetry alone: the finished bundle is neither " + "Keep only the staging run files: the finished bundle is neither " "inventoried nor uploaded (--no-staging already disables both)." ), ) diff --git a/packages/microcosm-build/src/microcosm/build/staging_v2.py b/packages/microcosm-build/src/microcosm/build/staging_v2.py index 46197a4c6..105c668b6 100644 --- a/packages/microcosm-build/src/microcosm/build/staging_v2.py +++ b/packages/microcosm-build/src/microcosm/build/staging_v2.py @@ -25,7 +25,6 @@ BestEffortUploadSession, HuggingFaceDatasetStorage, ) -from microcosm.build.telemetry_emitter import LocalTelemetryEmitter STAGING_CONTRACT_VERSION = 2 DEFAULT_STAGING_PREFIX = "runs" @@ -706,8 +705,8 @@ def _reject_record_collections(self, value: Any) -> None: self._reject_record_collections(item) -class StagingTelemetryV2: - """Record, validate, persist, and optionally upload telemetry version 2.""" +class StagingRunBundleWriterV2: + """Record, validate, persist, and optionally upload a version 2 run bundle.""" def __init__( self, @@ -729,7 +728,6 @@ def __init__( monotonic: Callable[[], float] = time.monotonic, content_policy: StagingContentPolicy | None = None, sleep: Callable[[float], None] = time.sleep, - emitter: LocalTelemetryEmitter | None = None, ) -> None: self.run_id = _safe_identifier(run_id, label="run_id") self.candidate_id = _safe_identifier(candidate_id, label="candidate_id") @@ -745,7 +743,8 @@ def __init__( raise StagingContractError("pipeline_version must be non-empty.") if delivery_mode == "disabled": raise StagingContractError( - "Do not construct telemetry for disabled staging; record an opt-out." + "Do not construct a staging run bundle when staging is disabled; " + "record an opt-out." ) if delivery_mode == "local_and_remote" and not (repo_id or "").strip(): raise StagingContractError( @@ -767,7 +766,6 @@ def __init__( self._monotonic = monotonic self._sleep = sleep self._content_policy = content_policy or StagingContentPolicy() - self.emitter = emitter self._transport = ( HuggingFaceDatasetStorage(self.repo_id, api=api) if self.delivery_mode == "local_and_remote" and self.repo_id @@ -973,22 +971,7 @@ def fail( timestamp=self.updated_at, ) self._persist_bundle() - try: - self._terminal_upload() - finally: - if self.emitter is not None: - self.emitter.emit( - event_type="run", - stage_id="failed", - status="failed", - message=self.message, - details={ - "error_type": error_type, - "failure_class": error_code.lower(), - "failed_during": stage, - }, - ) - self.emitter.close() + self._terminal_upload() def complete(self, *, message: str = "Staging run completed.") -> None: self._require_running("complete the run") @@ -1005,11 +988,7 @@ def complete(self, *, message: str = "Staging run completed.") -> None: timestamp=self.updated_at, ) self._persist_bundle() - try: - self._terminal_upload() - finally: - if self.emitter is not None: - self.emitter.close() + self._terminal_upload() def verify_remote(self) -> None: if self.status == "running": @@ -1087,20 +1066,6 @@ def _append_event( } self._content_policy.validate_payload(event["details"]) self._events.append(validate_v2_document(event)) - if self.emitter is not None: - collector_status = { - "completed": "completed", - "failed": "failed", - "progress": "progress", - "started": "started", - }.get(status, "progress") - self.emitter.emit( - event_type=("calibration" if event_type == "calibration" else "stage"), - stage_id=stage_id, - status=collector_status, - message=message, - details=details, - ) def _run_fields(self) -> dict[str, Any]: return { diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py index 20f639563..95151bf83 100644 --- a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py @@ -348,7 +348,10 @@ def transition_stage( collector_status = { "failed": "failed", + "completed": "completed", "passed": "completed", + "progress": "progress", + "started": "started", "running": "started", }.get(status, "progress") if collector_status == "started": diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py index a3202cfa6..cc89d2eca 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/full_build_cli.py @@ -174,10 +174,10 @@ from .rowwise_staging import ( STAGED_DATASET_PHASES, STAGING_UPLOAD_INTERVAL_SECONDS, - calibration_progress, - create_staging_telemetry, - fail_staging_telemetry, - finalize_staging_telemetry, + create_staging_run_bundle, + emit_calibration_progress, + fail_staging_run_bundle, + finalize_staging_run_bundle, gate_statuses, graph_progress, preflight_staged_dataset, @@ -215,7 +215,7 @@ ) -def _run_graph_with_progress(compiled, *, telemetry, **kwargs): +def _run_graph_with_progress(compiled, **kwargs): """Run a graph and emit one lightweight update per completed node.""" completed = 0 @@ -227,7 +227,6 @@ def observe(node_id, _population) -> None: now = time.monotonic() completed += 1 graph_progress( - telemetry, node_id=node_id, done=completed, total=total, @@ -603,11 +602,14 @@ def _solve_observer(args: argparse.Namespace, telemetry): sinks = [uk_solve_progress_callback(_stderr_progress)] sinks.append( - thinned_epochs( - lambda event: calibration_progress(telemetry, event), - every=staging_epoch_every(args), - ) + thinned_epochs(emit_calibration_progress, every=staging_epoch_every(args)) ) + if telemetry is not None: + sinks.append( + thinned_epochs( + telemetry.calibration_progress, every=staging_epoch_every(args) + ) + ) def observer(event: dict[str, object]) -> None: for sink in sinks: @@ -1149,11 +1151,11 @@ def _close_attempt( manifest=manifest, output_paths={"manifest": output / MANIFEST_FILENAME}, run_id=state.build_id if telemetry is None else telemetry.run_id, - telemetry=telemetry, + staging_bundle=telemetry, ) append_phase(state, STAGED_DATASET_PHASES[staged_dataset["status"]]) try: - finalize_staging_telemetry(args, telemetry) + finalize_staging_run_bundle(args, telemetry) finally: manifest["staging_delivery"] = staging_delivery(telemetry) manifest["staged_dataset"] = staged_dataset @@ -1167,7 +1169,7 @@ def _close_attempt( output / f"{stem}.h5", repository_hint=REPOSITORY ) else: - finalize_staging_telemetry(args, telemetry) + finalize_staging_run_bundle(args, telemetry) spool_path = record_candidate_attempt( state=state, started_at=attempt["started_at"], @@ -1316,7 +1318,6 @@ def _execute_full_build( continue checkpoint = _run_graph_with_progress( compile_graph(_through(graph, endpoint)), - telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1329,7 +1330,6 @@ def _execute_full_build( preflight_graph = _through(graph, "uk.full.gates.preflight") preflight = _run_graph_with_progress( compile_graph(preflight_graph), - telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1366,7 +1366,6 @@ def _execute_full_build( ) manifest = _run_graph_with_progress( compile_graph(_through(graph, "uk.full.gates.calibrated")), - telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1444,7 +1443,6 @@ def _execute_full_build( stage(telemetry, "output_bundle", "started") manifest = _run_graph_with_progress( compile_graph(graph), - telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1525,7 +1523,6 @@ def _execute_full_build( ) final = _run_graph_with_progress( compile_graph(graph), - telemetry=telemetry, sources={ **sources, "exported_dataset": dataset, @@ -1912,7 +1909,6 @@ def _execute_national_build( stage(telemetry, "input_loading", "started") checkpoint = _run_graph_with_progress( compile_graph(_through(graph, "uk.full.spine_checkpoint")), - telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1934,7 +1930,6 @@ def _execute_national_build( stage(telemetry, "target_compilation", "started") targets = _run_graph_with_progress( compile_graph(_through(graph, NATIONAL_TARGETS_NODE)), - telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -1988,7 +1983,6 @@ def _execute_national_build( ) manifest = _run_graph_with_progress( compile_graph(_through(graph, NATIONAL_GATES_NODE)), - telemetry=telemetry, sources=sources, store=store, kernels=kernels, @@ -2142,7 +2136,6 @@ def _execute_national_build( continued = add_uk_national_readback(graph, population=national.population) final = _run_graph_with_progress( compile_graph(continued), - telemetry=telemetry, sources={**sources, "exported_dataset": paths["dataset"]}, store=store, kernels=kernels, @@ -2281,7 +2274,7 @@ def _close_national_attempt( manifest=manifest, output_paths=paths, run_id=state.build_id if telemetry is None else telemetry.run_id, - telemetry=telemetry, + staging_bundle=telemetry, ) append_phase(state, STAGED_DATASET_PHASES[staged_dataset["status"]]) evaluation = national_role.evaluate_against_incumbent( @@ -2294,7 +2287,7 @@ def _close_national_attempt( out_dir=output, ) try: - finalize_staging_telemetry(args, telemetry) + finalize_staging_run_bundle(args, telemetry) finally: delivery = staging_delivery(telemetry) build_record = json.loads(paths["build_record"].read_text()) @@ -2328,7 +2321,7 @@ def _close_national_attempt( }, ) else: - finalize_staging_telemetry(args, telemetry) + finalize_staging_run_bundle(args, telemetry) spool_path = record_candidate_attempt( state=state, started_at=attempt["started_at"], @@ -2364,7 +2357,7 @@ def _national_main(args: argparse.Namespace) -> int: try: digest = preflight_digest(posture.pipeline) except BaseException as error: - fail_staging_telemetry(None, error) + fail_staging_run_bundle(None, error) raise state = AttemptState( # The staging run id and Logbook row use the same attempt id. @@ -2388,9 +2381,9 @@ def _national_main(args: argparse.Namespace) -> int: ).national.operation_inventory(), ) except BaseException as error: - fail_staging_telemetry(None, error) + fail_staging_run_bundle(None, error) raise - finalize_staging_telemetry(args, None) + finalize_staging_run_bundle(args, None) return status try: @@ -2400,9 +2393,9 @@ def _national_main(args: argparse.Namespace) -> int: refuse_occupied_national_output(args) preflight_staged_dataset(args) predecessor = resolve_predecessor(args.logbook_prev_row_digest) - telemetry = create_staging_telemetry(args, build_id=state.build_id) + telemetry = create_staging_run_bundle(args, build_id=state.build_id) except BaseException as error: - fail_staging_telemetry(None, error) + fail_staging_run_bundle(None, error) raise attempt = { "state": state, @@ -2428,13 +2421,13 @@ def _national_main(args: argparse.Namespace) -> int: pipeline=posture.pipeline, disposition="discarded", ) - fail_staging_telemetry(telemetry, interrupt) + fail_staging_run_bundle(telemetry, interrupt) raise except Exception as error: _record_failure( args, error, state=state, attempt=attempt, pipeline=posture.pipeline ) - fail_staging_telemetry(telemetry, error) + fail_staging_run_bundle(telemetry, error) print(f"UK national build failed: {error}", file=sys.stderr) return 1 @@ -2536,7 +2529,7 @@ def main(argv: list[str] | None = None) -> int: try: digest = preflight_digest(posture.pipeline) except BaseException as error: - fail_staging_telemetry(None, error) + fail_staging_run_bundle(None, error) raise state = AttemptState( build_id=build_id, @@ -2556,9 +2549,9 @@ def main(argv: list[str] | None = None) -> int: try: status = _dry_run(args) except BaseException as error: - fail_staging_telemetry(None, error) + fail_staging_run_bundle(None, error) raise - finalize_staging_telemetry(args, None) + finalize_staging_run_bundle(args, None) return status try: @@ -2566,9 +2559,9 @@ def main(argv: list[str] | None = None) -> int: # the emitter records their failures. preflight_staged_dataset(args) predecessor = resolve_predecessor(args.logbook_prev_row_digest) - telemetry = create_staging_telemetry(args, build_id=state.build_id) + telemetry = create_staging_run_bundle(args, build_id=state.build_id) except BaseException as error: - fail_staging_telemetry(None, error) + fail_staging_run_bundle(None, error) raise attempt = { "state": state, @@ -2593,7 +2586,7 @@ def main(argv: list[str] | None = None) -> int: prepared=prepared, disposition="discarded", ) - fail_staging_telemetry(telemetry, interrupt) + fail_staging_run_bundle(telemetry, interrupt) raise except Exception as error: _record_failure( @@ -2604,7 +2597,7 @@ def main(argv: list[str] | None = None) -> int: pipeline=posture.pipeline, prepared=prepared, ) - fail_staging_telemetry(telemetry, error) + fail_staging_run_bundle(telemetry, error) print(f"UK full build failed: {error}", file=sys.stderr) return 1 diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/national_role.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/national_role.py index 5d6f6cf94..5ffc89ce7 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/national_role.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/national_role.py @@ -34,7 +34,7 @@ parse_sha256sums, refresh_sha256sums_entry, ) -from microcosm.build.staging_v2 import StagingTelemetryV2 +from microcosm.build.staging_v2 import StagingRunBundleWriterV2 from microcosm.build.uk_runtime.calibration_run import runtime_provenance from microcosm.build.uk_runtime.diagnostics import uk_fit_by_family from microcosm.build.uk_runtime.frs_release import load_uk_frs_release @@ -480,7 +480,7 @@ def evaluate_against_incumbent( incumbent: Mapping[str, Any] | None, inputs: Mapping[str, Any], output_paths: Mapping[str, Path], - telemetry: StagingTelemetryV2 | None, + telemetry: StagingRunBundleWriterV2 | None, calibration_year: int, out_dir: Path, ) -> dict[str, Any]: diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py index b8894cd45..2f0ababfc 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/rowwise_staging.py @@ -1,6 +1,6 @@ -"""Staging telemetry and staged-dataset delivery of the UK rowwise roles. +"""Staging run bundles and staged-dataset delivery for UK rowwise roles. -Two destinations, one run id: reviewed aggregate telemetry goes to the +Two destinations, one run id: reviewed aggregate build files go to the staging repository under ``runs//``, the finished bundle a manifest vouches for to the private artifact repository under ``staged//``. Both are best-effort evidence about the build, never a release, and the @@ -32,7 +32,7 @@ from microcosm.build.staging_storage import HuggingFaceDatasetStorage from microcosm.build.staging_v2 import ( StagingContractError, - StagingTelemetryV2, + StagingRunBundleWriterV2, disabled_staging_delivery, ) from microcosm.build.telemetry_emitter import ( @@ -54,10 +54,10 @@ "STAGING_MAX_EPOCH_ROWS", "STAGING_UPLOAD_INTERVAL_SECONDS", "add_staging_artifact", - "calibration_progress", - "create_staging_telemetry", - "fail_staging_telemetry", - "finalize_staging_telemetry", + "create_staging_run_bundle", + "emit_calibration_progress", + "fail_staging_run_bundle", + "finalize_staging_run_bundle", "fit_summary", "gate_statuses", "graph_progress", @@ -73,13 +73,13 @@ "thinned_epochs", ] -# Best-effort telemetry upload cadence. The Hub allows about 128 commits per +# Best-effort staging-bundle upload cadence. The Hub allows about 128 commits per # hour per repository and one cycle is up to eight single-file commits, so # the shared 30-second default exhausts the budget on a multi-hour solve # and loses uploads (the v20 national run did); five minutes keeps a # 1,500-epoch run well inside it. STAGING_UPLOAD_INTERVAL_SECONDS = 300.0 -# Staging telemetry keeps one row per forwarded epoch in +# The staging bundle keeps one row per forwarded epoch in # calibration_progress.json and one event in events.ndjson, both under the # contract's 5 MiB remote cap. A size run at 2,000 epochs solves the dense # pool, up to ten full-length L0 probes and the refit: about 24,000 epochs, @@ -103,7 +103,7 @@ def _hub_api() -> Any: - """The Hub client used for telemetry and the staged dataset (test seam).""" + """The Hub client used for the run bundle and staged dataset (test seam).""" from huggingface_hub import HfApi @@ -191,17 +191,16 @@ def _require_write_credential(storage: HuggingFaceDatasetStorage, *, hint: str) ) -def create_staging_telemetry( +def create_staging_run_bundle( args: argparse.Namespace, *, build_id: str -) -> StagingTelemetryV2 | None: +) -> StagingRunBundleWriterV2 | None: posture = posture_of(args) run_id = args.staging_run_id or build_id - emitter = start_telemetry_emitter(args, build_id=build_id) if args.no_staging: return None local_only = bool(args.staging_local_only) out_dir = args.out.expanduser().resolve() - return StagingTelemetryV2( + return StagingRunBundleWriterV2( run_id=run_id, country_code="GB", operation_id=posture.staging_operation_id, @@ -214,7 +213,6 @@ def create_staging_telemetry( repo_id=None if local_only else args.staging_repo_id, upload_interval_seconds=args.staging_upload_interval_seconds, api=None if local_only else _hub_api(), - emitter=emitter, ) @@ -247,16 +245,16 @@ def start_telemetry_emitter( return _ACTIVE_EMITTER -_create_staging_telemetry = create_staging_telemetry +_create_staging_run_bundle = create_staging_run_bundle def stage( - telemetry: StagingTelemetryV2 | None, + staging_bundle: StagingRunBundleWriterV2 | None, stage_id: str, event_status: str = "started", **details: Any, ) -> None: - """Forward one stage event to the best-effort telemetry. + """Report one stage independently to the emitter and staging bundle. A contract or content refusal of the event is reported and the event dropped; the build must never abort on its own progress report. Each @@ -264,19 +262,25 @@ def stage( the ones that follow. """ - if telemetry is None: - if _ACTIVE_EMITTER is not None: - _ACTIVE_EMITTER.stage( - stage_id, - status=event_status, - **details, - ) + emitter_details = dict(details) + message = emitter_details.pop("message", None) + emitter_details.pop("force_upload", None) + if "status" in emitter_details: + emitter_details["result_status"] = emitter_details.pop("status") + if _ACTIVE_EMITTER is not None: + _ACTIVE_EMITTER.transition_stage( + stage_id, + status=event_status, + message=message, + **emitter_details, + ) + if staging_bundle is None: return try: - telemetry.stage(stage_id, event_status=event_status, **details) + staging_bundle.stage(stage_id, event_status=event_status, **details) except StagingContractError as error: print( - f"warning: staging telemetry refused the {stage_id!r} stage event " + f"warning: staging run bundle refused the {stage_id!r} stage event " f"({type(error).__name__}: {error}); the event is not staged, the " "build continues.", file=sys.stderr, @@ -287,20 +291,14 @@ def stage( _stage = stage -def calibration_progress( - telemetry: StagingTelemetryV2 | None, - event: Mapping[str, Any], -) -> None: - """Forward calibration progress with or without staging artifacts.""" +def emit_calibration_progress(event: Mapping[str, Any]) -> None: + """Report calibration progress only through the local emitter service.""" - if telemetry is not None: - telemetry.calibration_progress(event) - elif _ACTIVE_EMITTER is not None: + if _ACTIVE_EMITTER is not None: _ACTIVE_EMITTER.transition_calibration_progress(event) def graph_progress( - telemetry: StagingTelemetryV2 | None, *, node_id: str, done: int, @@ -309,10 +307,9 @@ def graph_progress( ) -> None: """Send graph work progress only through the local emitter service.""" - emitter = telemetry.emitter if telemetry is not None else _ACTIVE_EMITTER - if emitter is None: + if _ACTIVE_EMITTER is None: return - emitter.emit( + _ACTIVE_EMITTER.emit( event_type="progress", stage_id=node_id, status="completed", @@ -326,9 +323,9 @@ def graph_progress( def stage_sample( - telemetry: StagingTelemetryV2 | None, *, sample_fraction: float + staging_bundle: StagingRunBundleWriterV2 | None, *, sample_fraction: float ) -> None: - """Record the run's sampling evidence on the staging telemetry. + """Record the run's sampling evidence in the staging bundle. The contract's only sampling statement is ``{"mode": "full"}``, the f100 rung; a rung below f100 stages a null sample, as the spine builder does. @@ -337,9 +334,9 @@ def stage_sample( own fraction), so the two release roles stage the same evidence. """ - if telemetry is None or float(sample_fraction) != 1.0: + if staging_bundle is None or float(sample_fraction) != 1.0: return - telemetry.set_sample({"mode": "full"}) + staging_bundle.set_sample({"mode": "full"}) def staging_epoch_every(args: argparse.Namespace) -> int: @@ -391,7 +388,7 @@ def callback(event: dict[str, object]) -> None: except StagingContractError as error: disabled = True print( - "warning: staging telemetry refused a calibration progress row " + "warning: staging run bundle refused a calibration progress row " f"({type(error).__name__}); epoch progress is no longer forwarded, " "the solve continues.", file=sys.stderr, @@ -415,50 +412,51 @@ def gate_statuses(gate_report: Mapping[str, Any]) -> dict[str, str]: _gate_statuses = gate_statuses -def fail_staging_telemetry( - telemetry: StagingTelemetryV2 | None, error: BaseException +def fail_staging_run_bundle( + staging_bundle: StagingRunBundleWriterV2 | None, error: BaseException ) -> None: - if telemetry is None: - if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: - _ACTIVE_EMITTER.fail(error) + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.fail(error) + if staging_bundle is None: return - if telemetry.status != "running": + if staging_bundle.status != "running": return try: - telemetry.fail(error) - telemetry.validate_local_bundle() + staging_bundle.fail(error) + staging_bundle.validate_local_bundle() except Exception: pass -_fail_staging_telemetry = fail_staging_telemetry +_fail_staging_run_bundle = fail_staging_run_bundle -def finalize_staging_telemetry( - args: argparse.Namespace, telemetry: StagingTelemetryV2 | None +def finalize_staging_run_bundle( + args: argparse.Namespace, staging_bundle: StagingRunBundleWriterV2 | None ) -> None: - if telemetry is None: - if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: - _ACTIVE_EMITTER.complete() - return - try: - telemetry.complete(message="UK rowwise candidate staging run completed.") - except StagingContractError as error: - _warn_telemetry("could not close the staging run", error) - return - try: - if args.staging_read_back: - # Requested explicitly, so a failed read-back is the run's failure, - # as on the national command. - telemetry.verify_remote() - finally: + if staging_bundle is not None: try: - telemetry.validate_local_bundle() + staging_bundle.complete( + message="UK rowwise candidate staging run completed." + ) except StagingContractError as error: - _warn_telemetry("the local staging bundle does not validate", error) + _warn_telemetry("could not close the staging run", error) + else: + try: + if args.staging_read_back: + # Requested explicitly, so a failed read-back is the run's + # failure, as on the national command. + staging_bundle.verify_remote() + finally: + try: + staging_bundle.validate_local_bundle() + except StagingContractError as error: + _warn_telemetry("the local staging bundle does not validate", error) + if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: + _ACTIVE_EMITTER.complete() -_finalize_staging_telemetry = finalize_staging_telemetry +_finalize_staging_run_bundle = finalize_staging_run_bundle def _warn_telemetry(what: str, error: BaseException) -> None: @@ -470,10 +468,12 @@ def _warn_telemetry(what: str, error: BaseException) -> None: ) -def staging_delivery(telemetry: StagingTelemetryV2 | None) -> dict[str, Any]: - if telemetry is None: +def staging_delivery( + staging_bundle: StagingRunBundleWriterV2 | None, +) -> dict[str, Any]: + if staging_bundle is None: return disabled_staging_delivery("--no-staging") - return telemetry.delivery_summary + return staging_bundle.delivery_summary _staging_delivery = staging_delivery @@ -485,7 +485,7 @@ def stage_dataset( manifest: Mapping[str, Any], output_paths: Mapping[str, Path], run_id: str, - telemetry: StagingTelemetryV2 | None, + staging_bundle: StagingRunBundleWriterV2 | None, ) -> dict[str, Any]: """Stage the published bundle under ``staged//``; record, never raise. @@ -502,7 +502,13 @@ def stage_dataset( repository = ( None if mode == "local_only" else str(args.staged_dataset_repo_id).strip() ) - stage(telemetry, "dataset_staging", "started", mode=mode, repository=repository) + stage( + staging_bundle, + "dataset_staging", + "started", + mode=mode, + repository=repository, + ) statuses = gate_statuses(getattr(args, "_gate_report", {}) or {}) bundle = StagedDatasetBundle.from_manifest( output_paths["manifest"].parent, @@ -519,11 +525,11 @@ def stage_dataset( ) telemetry_reference = ( None - if telemetry is None + if staging_bundle is None else { - "repository": telemetry.repo_id, - "prefix": telemetry.repo_run_prefix, - "mode": telemetry.delivery_mode, + "repository": staging_bundle.repo_id, + "prefix": staging_bundle.repo_run_prefix, + "mode": staging_bundle.delivery_mode, } ) write_sidecars( @@ -547,7 +553,7 @@ def stage_dataset( ) print(_staged_dataset_line(delivery), file=sys.stderr, flush=True) stage( - telemetry, + staging_bundle, "dataset_staging", "completed", status=delivery["status"], @@ -557,14 +563,14 @@ def stage_dataset( file_count=len(delivery["files"]), ) add_staging_artifact( - telemetry, + staging_bundle, "staged_dataset", delivery, artifact_kind="build_metadata", classification="non_row_level", ) add_staging_artifact( - telemetry, + staging_bundle, "fit_summary", fit_summary( manifest, @@ -601,26 +607,26 @@ def _staged_dataset_line(delivery: Mapping[str, Any]) -> str: def add_staging_artifact( - telemetry: StagingTelemetryV2 | None, + staging_bundle: StagingRunBundleWriterV2 | None, logical_name: str, payload: Mapping[str, Any], *, artifact_kind: str, classification: str, ) -> None: - """Attach a reviewed aggregate JSON artifact to the telemetry run. + """Attach a reviewed aggregate JSON artifact to the staging run bundle. A content-policy refusal is reported and skipped: the telemetry is best-effort and must never fail a finished build. """ - if telemetry is None: + if staging_bundle is None: return with tempfile.TemporaryDirectory(prefix=".staging-artifact.") as scratch: source = Path(scratch) / f"{logical_name}.json" source.write_text(json_text(payload), encoding="utf-8") try: - telemetry.add_artifact( + staging_bundle.add_artifact( logical_name, source, artifact_kind=artifact_kind, diff --git a/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py b/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py index 87c8a5f21..b0d92ccfe 100644 --- a/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py +++ b/packages/microcosm-build/src/microcosm/build/uk_runtime/spine_build.py @@ -54,7 +54,7 @@ validate_staging_arguments, ) from microcosm.build.staging_v2 import ( - StagingTelemetryV2, + StagingRunBundleWriterV2, disabled_staging_delivery, ) from microcosm.build.telemetry_emitter import ( @@ -1016,13 +1016,18 @@ def checkpoint_metadata(self) -> dict[str, object]: return dict(hook()) -def _staging_stage_observer(telemetry: StagingTelemetryV2) -> StageObserver: - """Translate a shared stage observation into staging telemetry.""" +def _stage_observer( + staging_bundle: StagingRunBundleWriterV2 | None, + emitter: LocalTelemetryEmitter | None, +) -> StageObserver: + """Report a stage independently to the emitter and staging bundle.""" def observe(observation: StageObservation) -> None: - telemetry.stage( + _record_stage( + staging_bundle, + emitter, observation.stage_id, - event_status=observation.status, + observation.status, elapsed_seconds=observation.elapsed_seconds, entity_row_counts=dict(observation.entity_row_counts), produced_column_count=observation.produced_column_count, @@ -1031,19 +1036,19 @@ def observe(observation: StageObservation) -> None: return observe -def _emitter_stage_observer(emitter: LocalTelemetryEmitter) -> StageObserver: - """Report graph stage observations when file staging is disabled.""" - - def observe(observation: StageObservation) -> None: - emitter.stage( - observation.stage_id, - status=observation.status, - elapsed_seconds=observation.elapsed_seconds, - entity_row_counts=dict(observation.entity_row_counts), - produced_column_count=observation.produced_column_count, - ) +def _record_stage( + staging_bundle: StagingRunBundleWriterV2 | None, + emitter: LocalTelemetryEmitter | None, + stage_id: str, + status: str, + **details: object, +) -> None: + """Send one stage update to two independent destinations.""" - return observe + if emitter is not None: + emitter.transition_stage(stage_id, status=status, **details) + if staging_bundle is not None: + staging_bundle.stage(stage_id, event_status=status, **details) def _graph_progress_observer(emitter: LocalTelemetryEmitter | None, *, total: int): @@ -1250,20 +1255,17 @@ def _exception_chain_contains(error: BaseException, text: str) -> bool: return False -def _create_staging_telemetry( +def _create_staging_run_bundle( args: argparse.Namespace, *, state: AttemptState, - emitter: LocalTelemetryEmitter | None = None, -) -> StagingTelemetryV2 | None: - global _ACTIVE_EMITTER +) -> StagingRunBundleWriterV2 | None: run_id = args.staging_run_id or state.build_id - _ACTIVE_EMITTER = emitter or _start_telemetry_emitter(args, state=state) if args.no_staging: return None local_dir = args.staging_dir or args.spine_h5.parent / "staging" local_only = args.staging_local_only - return StagingTelemetryV2( + return StagingRunBundleWriterV2( run_id=run_id, country_code="GB", operation_id="uk_frs_spine", @@ -1275,7 +1277,6 @@ def _create_staging_telemetry( delivery_mode="local_only" if local_only else "local_and_remote", repo_id=None if local_only else args.staging_repo_id, upload_interval_seconds=args.staging_upload_interval_seconds, - emitter=_ACTIVE_EMITTER, ) @@ -1311,11 +1312,11 @@ def _telemetry_sample( def _staging_delivery( - args: argparse.Namespace, telemetry: StagingTelemetryV2 | None + args: argparse.Namespace, staging_bundle: StagingRunBundleWriterV2 | None ) -> dict[str, object]: - if telemetry is None: + if staging_bundle is None: return disabled_staging_delivery("--no-staging") - return telemetry.delivery_summary + return staging_bundle.delivery_summary @dataclass(frozen=True) @@ -1353,7 +1354,7 @@ def prepare_uk_spine_execution( roster, the private-input checks, the stage implementations (licensed or synthetic fixture), the sampling root, the optional observation wrap, the declared graph sources, and the gate nodes bound to the rules engine. - ``observer`` is the staging telemetry's stage observer when telemetry is + ``observer`` is the staging run bundle's stage observer when that bundle is enabled; ``main`` supplies it, and a caller without telemetry passes none. """ @@ -1759,7 +1760,7 @@ def main(argv: list[str] | None = None) -> int: ) code_pin = "unresolved-local-git-code-pin" spool_dir = args.spine_h5.parent / "logbook-spool" - telemetry: StagingTelemetryV2 | None = None + staging_bundle: StagingRunBundleWriterV2 | None = None spine_battery: GateBatteryRun | None = None emitter = _start_telemetry_emitter(args, state=state) try: @@ -1784,28 +1785,24 @@ def main(argv: list[str] | None = None) -> int: stale_outputs.append(args.emit_nonzero_shares) for stale in stale_outputs: stale.unlink(missing_ok=True) - telemetry = _create_staging_telemetry(args, state=state, emitter=emitter) - if telemetry is not None: + staging_bundle = _create_staging_run_bundle(args, state=state) + if staging_bundle is not None: initial_sample = _telemetry_sample(args, None) if initial_sample is not None: - telemetry.set_sample(initial_sample) - telemetry.stage( - "configuration", - event_status="completed", - smoke=args.smoke, - sample_mode=("fraction" if args.sample_fraction != 1.0 else "full"), - ) + staging_bundle.set_sample(initial_sample) + _record_stage( + staging_bundle, + emitter, + "configuration", + "completed", + smoke=args.smoke, + sample_mode=("fraction" if args.sample_fraction != 1.0 else "full"), + ) code_pin = git_code_pin(_REPOSITORY) append_phase(state, "configured") prepared = prepare_uk_spine_execution( args, - observer=( - _staging_stage_observer(telemetry) - if telemetry is not None - else _emitter_stage_observer(_ACTIVE_EMITTER) - if _ACTIVE_EMITTER is not None - else None - ), + observer=_stage_observer(staging_bundle, emitter), ) spec = prepared.spec graph = prepared.graph @@ -1835,13 +1832,14 @@ def main(argv: list[str] | None = None) -> int: canonical_json_bytes(run_config) ).hexdigest() append_phase(state, "inputs_pinned") - if telemetry is not None: - telemetry.stage( - "input_verification", - event_status="completed", - stage_count=len(stage_names), - input_artifact_count=len(artifact_pins) + len(input_artifact_pins), - ) + _record_stage( + staging_bundle, + emitter, + "input_verification", + "completed", + stage_count=len(stage_names), + input_artifact_count=len(artifact_pins) + len(input_artifact_pins), + ) stochastic_contract = prepared.stochastic_contract frs_release = prepared.frs_release spine_gate_path = _spine_gate_report_path(args.spine_h5) @@ -1891,23 +1889,24 @@ def main(argv: list[str] | None = None) -> int: ) stored_evidence = spine_sidecar_evidence(stored_stages) sampling = stored_evidence["sampling"] - if telemetry is not None: - sample = _telemetry_sample(args, sampling) - if sample is not None: - telemetry.set_sample(sample) - telemetry.stage( - "sampling", - event_status="completed", - realized_household_rows=( - len(frame.table("household")) - if sampling is None - else sampling.get( - "realized_household_rows", - sampling.get("post_household_count"), - ) - ), - ) - telemetry.stage("validation", event_status="started") + sample = _telemetry_sample(args, sampling) + if staging_bundle is not None and sample is not None: + staging_bundle.set_sample(sample) + _record_stage( + staging_bundle, + emitter, + "sampling", + "completed", + realized_household_rows=( + len(frame.table("household")) + if sampling is None + else sampling.get( + "realized_household_rows", + sampling.get("post_household_count"), + ) + ), + ) + _record_stage(staging_bundle, emitter, "validation", "started") if spine_battery is not None: materialize_spine_gate_reports( graph_manifest, @@ -1917,24 +1916,25 @@ def main(argv: list[str] | None = None) -> int: ) if spine_battery is not None: append_phase(state, "spine_gates_evaluated") - if telemetry is not None: - telemetry.stage( - "validation", - event_status="completed", - entity_row_counts=_entity_row_counts(frame), - ) + _record_stage( + staging_bundle, + emitter, + "validation", + "completed", + entity_row_counts=_entity_row_counts(frame), + ) append_phase(state, "spine_built") - if telemetry is not None: - telemetry.stage("spine_h5_creation", event_status="started") + _record_stage(staging_bundle, emitter, "spine_h5_creation", "started") output = write_uk_national_frame(frame, args.spine_h5) if args.smoke: _mark_non_release_h5(output, build_id=state.build_id) - if telemetry is not None: - telemetry.stage( - "spine_h5_creation", - event_status="completed", - size_bytes=output.stat().st_size, - ) + _record_stage( + staging_bundle, + emitter, + "spine_h5_creation", + "completed", + size_bytes=output.stat().st_size, + ) append_phase(state, "spine_written") if args.checkpoint_dir is not None: args.checkpoint_dir.mkdir(parents=True, exist_ok=True) @@ -1956,8 +1956,7 @@ def main(argv: list[str] | None = None) -> int: "report_kind": str(json.loads(replay_bytes).get("report_kind", "")), "sha256": hashlib.sha256(replay_bytes).hexdigest(), } - if telemetry is not None: - telemetry.stage("sidecar_creation", event_status="started") + _record_stage(staging_bundle, emitter, "sidecar_creation", "started") sidecar = _build_sidecar( frame=frame, stages=stages, @@ -1978,7 +1977,7 @@ def main(argv: list[str] | None = None) -> int: else "development" ), synthetic_fixture=synthetic_fixture, - staging_delivery=_staging_delivery(args, telemetry), + staging_delivery=_staging_delivery(args, staging_bundle), spine_gate_report=( { "path": str(spine_gate_path), @@ -2000,12 +1999,13 @@ def main(argv: list[str] | None = None) -> int: if fit_weight_records: sidecar["fit_weight_records"] = fit_weight_records atomic_write_json(sidecar_path, sidecar) - if telemetry is not None: - telemetry.stage( - "sidecar_creation", - event_status="completed", - size_bytes=sidecar_path.stat().st_size, - ) + _record_stage( + staging_bundle, + emitter, + "sidecar_creation", + "completed", + size_bytes=sidecar_path.stat().st_size, + ) append_phase(state, "build_sidecar_written") if args.emit_nonzero_shares is not None: final_columns = list( @@ -2024,8 +2024,8 @@ def main(argv: list[str] | None = None) -> int: }, ) append_phase(state, "nonzero_shares_written") - if telemetry is not None: - telemetry.complete( + if staging_bundle is not None: + staging_bundle.complete( message=( "Non-release smoke verification completed." if args.smoke @@ -2033,12 +2033,12 @@ def main(argv: list[str] | None = None) -> int: ) ) if args.staging_read_back: - telemetry.verify_remote() - telemetry.validate_local_bundle() - sidecar["staging_delivery"] = telemetry.delivery_summary + staging_bundle.verify_remote() + staging_bundle.validate_local_bundle() + sidecar["staging_delivery"] = staging_bundle.delivery_summary atomic_write_json(sidecar_path, sidecar) - elif _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: - _ACTIVE_EMITTER.complete() + if emitter.available: + emitter.complete() state.artifact_location = local_artifact_reference( output, repository_hint=_REPOSITORY, @@ -2085,14 +2085,14 @@ def main(argv: list[str] | None = None) -> int: materialize_blocked_spine_gate_report(error, battery=spine_battery) except GateBatteryBlockedError as blocked: error = blocked - if telemetry is not None and telemetry.status == "running": + if staging_bundle is not None and staging_bundle.status == "running": try: - telemetry.fail(error) - telemetry.validate_local_bundle() + staging_bundle.fail(error) + staging_bundle.validate_local_bundle() except Exception: pass - if _ACTIVE_EMITTER is not None and _ACTIVE_EMITTER.available: - _ACTIVE_EMITTER.fail(error) + if emitter.available: + emitter.fail(error) if _is_sampled(args) and _exception_chain_contains( error, _RUNG_NAMED_EDGE_SIGNATURE ): diff --git a/packages/microcosm-build/tests/engine_free/shared/test_staging.py b/packages/microcosm-build/tests/engine_free/shared/test_staging.py index be9490272..3d6b62c03 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_staging.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_staging.py @@ -1,7 +1,8 @@ +import inspect import json import microcosm.build.staging as staging_module -from microcosm.build.staging import StagingTelemetry +from microcosm.build.staging import StagingRunBundleWriter from microcosm.build.staging_storage import BestEffortUploadSession from test_support.paths import paths_for @@ -10,6 +11,10 @@ V1_FIXTURE = _TEST_PATHS.tests / "fixtures" / "staging" / "v1" +def test_staging_run_bundle_writer_has_no_emitter_dependency() -> None: + assert "emitter" not in inspect.signature(StagingRunBundleWriter).parameters + + class FakeApi: def __init__(self) -> None: self.uploads: list[tuple[str, str, str, str]] = [] @@ -26,8 +31,8 @@ def upload_file( return None -def test_staging_telemetry_writes_local_progress(tmp_path) -> None: - telemetry = StagingTelemetry( +def test_staging_run_bundle_writer_writes_local_progress(tmp_path) -> None: + telemetry = StagingRunBundleWriter( run_id="run-a", candidate_release_id="populace-us-2024-abc-20260618T000000Z", run_dir=tmp_path / "run-a", @@ -51,9 +56,9 @@ def test_staging_telemetry_writes_local_progress(tmp_path) -> None: assert len(events) >= 3 -def test_staging_telemetry_uploads_repo_paths(tmp_path) -> None: +def test_staging_run_bundle_writer_uploads_repo_paths(tmp_path) -> None: api = FakeApi() - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-b", candidate_release_id="populace-us-2024-def-20260618T000000Z", run_dir=tmp_path / "run-b", @@ -96,7 +101,7 @@ class FakeApiWithDownload(FakeApi): def hf_hub_download(self, *, repo_id, filename, repo_type, **kwargs): return str(existing) - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="newer", candidate_release_id="populace-us-2024-new-20260620T000000Z", run_dir=tmp_path / "newer", @@ -120,7 +125,7 @@ def upload_file(self, **kwargs): def hf_hub_download(self, **kwargs): raise RuntimeError("401 Unauthorized") - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-1", candidate_release_id="run-1", run_dir=tmp_path / "run", @@ -143,7 +148,7 @@ def hf_hub_download(self, **kwargs): def test_uploads_succeeded_counts_files_that_reached_the_repo(tmp_path): api = FakeApi() - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-c", candidate_release_id="run-c", run_dir=tmp_path / "run-c", @@ -162,7 +167,7 @@ def test_blank_path_prefix_falls_back_to_the_default(tmp_path): # A blank or slash-only prefix would put run files at the repo root, where # the dashboard's runs/ paths cannot find them. for blank in ("", " ", "/", " / "): - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-e", candidate_release_id="run-e", run_dir=tmp_path / "run-e", @@ -173,7 +178,7 @@ def test_blank_path_prefix_falls_back_to_the_default(tmp_path): def test_path_prefix_is_trimmed_but_otherwise_respected(tmp_path): - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-f", candidate_release_id="run-f", run_dir=tmp_path / "run-f", @@ -184,7 +189,7 @@ def test_path_prefix_is_trimmed_but_otherwise_respected(tmp_path): def test_uploads_succeeded_is_zero_for_a_local_only_run(tmp_path): - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="run-d", candidate_release_id="run-d", run_dir=tmp_path / "run-d", @@ -215,7 +220,7 @@ def hf_hub_download(self, **kwargs): raise FileNotFoundError run_dir = tmp_path / "v1-us-fixture" - telemetry = StagingTelemetry( + telemetry = StagingRunBundleWriter( run_id="v1-us-fixture", candidate_release_id="populace-us-2024-v1-fixture", run_dir=run_dir, diff --git a/packages/microcosm-build/tests/engine_free/shared/test_staging_v2.py b/packages/microcosm-build/tests/engine_free/shared/test_staging_v2.py index a4030f8ad..357c08acc 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_staging_v2.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_staging_v2.py @@ -1,5 +1,6 @@ import argparse import hashlib +import inspect import json import subprocess import sys @@ -26,7 +27,7 @@ StagingContentError, StagingContractError, StagingReadBackError, - StagingTelemetryV2, + StagingRunBundleWriterV2, disabled_staging_delivery, validate_staging_delivery, validate_v2_bundle, @@ -80,7 +81,7 @@ def _sample() -> dict: return {"mode": "full"} -def _recorder(tmp_path, **kwargs) -> StagingTelemetryV2: +def _recorder(tmp_path, **kwargs) -> StagingRunBundleWriterV2: defaults = { "run_id": "uk-smoke-5-42", "country_code": "GB", @@ -96,7 +97,11 @@ def _recorder(tmp_path, **kwargs) -> StagingTelemetryV2: "clock": Clock(), } defaults.update(kwargs) - return StagingTelemetryV2(**defaults) + return StagingRunBundleWriterV2(**defaults) + + +def test_staging_run_bundle_writer_has_no_emitter_dependency() -> None: + assert "emitter" not in inspect.signature(StagingRunBundleWriterV2).parameters def test_version_2_local_bundle_has_explicit_schemas_and_ordered_events(tmp_path): diff --git a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py index a1ba74b76..75ce48358 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py @@ -160,7 +160,7 @@ def test_sequential_stage_updates_close_the_previous_stage() -> None: emitter.transition_stage("load", message="Loading.") emitter.transition_stage("compile", message="Compiling.") - emitter.transition_stage("compile", status="passed", batches=4) + emitter.transition_stage("compile", status="completed", batches=4) events = [message["event"] for message in messages] assert [(event["stage_id"], event["status"]) for event in events] == [ diff --git a/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py b/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py index cf5e58c33..97c21f4dd 100644 --- a/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py +++ b/packages/microcosm-build/tests/engine_free/uk/test_uk_frs_spine.py @@ -108,15 +108,20 @@ def _stored_fit_weight_records( } -def test_staging_stage_observer_translates_shared_observation() -> None: +def test_stage_observer_reports_to_emitter_and_staging_bundle_independently() -> None: tool = _load_tool() - calls = [] + bundle_calls = [] + emitter_calls = [] - class RecordingTelemetry: + class RecordingBundle: def stage(self, stage_id: str, **payload: object) -> None: - calls.append((stage_id, payload)) + bundle_calls.append((stage_id, payload)) - observer = tool._staging_stage_observer(RecordingTelemetry()) + class RecordingEmitter: + def transition_stage(self, stage_id: str, **payload: object) -> None: + emitter_calls.append((stage_id, payload)) + + observer = tool._stage_observer(RecordingBundle(), RecordingEmitter()) observer( StageObservation( stage_id="frs_spine", @@ -127,17 +132,21 @@ def stage(self, stage_id: str, **payload: object) -> None: ) ) - assert calls == [ + details = { + "elapsed_seconds": 1.25, + "entity_row_counts": {"household": 2}, + "produced_column_count": 4, + } + assert bundle_calls == [ ( "frs_spine", { "event_status": "completed", - "elapsed_seconds": 1.25, - "entity_row_counts": {"household": 2}, - "produced_column_count": 4, + **details, }, ) ] + assert emitter_calls == [("frs_spine", {"status": "completed", **details})] def _write_tab(root: Path, table: str, rows: list[dict[str, object]]) -> None: diff --git a/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py b/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py index 9bf43e60c..de62bb5be 100644 --- a/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py +++ b/packages/microcosm-build/tests/engine_free/uk/test_uk_full_build_cli.py @@ -521,7 +521,7 @@ def transition_stage(self, stage_id, **details): ) monkeypatch.setattr( cli, - "finalize_staging_telemetry", + "finalize_staging_run_bundle", lambda requested, telemetry: events.append(("complete", telemetry)), ) @@ -554,12 +554,12 @@ def refuse_preflight(requested): monkeypatch.setattr(cli, "preflight_staged_dataset", refuse_preflight) monkeypatch.setattr( cli, - "fail_staging_telemetry", + "fail_staging_run_bundle", lambda telemetry, error: events.append(("failed", telemetry, str(error))), ) monkeypatch.setattr( cli, - "create_staging_telemetry", + "create_staging_run_bundle", lambda *args, **kwargs: pytest.fail("preflight failure reached staging setup"), ) @@ -574,6 +574,38 @@ def refuse_preflight(requested): assert events[-1][1] is None +def test_hosted_stage_reporting_survives_staging_bundle_refusal(monkeypatch, capsys): + from microcosm.build.staging_v2 import StagingContentError + from microcosm.build.uk_runtime import rowwise_staging + + events = [] + + class RefusingBundle: + def stage(self, *args, **kwargs): + raise StagingContentError("synthetic staging bundle refusal") + + class Emitter: + def transition_stage(self, stage_id, **details): + events.append((stage_id, details)) + + monkeypatch.setattr(rowwise_staging, "_ACTIVE_EMITTER", Emitter()) + + rowwise_staging.stage( + RefusingBundle(), + "target_compilation", + "started", + selected_target_count=10, + ) + + assert events == [ + ( + "target_compilation", + {"status": "started", "message": None, "selected_target_count": 10}, + ) + ] + assert "synthetic staging bundle refusal" in capsys.readouterr().err + + def test_rejected_output_inside_source_never_writes_failure_sidecar( tmp_path, monkeypatch ): @@ -952,11 +984,11 @@ def test_invalid_local_telemetry_bundle_is_a_warning_not_the_runs_failure( from microcosm.build.staging_v2 import StagingContractError from microcosm.build.uk_runtime import rowwise_staging - class Invalid(rowwise_staging.StagingTelemetryV2): + class Invalid(rowwise_staging.StagingRunBundleWriterV2): def validate_local_bundle(self): raise StagingContractError("synthetic bundle defect") - monkeypatch.setattr(rowwise_staging, "StagingTelemetryV2", Invalid) + monkeypatch.setattr(rowwise_staging, "StagingRunBundleWriterV2", Invalid) status, out = run_dense_main(tmp_path, monkeypatch, staging="--staging-local-only") assert status == 0 err = capsys.readouterr().err @@ -975,11 +1007,11 @@ def test_telemetry_content_refusal_never_aborts_the_solve( from microcosm.build.staging_v2 import StagingContentError, validate_v2_bundle from microcosm.build.uk_runtime import rowwise_staging - class Refusing(rowwise_staging.StagingTelemetryV2): + class Refusing(rowwise_staging.StagingRunBundleWriterV2): def calibration_progress(self, event): raise StagingContentError("Staging file exceeds the 5242880-byte limit.") - monkeypatch.setattr(rowwise_staging, "StagingTelemetryV2", Refusing) + monkeypatch.setattr(rowwise_staging, "StagingRunBundleWriterV2", Refusing) status, out = run_dense_main( tmp_path, monkeypatch, staging="--staging-local-only", on_prepare=_drive_epochs ) diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_exact_k_ladder_launcher.py b/packages/microcosm-build/tests/engine_free/us/test_us_exact_k_ladder_launcher.py index 484e548f5..bb1623eb3 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_exact_k_ladder_launcher.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_exact_k_ladder_launcher.py @@ -365,7 +365,7 @@ def __init__(self, **kwargs): monkeypatch.setattr( launcher.fiscal_release, - "StagingTelemetry", + "StagingRunBundleWriter", UnexpectedTelemetry, ) argv = launcher._builder_argv( @@ -376,13 +376,27 @@ def __init__(self, **kwargs): ) parsed = launcher.fiscal_release._parse_args(argv) - telemetry = launcher.fiscal_release._staging_telemetry( + class Emitter: + def transition_stage(self, *args, **kwargs): + pass + + def transition_calibration_progress(self, *args, **kwargs): + pass + + def fail(self, *args, **kwargs): + pass + + def complete(self): + pass + + progress = launcher.fiscal_release._build_progress( parsed, release_root=tmp_path / "out", release_id=config.release_id, + emitter=Emitter(), ) assert parsed.no_staging is True - assert telemetry is None + assert progress.staging_bundle is None assert constructed is False assert not (tmp_path / "out" / "staging").exists() diff --git a/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py b/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py index 33624246c..73cc7bfad 100644 --- a/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py +++ b/packages/microcosm-build/tests/engine_free/us/test_us_fiscal_refresh_builder.py @@ -6835,6 +6835,8 @@ class LiveTelemetry: run_id = "live-telemetry-test" repo_id = "policyengine/populace-us-staging" uploads_succeeded = 3 + staging_bundle = object() + staging_opt_out_reason = None def stage(self, stage, **details): captured.setdefault("telemetry_events", []).append(("stage", stage)) @@ -6867,7 +6869,7 @@ def complete(self): live_telemetry = LiveTelemetry() monkeypatch.setattr( builder, - "_staging_telemetry", + "_build_progress", lambda *args, **kwargs: live_telemetry, ) if terminal_mode in { @@ -13388,7 +13390,28 @@ def test_us_release_id_guard() -> None: raise AssertionError("Expected non-US release id to fail.") -def test_staging_telemetry_defaults_on_and_no_staging_disables(tmp_path, monkeypatch): +class _TestEmitter: + available = True + + def __init__(self) -> None: + self.events = [] + + def transition_stage(self, stage, **details): + self.events.append(("stage", stage, details)) + + def transition_calibration_progress(self, event): + self.events.append(("calibration", event)) + + def fail(self, error): + self.events.append(("failed", error)) + + def complete(self): + self.events.append(("completed",)) + + +def test_staging_bundle_defaults_on_and_no_staging_disables_files( + tmp_path, monkeypatch +): module = _load_builder_module() # The parser defaults staging uploads ON (overridable by env). @@ -13420,20 +13443,51 @@ def namespace(no_staging: bool) -> SimpleNamespace: staging_upload_interval_seconds=60.0, ) - telemetry = module._staging_telemetry( - namespace(no_staging=False), release_root=tmp_path, release_id="rel-1" + progress = module._build_progress( + namespace(no_staging=False), + release_root=tmp_path, + release_id="rel-1", + emitter=_TestEmitter(), ) - assert telemetry is not None - assert telemetry.run_id == "rel-1" - assert telemetry.repo_id is None + assert progress.run_id == "rel-1" + assert progress.staging_bundle is not None + assert progress.repo_id is None - # --no-staging wins even when a staging destination is configured. - assert ( - module._staging_telemetry( - namespace(no_staging=True), release_root=tmp_path, release_id="rel-1" - ) - is None + # --no-staging disables files without disabling hosted progress. + progress = module._build_progress( + namespace(no_staging=True), + release_root=tmp_path, + release_id="rel-1", + emitter=_TestEmitter(), ) + assert progress.staging_bundle is None + assert progress.staging_opt_out_reason == "--no-staging" + + +def test_hosted_progress_survives_staging_bundle_failure() -> None: + module = _load_builder_module() + emitter = _TestEmitter() + + class FailingBundle: + def stage(self, *args, **kwargs): + raise OSError("staging disk unavailable") + + progress = module._BuildProgress( + run_id="rel-1", + emitter=emitter, + staging_bundle=FailingBundle(), + ) + + with pytest.raises(OSError, match="staging disk unavailable"): + progress.stage("target_compilation", batches=4) + + assert emitter.events == [ + ( + "stage", + "target_compilation", + {"status": "running", "message": None, "batches": 4}, + ) + ] def test_blank_staging_repo_id_is_refused_at_parse_time(monkeypatch, capsys) -> None: @@ -13483,7 +13537,7 @@ class Telemetry: def fail(self, error): recorded.append(error) - monkeypatch.setattr(module, "_ACTIVE_TELEMETRY", Telemetry()) + monkeypatch.setattr(module, "_ACTIVE_PROGRESS", Telemetry()) monkeypatch.setattr( module, "_main", @@ -13503,7 +13557,7 @@ class ExplodingTelemetry: def fail(self, error): raise RuntimeError("telemetry itself is broken") - monkeypatch.setattr(module, "_ACTIVE_TELEMETRY", ExplodingTelemetry()) + monkeypatch.setattr(module, "_ACTIVE_PROGRESS", ExplodingTelemetry()) monkeypatch.setattr( module, "_main", @@ -13516,7 +13570,7 @@ def fail(self, error): assert "could not record the staging run as failed" in capsys.readouterr().err -def test_staging_telemetry_clears_any_previous_active_run(tmp_path) -> None: +def test_build_progress_replaces_any_previous_active_run(tmp_path) -> None: module = _load_builder_module() args = SimpleNamespace( no_staging=False, @@ -13526,15 +13580,24 @@ def test_staging_telemetry_clears_any_previous_active_run(tmp_path) -> None: staging_prefix=module.DEFAULT_STAGING_PREFIX, staging_upload_interval_seconds=60.0, ) - module._staging_telemetry(args, release_root=tmp_path, release_id="rel-1") - assert module._ACTIVE_TELEMETRY is not None + first = module._build_progress( + args, + release_root=tmp_path, + release_id="rel-1", + emitter=_TestEmitter(), + ) + assert module._ACTIVE_PROGRESS is first + assert first.staging_bundle is not None args.no_staging = True - assert ( - module._staging_telemetry(args, release_root=tmp_path, release_id="rel-2") - is None + second = module._build_progress( + args, + release_root=tmp_path, + release_id="rel-2", + emitter=_TestEmitter(), ) - assert module._ACTIVE_TELEMETRY is None + assert module._ACTIVE_PROGRESS is second + assert second.staging_bundle is None def test_staging_manifest_block_distinguishes_opt_out_from_delivery() -> None: @@ -13549,6 +13612,8 @@ class Delivered: run_id = "rel-1" repo_id = "policyengine/populace-us-staging" uploads_succeeded = 7 + staging_bundle = object() + staging_opt_out_reason = None assert module._staging_manifest_block(Delivered()) == { "enabled": True, @@ -13561,6 +13626,8 @@ class Undelivered: run_id = "rel-2" repo_id = None uploads_succeeded = 0 + staging_bundle = object() + staging_opt_out_reason = None block = module._staging_manifest_block(Undelivered()) assert block["enabled"] is True @@ -13568,7 +13635,7 @@ class Undelivered: assert block["repo_id"] is None -def test_staging_telemetry_refuses_a_destinationless_namespace(tmp_path) -> None: +def test_staging_bundle_refuses_a_destinationless_namespace(tmp_path) -> None: module = _load_builder_module() args = SimpleNamespace( no_staging=False, @@ -13580,7 +13647,12 @@ def test_staging_telemetry_refuses_a_destinationless_namespace(tmp_path) -> None ) with pytest.raises(ValueError, match="no destination"): - module._staging_telemetry(args, release_root=tmp_path, release_id="rel-1") + module._build_progress( + args, + release_root=tmp_path, + release_id="rel-1", + emitter=_TestEmitter(), + ) # --------------------------------------------------------------------------- @@ -14787,7 +14859,7 @@ def _record_qrf_tail(builder, tmp_path, frame, *, register, allow, failures): allow_concentration=allow, terminal_gate_failures=failures, release_dir=tmp_path, - telemetry=builder._TerminalBatchTelemetry(recorder, failures), + telemetry=builder._TerminalBatchProgress(recorder, failures), ) return register_failures, recorder diff --git a/tools/build_us_fiscal_refresh_release.py b/tools/build_us_fiscal_refresh_release.py index 6729b34c8..630194a81 100644 --- a/tools/build_us_fiscal_refresh_release.py +++ b/tools/build_us_fiscal_refresh_release.py @@ -49,7 +49,7 @@ from datetime import UTC, datetime from pathlib import Path from types import MethodType -from typing import Any, Protocol +from typing import Any import numpy as np import pandas as pd @@ -68,7 +68,7 @@ ) from microcosm.build.ledger_artifact import load_ledger_consumer_artifact from microcosm.build.source_runtime import SourceRuntimeConfig, run_source_stage -from microcosm.build.staging import DEFAULT_STAGING_PREFIX, StagingTelemetry +from microcosm.build.staging import DEFAULT_STAGING_PREFIX, StagingRunBundleWriter from microcosm.build.telemetry_emitter import ( LocalTelemetryEmitter, start_local_telemetry_emitter_service, @@ -1778,7 +1778,7 @@ def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: "--staging-dir", type=Path, help=( - "Optional local directory for staging telemetry artifacts. Defaults " + "Optional local directory for staging run files. Defaults " "to /staging/runs/ when --staging-repo-id is set." ), ) @@ -1786,7 +1786,7 @@ def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: "--staging-repo-id", default=_env_default("POPULACE_STAGING_REPO_ID", STAGING_REPO_ID), help=( - "Hugging Face dataset repo to upload staging telemetry to while " + "Hugging Face dataset repo to upload staging run files to while " "the build runs. On by default (uploads are best-effort and never " "fail the build); override with POPULACE_STAGING_REPO_ID or " "disable with --no-staging. An empty POPULACE_STAGING_REPO_ID is " @@ -1815,7 +1815,10 @@ def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: parser.add_argument( "--no-staging", action="store_true", - help="Disable staging telemetry (local staging dir and uploads) for this build.", + help=( + "Disable staging run files and their uploads for this build; hosted " + "telemetry remains active." + ), ) parser.add_argument( "--staging-prefix", @@ -1912,7 +1915,7 @@ def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: # effect of a blank repo id. parser.error( "--staging-repo-id is empty and no --staging-dir is set, so staging " - "telemetry would silently do nothing. Pass --no-staging to skip " + "run files would have no destination. Pass --no-staging to skip " "staging deliberately, or --staging-dir for a local-only run." ) if args.evidence_failure_owners is not None and not args.evidence_release: @@ -8051,7 +8054,7 @@ def _record_qrf_tail_concentration_gate( allow_concentration: bool, terminal_gate_failures: list[str], release_dir: Path, - telemetry: _TerminalBatchTelemetry, + telemetry: _TerminalBatchProgress, ) -> list[str]: """Evaluate the terminal QRF tail gate and record everything it measured. @@ -8814,7 +8817,7 @@ def _enforce_ssi_take_up_delivery( *, targets: Mapping[str, float], release_dir: Path, - telemetry: _BuildTelemetry | None, + telemetry: _BuildProgress | None, enforcement_fences: Mapping[str, str] | None = None, ) -> tuple[list[str], GateResult]: """Fail the release on an enforced-band delivery miss, via the batch. @@ -11457,52 +11460,33 @@ def _assert_exact_k_original_pool_alignment( _ACTIVE_EMITTER: LocalTelemetryEmitter | None = None -class _BuildTelemetry(Protocol): - """Telemetry operations used by the release build body.""" - - run_id: str - repo_id: str | None - uploads_succeeded: int - - def stage( - self, - stage: str, - *, - message: str | None = None, - status: str = "running", - force_upload: bool = False, - **details: Any, - ) -> None: ... - - def calibration_progress(self, event: dict[str, object]) -> None: ... - - def attach_artifact( - self, - name: str, - path: Path | str, - **details: Any, - ) -> None: ... - - def fail(self, error: BaseException) -> None: ... - - def complete(self) -> None: ... - - -class _EmitterOnlyTelemetry: - """Expose the build telemetry interface without writing staging files.""" +class _BuildProgress: + """Report progress to an emitter and, independently, a staging bundle.""" def __init__( self, *, run_id: str, emitter: LocalTelemetryEmitter, - reason: str, + staging_bundle: StagingRunBundleWriter | None, + staging_opt_out_reason: str | None = None, ) -> None: self.run_id = run_id - self.repo_id: str | None = None - self.uploads_succeeded = 0 - self.reason = reason self.emitter = emitter + self.staging_bundle = staging_bundle + self.staging_opt_out_reason = staging_opt_out_reason + + @property + def repo_id(self) -> str | None: + if self.staging_bundle is None: + return None + return self.staging_bundle.repo_id + + @property + def uploads_succeeded(self) -> int: + if self.staging_bundle is None: + return 0 + return self.staging_bundle.uploads_succeeded def stage( self, @@ -11513,16 +11497,25 @@ def stage( force_upload: bool = False, **details: Any, ) -> None: - del force_upload self.emitter.transition_stage( stage, status=status, message=message, **details, ) + if self.staging_bundle is not None: + self.staging_bundle.stage( + stage, + message=message, + status=status, + force_upload=force_upload, + **details, + ) def calibration_progress(self, event: dict[str, object]) -> None: self.emitter.transition_calibration_progress(event) + if self.staging_bundle is not None: + self.staging_bundle.calibration_progress(event) def attach_artifact( self, @@ -11530,19 +11523,23 @@ def attach_artifact( path: Path | str, **details: Any, ) -> None: - del name, path, details + if self.staging_bundle is not None: + self.staging_bundle.attach_artifact(name, path, **details) def fail(self, error: BaseException) -> None: self.emitter.fail(error) + if self.staging_bundle is not None: + self.staging_bundle.fail(error) def complete(self) -> None: + if self.staging_bundle is not None: + self.staging_bundle.complete() self.emitter.complete() -#: The telemetry interface for the build in flight, so the entry point can -#: report failure on the way out without threading another handle through the -#: release call stack. -_ACTIVE_TELEMETRY: _BuildTelemetry | None = None +#: The progress coordinator for the build in flight, so the entry point can +#: report failure without threading another handle back through the call stack. +_ACTIVE_PROGRESS: _BuildProgress | None = None class _WorkCounter: @@ -11583,7 +11580,8 @@ class _ReleaseDryRun: ``_main`` runs the release as it would build it, up to the point where the staged frame goes to target materialization. It differs in four ways: - * Nothing is written under ``--out`` and staging telemetry stays off. + * Nothing is written under ``--out`` and staging run files stay off; + hosted progress reporting remains active. * The input-mass, degenerate-input and eCPS parity gates take their degraded-mode branch and batch instead of raising, so one report carries them with every register. @@ -11750,7 +11748,7 @@ def _write(self, report: ReleaseDryRunReport) -> int: #: The dry run in flight, so :func:`main` can turn a refusal before the stop -#: point into that run's report. Like :data:`_ACTIVE_TELEMETRY`, a module-level +#: point into that run's report. Like :data:`_ACTIVE_PROGRESS`, a module-level #: handle rather than a second object threaded back up the build's call stack. _ACTIVE_DRY_RUN: _ReleaseDryRun | None = None @@ -12172,17 +12170,20 @@ def zero_support() -> CheckResult: return checks -def _staging_manifest_block(telemetry: _BuildTelemetry | None) -> dict[str, object]: +def _staging_manifest_block(telemetry: _BuildProgress | None) -> dict[str, object]: """Record what staging did and distinguish an opt-out from non-delivery. Uploads are best-effort and self-disable after repeated failures, so a configured destination is not evidence that anything reached it. """ - if telemetry is None: - return {"enabled": False, "reason": "--no-staging"} - if isinstance(telemetry, _EmitterOnlyTelemetry): - return {"enabled": False, "reason": telemetry.reason} + if telemetry is None or telemetry.staging_bundle is None: + reason = ( + "--no-staging" + if telemetry is None + else telemetry.staging_opt_out_reason or "--no-staging" + ) + return {"enabled": False, "reason": reason} return { "enabled": True, "run_id": telemetry.run_id, @@ -12191,28 +12192,27 @@ def _staging_manifest_block(telemetry: _BuildTelemetry | None) -> dict[str, obje } -def _staging_telemetry( +def _build_progress( args: argparse.Namespace, *, release_root: Path, release_id: str, run_id: str | None = None, - emitter: LocalTelemetryEmitter | None = None, -) -> _BuildTelemetry | None: - global _ACTIVE_TELEMETRY + emitter: LocalTelemetryEmitter, +) -> _BuildProgress: + global _ACTIVE_PROGRESS # Each call establishes the current run, so a handle from a previous one # can never be marked failed in place of this build's. - _ACTIVE_TELEMETRY = None + _ACTIVE_PROGRESS = None run_id = run_id or args.staging_run_id or release_id if args.no_staging: - if emitter is None: - return None - _ACTIVE_TELEMETRY = _EmitterOnlyTelemetry( + _ACTIVE_PROGRESS = _BuildProgress( run_id=run_id, emitter=emitter, - reason="--no-staging", + staging_bundle=None, + staging_opt_out_reason="--no-staging", ) - return _ACTIVE_TELEMETRY + return _ACTIVE_PROGRESS if not args.staging_dir and not args.staging_repo_id: # The parser rejects this combination, so reaching it means a caller # built the namespace directly. Returning None here would reinstate @@ -12223,19 +12223,23 @@ def _staging_telemetry( "or give a staging_dir for a local-only run." ) run_dir = args.staging_dir or release_root / "staging" / "runs" / run_id - _ACTIVE_TELEMETRY = StagingTelemetry( + staging_bundle = StagingRunBundleWriter( run_id=run_id, candidate_release_id=release_id, run_dir=run_dir, repo_id=args.staging_repo_id, path_prefix=args.staging_prefix, upload_interval_seconds=args.staging_upload_interval_seconds, + ) + _ACTIVE_PROGRESS = _BuildProgress( + run_id=run_id, emitter=emitter, + staging_bundle=staging_bundle, ) - return _ACTIVE_TELEMETRY + return _ACTIVE_PROGRESS -class _TerminalBatchTelemetry: +class _TerminalBatchProgress: """Turn terminal-batch telemetry crashes into release-gate failures. The proxy is deliberately scoped to the post-diagnostics terminal batch. @@ -12246,7 +12250,7 @@ class _TerminalBatchTelemetry: def __init__( self, - telemetry: _BuildTelemetry | None, + telemetry: _BuildProgress | None, terminal_gate_failures: list[str], ) -> None: self._telemetry = telemetry @@ -12344,9 +12348,9 @@ def main(argv: Sequence[str] | None = None) -> None: failure_class="dry_run_refusal", ) raise SystemExit(dry_run.refused(error)) from error - if _ACTIVE_TELEMETRY is not None: + if _ACTIVE_PROGRESS is not None: try: - _ACTIVE_TELEMETRY.fail(error) + _ACTIVE_PROGRESS.fail(error) except Exception as telemetry_error: # pragma: no cover - defensive # A failing failure-report must not replace the real traceback. print( @@ -12404,9 +12408,9 @@ def _check_committed_us_ledger_feed_pin( def _main(argv: Sequence[str] | None = None) -> int | None: - global _ACTIVE_EMITTER, _ACTIVE_TELEMETRY + global _ACTIVE_EMITTER, _ACTIVE_PROGRESS _ACTIVE_EMITTER = None - _ACTIVE_TELEMETRY = None + _ACTIVE_PROGRESS = None args = _parse_args(argv) build_started = time.perf_counter() attempt_started_at = datetime.now(UTC) @@ -12724,7 +12728,7 @@ def _main(argv: Sequence[str] | None = None) -> int | None: # A dry run writes only its report under --out and creates no staging # artifacts. Hosted telemetry identifies it separately from a release. if dry_run is None: - telemetry = _staging_telemetry( + telemetry = _build_progress( args, release_root=release_root, release_id=release_id, @@ -12732,12 +12736,13 @@ def _main(argv: Sequence[str] | None = None) -> int | None: emitter=_ACTIVE_EMITTER, ) else: - telemetry = _EmitterOnlyTelemetry( + telemetry = _BuildProgress( run_id=run_id, emitter=_ACTIVE_EMITTER, - reason="dry run", + staging_bundle=None, + staging_opt_out_reason="dry run", ) - _ACTIVE_TELEMETRY = telemetry + _ACTIVE_PROGRESS = telemetry if telemetry is not None: telemetry.stage( "target_registry", @@ -15261,7 +15266,7 @@ def _main(argv: Sequence[str] | None = None) -> int | None: # An unbuildable post-export plan refuses here, before the export write, # with every other terminal group still evaluated (microcosm#956). terminal_gate_failures.extend(post_export_scoring_plan.terminal_failures()) - terminal_batch_telemetry = _TerminalBatchTelemetry( + terminal_batch_telemetry = _TerminalBatchProgress( telemetry, terminal_gate_failures, ) @@ -15620,7 +15625,7 @@ def _main(argv: Sequence[str] | None = None) -> int | None: # The owned failures ride into the release manifest's known_failures # block instead of aborting the export; the H5 written below carries # the calibrated weights, so the sidecar is not written on this path. - # Failures appended AFTER this point (a _TerminalBatchTelemetry crash + # Failures appended AFTER this point (a _TerminalBatchProgress crash # line, the smoke/take-up/coverage recordings) are owner-checked at # their own append sites and again before the manifest write; if one # is unowned the run dies post-H5 — weights retained in the written diff --git a/tools/generate_staging_contract_fixtures.py b/tools/generate_staging_contract_fixtures.py index a1733a584..71563c8cd 100644 --- a/tools/generate_staging_contract_fixtures.py +++ b/tools/generate_staging_contract_fixtures.py @@ -9,7 +9,7 @@ import tempfile from pathlib import Path -from microcosm.build.staging_v2 import StagingTelemetryV2 +from microcosm.build.staging_v2 import StagingRunBundleWriterV2 ROOT = Path(__file__).resolve().parents[1] DEFAULT_OUTPUT = ( @@ -29,7 +29,7 @@ def __call__(self) -> str: def _completed_spine(root: Path) -> None: - recorder = StagingTelemetryV2( + recorder = StagingRunBundleWriterV2( run_id="uk-spine-v2-fixture", country_code="GB", operation_id="uk_frs_spine", @@ -54,7 +54,7 @@ def _completed_spine(root: Path) -> None: def _calibration(root: Path) -> None: - recorder = StagingTelemetryV2( + recorder = StagingRunBundleWriterV2( run_id="uk-calibration-v2-fixture", country_code="GB", operation_id="uk_national_calibration", @@ -86,7 +86,7 @@ def _calibration(root: Path) -> None: def _failed(root: Path) -> None: - recorder = StagingTelemetryV2( + recorder = StagingRunBundleWriterV2( run_id="uk-failed-v2-fixture", country_code="GB", operation_id="uk_frs_spine", From ddcdecf8cbeabca16b1550d6eb45d6f7a9f26712 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:37:36 +0400 Subject: [PATCH 7/8] Refactor telemetry emitter service modules --- .../src/microcosm/build/telemetry_emitter.py | 299 +++---- .../build/telemetry_emitter_constants.py | 38 + .../build/telemetry_emitter_service.py | 844 ------------------ .../telemetry_emitter_service/__init__.py | 17 + .../telemetry_emitter_service/__main__.py | 5 + .../telemetry_emitter_service/collector.py | 273 ++++++ .../telemetry_emitter_service/constants.py | 69 ++ .../build/telemetry_emitter_service/main.py | 53 ++ .../telemetry_emitter_service/resources.py | 137 +++ .../telemetry_emitter_service/runtime.py | 215 +++++ .../build/telemetry_emitter_service/spool.py | 342 +++++++ .../telemetry_emitter_service/timestamps.py | 11 + .../src/microcosm/build/telemetry_identity.py | 53 ++ .../src/microcosm/build/telemetry_protocol.py | 66 ++ .../microcosm/build/telemetry_sanitization.py | 94 ++ .../shared/test_telemetry_emitter.py | 63 +- 16 files changed, 1533 insertions(+), 1046 deletions(-) create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_constants.py delete mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__init__.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__main__.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/collector.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/constants.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/main.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/resources.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/runtime.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/timestamps.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_identity.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_protocol.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_sanitization.py diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py index 95151bf83..b7d97a6bd 100644 --- a/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter.py @@ -9,7 +9,6 @@ import json import os -import re import socket import subprocess import sys @@ -19,21 +18,59 @@ from collections.abc import Mapping from dataclasses import dataclass, field from datetime import UTC, datetime -from importlib import metadata from pathlib import Path -from platform import platform -from typing import Any, Literal - -_SENSITIVE_KEY_PARTS = ( - "authorization", - "credential", - "password", - "secret", - "token", - "traceback", +from typing import Any + +from microcosm.build.telemetry_emitter_constants import ( + DEFAULT_HEARTBEAT_SECONDS, + DEFAULT_SEND_TIMEOUT_SECONDS, + DEFAULT_STARTUP_TIMEOUT_SECONDS, + INVALID_ACKNOWLEDGEMENT_ERROR, + NO_LOCAL_SOCKET_WARNING, + QUEUE_WARNING, + RUNTIME_DIRECTORY_PREFIX, + SERVICE_NOT_READY_WARNING, + SERVICE_PROCESS_EXIT_TIMEOUT_SECONDS, + SERVICE_READY_TIMEOUT_SECONDS, + SERVICE_START_WARNING, + SOCKET_FILENAME, + STARTUP_POLL_SECONDS, + TELEMETRY_CACHE_PARTS, + TELEMETRY_SERVICE_MODULE, + TELEMETRY_SPOOL_FILENAME, + TEMPORARY_DIRECTORY_ALIAS, ) -_SECRET_TEXT = re.compile( - r"(?i)(?:bearer\s+[^\s]+|(?:token|secret|password|credential)\s*[:=]\s*[^\s]+|hf_[A-Za-z0-9_-]{8,})" +from microcosm.build.telemetry_identity import runtime_identity +from microcosm.build.telemetry_protocol import ( + ACTION_CLOSE, + ACTION_EVENT, + BUILD_COMPLETED_MESSAGE, + BUILD_STARTED_MESSAGE, + CALIBRATION_EVENT_KIND, + EVENT_TYPE_CALIBRATION, + EVENT_TYPE_PROGRESS, + EVENT_TYPE_RUN, + EVENT_TYPE_STAGE, + LOCAL_ACKNOWLEDGEMENT_OK, + LOCAL_ACKNOWLEDGEMENT_READ_BYTES, + LOCAL_MESSAGE_DELIMITER, + LOCAL_PING_MESSAGE, + MAX_TELEMETRY_MESSAGE_CHARS, + SEQUENTIAL_STATUS_MAP, + STAGE_CALIBRATING, + STAGE_COMPLETE, + STAGE_CREATED, + STAGE_FAILED, + STATUS_COMPLETED, + STATUS_FAILED, + STATUS_PROGRESS, + STATUS_STARTED, + TelemetryEventType, + TelemetryStatus, +) +from microcosm.build.telemetry_sanitization import ( + sanitize_details, + sanitize_text, ) @@ -44,96 +81,7 @@ def _now() -> str: def _cache_dir() -> Path: configured = os.environ.get("XDG_CACHE_HOME", "").strip() root = Path(configured).expanduser() if configured else Path.home() / ".cache" - return root / "microcosm" / "telemetry" - - -def _safe_text(value: str, *, limit: int = 2_000) -> str: - return _SECRET_TEXT.sub("[redacted]", value)[:limit] - - -def _safe_json(value: Any, *, depth: int = 0) -> Any: - """Return JSON-compatible telemetry without retaining model objects.""" - - if depth >= 6: - return "[maximum depth]" - if isinstance(value, Mapping): - result = {} - for key, item in list(value.items())[:200]: - name = str(key) - if any(part in name.lower() for part in _SENSITIVE_KEY_PARTS): - result[name] = "[redacted]" - else: - result[name] = _safe_json(item, depth=depth + 1) - return result - if isinstance(value, (list, tuple)): - return [_safe_json(item, depth=depth + 1) for item in value[:200]] - if isinstance(value, Path): - return str(value) - if isinstance(value, float): - return value if value == value and abs(value) != float("inf") else None - if isinstance(value, str): - return _safe_text(value) - if isinstance(value, (int, bool)) or value is None: - return value - item = getattr(value, "item", None) - if callable(item): - return _safe_json(item()) - return _safe_text(str(value)) - - -def _safe_details(value: Mapping[str, Any]) -> dict[str, Any]: - sanitized = _safe_json(value) - assert isinstance(sanitized, dict) - if len(json.dumps(sanitized, separators=(",", ":")).encode()) <= 8_192: - return sanitized - compact: dict[str, Any] = {"telemetry_details_truncated": True} - for key, item in sanitized.items(): - if not (isinstance(item, (str, int, float, bool)) or item is None): - continue - candidate = {**compact, key: item} - if len(json.dumps(candidate, separators=(",", ":")).encode()) > 8_192: - break - compact[key] = item - return compact - - -def _run_identity() -> dict[str, Any]: - try: - commit = subprocess.run( - ["git", "rev-parse", "HEAD"], - check=True, - capture_output=True, - text=True, - timeout=1, - ).stdout.strip() - except (OSError, subprocess.SubprocessError): - commit = None - try: - memory_bytes = int(os.sysconf("SC_PHYS_PAGES")) * int( - os.sysconf("SC_PAGE_SIZE") - ) - except (AttributeError, OSError, ValueError): - memory_bytes = None - versions = {} - for distribution in ( - "microcosm-build", - "microcosm-graph", - "policyengine-us", - "policyengine-uk", - ): - try: - versions[distribution] = metadata.version(distribution) - except metadata.PackageNotFoundError: - continue - return { - "git_commit": commit, - "host": { - "platform": platform(), - "cpu_count": os.cpu_count(), - "memory_bytes": memory_bytes, - }, - "runtime": versions, - } + return root.joinpath(*TELEMETRY_CACHE_PARTS) @dataclass(frozen=True) @@ -170,7 +118,7 @@ def __init__( process: subprocess.Popen[bytes] | None, socket_path: Path | None, runtime_dir: Path | None, - send_timeout_seconds: float = 0.2, + send_timeout_seconds: float = DEFAULT_SEND_TIMEOUT_SECONDS, ) -> None: self.run = run self._process = process @@ -192,8 +140,8 @@ def start( release_id: str | None = None, run_kind: str = "build", development_collector_url: str | None = None, - heartbeat_seconds: float = 60.0, - startup_timeout_seconds: float = 3.0, + heartbeat_seconds: float = DEFAULT_HEARTBEAT_SECONDS, + startup_timeout_seconds: float = DEFAULT_STARTUP_TIMEOUT_SECONDS, spool_path: Path | str | None = None, ) -> LocalTelemetryEmitter: """Start the service, returning a harmless disabled handle on failure.""" @@ -207,26 +155,26 @@ def start( run_kind=run_kind, ) if not hasattr(socket, "AF_UNIX"): - print( - "warning: Microcosm telemetry is unavailable because this " - "platform has no local Unix sockets.", - file=sys.stderr, - ) + print(NO_LOCAL_SOCKET_WARNING, file=sys.stderr) return cls(run=run, process=None, socket_path=None, runtime_dir=None) # macOS limits AF_UNIX paths to roughly 100 bytes. Its default # temporary directory is already long, so use the short system alias. - temporary_root = Path("/tmp") if Path("/tmp").is_dir() else None + temporary_root = ( + TEMPORARY_DIRECTORY_ALIAS if TEMPORARY_DIRECTORY_ALIAS.is_dir() else None + ) runtime_dir = Path( - tempfile.mkdtemp(prefix="microcosm-telemetry-", dir=temporary_root) + tempfile.mkdtemp(prefix=RUNTIME_DIRECTORY_PREFIX, dir=temporary_root) ) runtime_dir.chmod(0o700) - socket_path = runtime_dir / "emitter.sock" - queue_path = Path(spool_path) if spool_path else _cache_dir() / "events.sqlite3" + socket_path = runtime_dir / SOCKET_FILENAME + queue_path = ( + Path(spool_path) if spool_path else _cache_dir() / TELEMETRY_SPOOL_FILENAME + ) command = [ sys.executable, "-m", - "microcosm.build.telemetry_emitter_service", + TELEMETRY_SERVICE_MODULE, "--socket", str(socket_path), "--spool", @@ -258,28 +206,26 @@ def start( runtime_dir=runtime_dir, ) emitter.emit( - event_type="run", - stage_id="created", - status="started", - message="Microcosm build started.", - details={"identity": _run_identity()}, + event_type=EVENT_TYPE_RUN, + stage_id=STAGE_CREATED, + status=STATUS_STARTED, + message=BUILD_STARTED_MESSAGE, + details={"identity": runtime_identity()}, ) return emitter if process.poll() is not None: break - time.sleep(0.02) + time.sleep(STARTUP_POLL_SECONDS) except Exception as error: print( - "warning: the local telemetry emitter service could not start: " - f"{type(error).__name__}: {error}", + SERVICE_START_WARNING.format( + error_type=type(error).__name__, + error=error, + ), file=sys.stderr, ) else: - print( - "warning: the local telemetry emitter service did not become ready; " - "the build will continue without hosted telemetry.", - file=sys.stderr, - ) + print(SERVICE_NOT_READY_WARNING, file=sys.stderr) if process is not None: _terminate_process(process) try: @@ -296,8 +242,8 @@ def available(self) -> bool: def emit( self, *, - event_type: Literal["run", "stage", "progress", "calibration", "heartbeat"], - status: Literal["started", "progress", "completed", "failed"], + event_type: TelemetryEventType, + status: TelemetryStatus, stage_id: str | None = None, message: str | None = None, details: Mapping[str, Any] | None = None, @@ -308,14 +254,18 @@ def emit( return self._send( { - "action": "event", + "action": ACTION_EVENT, "event": { "timestamp": _now(), "event_type": event_type, "stage_id": stage_id, "status": status, - "message": _safe_text(message, limit=500) if message else None, - "details": _safe_details(details or {}), + "message": ( + sanitize_text(message, limit=MAX_TELEMETRY_MESSAGE_CHARS) + if message + else None + ), + "details": sanitize_details(details or {}), }, } ) @@ -324,12 +274,12 @@ def stage( self, stage_id: str, *, - status: Literal["started", "progress", "completed", "failed"] = "started", + status: TelemetryStatus = STATUS_STARTED, message: str | None = None, **details: Any, ) -> None: self.emit( - event_type="stage", + event_type=EVENT_TYPE_STAGE, stage_id=stage_id, status=status, message=message, @@ -346,19 +296,12 @@ def transition_stage( ) -> None: """Translate a sequential stage update into explicit lifecycle events.""" - collector_status = { - "failed": "failed", - "completed": "completed", - "passed": "completed", - "progress": "progress", - "started": "started", - "running": "started", - }.get(status, "progress") - if collector_status == "started": + collector_status = SEQUENTIAL_STATUS_MAP.get(status, STATUS_PROGRESS) + if collector_status == STATUS_STARTED: if self._transition_stage == stage_id: self.stage( stage_id, - status="progress", + status=STATUS_PROGRESS, message=message, **details, ) @@ -386,29 +329,29 @@ def progress( **details: Any, ) -> None: self.emit( - event_type="progress", + event_type=EVENT_TYPE_PROGRESS, stage_id=stage_id, - status="progress", + status=STATUS_PROGRESS, details={"done": done, "total": total, "unit": unit, **details}, ) def calibration_progress(self, event: Mapping[str, Any]) -> None: - if event.get("kind") != "calibration_epoch": + if event.get("kind") != CALIBRATION_EVENT_KIND: return self.emit( - event_type="calibration", - stage_id="calibrating", - status="progress", + event_type=EVENT_TYPE_CALIBRATION, + stage_id=STAGE_CALIBRATING, + status=STATUS_PROGRESS, details=event, ) def transition_calibration_progress(self, event: Mapping[str, Any]) -> None: """Enter the sequential calibration stage, then report one epoch.""" - if event.get("kind") != "calibration_epoch": + if event.get("kind") != CALIBRATION_EVENT_KIND: return - if self._transition_stage != "calibrating": - self.transition_stage("calibrating") + if self._transition_stage != STAGE_CALIBRATING: + self.transition_stage(STAGE_CALIBRATING) self.calibration_progress(event) def fail( @@ -421,10 +364,10 @@ def fail( failed_stage = failed_during or self._transition_stage self._close_transition_stage() self.emit( - event_type="run", - stage_id="failed", - status="failed", - message=str(error)[:500], + event_type=EVENT_TYPE_RUN, + stage_id=STAGE_FAILED, + status=STATUS_FAILED, + message=str(error)[:MAX_TELEMETRY_MESSAGE_CHARS], details={ "error_type": type(error).__name__, "failure_class": failure_class, @@ -436,10 +379,10 @@ def fail( def complete(self) -> None: self._close_transition_stage() self.emit( - event_type="run", - stage_id="complete", - status="completed", - message="Microcosm build completed.", + event_type=EVENT_TYPE_RUN, + stage_id=STAGE_COMPLETE, + status=STATUS_COMPLETED, + message=BUILD_COMPLETED_MESSAGE, ) self.close() @@ -448,7 +391,7 @@ def close(self) -> None: if self._closed: return - self._send({"action": "close"}) + self._send({"action": ACTION_CLOSE}) self._closed = True def _close_transition_stage(self) -> None: @@ -456,7 +399,7 @@ def _close_transition_stage(self) -> None: return stage_id = self._transition_stage self._transition_stage = None - self.stage(stage_id, status="completed") + self.stage(stage_id, status=STATUS_COMPLETED) def _send(self, payload: Mapping[str, Any]) -> None: if self._socket_path is None: @@ -467,16 +410,15 @@ def _send(self, payload: Mapping[str, Any]) -> None: client.connect(str(self._socket_path)) client.sendall( json.dumps(payload, separators=(",", ":"), allow_nan=False).encode() - + b"\n" + + LOCAL_MESSAGE_DELIMITER ) - acknowledgement = client.recv(16) - if acknowledgement != b"ok\n": - raise OSError("invalid acknowledgement") + acknowledgement = client.recv(LOCAL_ACKNOWLEDGEMENT_READ_BYTES) + if acknowledgement != LOCAL_ACKNOWLEDGEMENT_OK: + raise OSError(INVALID_ACKNOWLEDGEMENT_ERROR) except Exception as error: if not self._warned: print( - "warning: the local telemetry emitter service could not queue " - f"an update ({type(error).__name__}); the build will continue.", + QUEUE_WARNING.format(error_type=type(error).__name__), file=sys.stderr, ) self._warned = True @@ -493,10 +435,13 @@ def start_local_telemetry_emitter_service( def _service_ready(socket_path: Path) -> bool: try: with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as client: - client.settimeout(0.1) + client.settimeout(SERVICE_READY_TIMEOUT_SECONDS) client.connect(str(socket_path)) - client.sendall(b'{"action":"ping"}\n') - return client.recv(16) == b"ok\n" + client.sendall(LOCAL_PING_MESSAGE) + return ( + client.recv(LOCAL_ACKNOWLEDGEMENT_READ_BYTES) + == LOCAL_ACKNOWLEDGEMENT_OK + ) except OSError: return False @@ -507,9 +452,9 @@ def _terminate_process(process: subprocess.Popen[bytes]) -> None: try: if process.poll() is None: process.terminate() - process.wait(timeout=1.0) + process.wait(timeout=SERVICE_PROCESS_EXIT_TIMEOUT_SECONDS) except subprocess.TimeoutExpired: process.kill() - process.wait(timeout=1.0) + process.wait(timeout=SERVICE_PROCESS_EXIT_TIMEOUT_SECONDS) except OSError: pass diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_constants.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_constants.py new file mode 100644 index 000000000..57a477f82 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_constants.py @@ -0,0 +1,38 @@ +"""Configuration and user-facing messages for the telemetry emitter client.""" + +from __future__ import annotations + +from pathlib import Path +from typing import Final + +TELEMETRY_SERVICE_MODULE: Final = "microcosm.build.telemetry_emitter_service" +TELEMETRY_CACHE_PARTS: Final = ("microcosm", "telemetry") +TELEMETRY_SPOOL_FILENAME: Final = "events.sqlite3" +RUNTIME_DIRECTORY_PREFIX: Final = "microcosm-telemetry-" +SOCKET_FILENAME: Final = "emitter.sock" +TEMPORARY_DIRECTORY_ALIAS: Final = Path("/tmp") + +DEFAULT_SEND_TIMEOUT_SECONDS: Final = 0.2 +DEFAULT_HEARTBEAT_SECONDS: Final = 60.0 +DEFAULT_STARTUP_TIMEOUT_SECONDS: Final = 3.0 +STARTUP_POLL_SECONDS: Final = 0.02 +SERVICE_READY_TIMEOUT_SECONDS: Final = 0.1 +SERVICE_PROCESS_EXIT_TIMEOUT_SECONDS: Final = 1.0 + +NO_LOCAL_SOCKET_WARNING: Final = ( + "warning: Microcosm telemetry is unavailable because this platform has no " + "local Unix sockets." +) +SERVICE_START_WARNING: Final = ( + "warning: the local telemetry emitter service could not start: " + "{error_type}: {error}" +) +SERVICE_NOT_READY_WARNING: Final = ( + "warning: the local telemetry emitter service did not become ready; " + "the build will continue without hosted telemetry." +) +QUEUE_WARNING: Final = ( + "warning: the local telemetry emitter service could not queue an update " + "({error_type}); the build will continue." +) +INVALID_ACKNOWLEDGEMENT_ERROR: Final = "invalid acknowledgement" diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py deleted file mode 100644 index 05be6b9d5..000000000 --- a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service.py +++ /dev/null @@ -1,844 +0,0 @@ -"""Standalone local service that delivers Microcosm telemetry to a collector.""" - -from __future__ import annotations - -import argparse -import json -import os -import socket -import sqlite3 -import sys -import threading -import time -import urllib.error -import urllib.request -import uuid -from collections.abc import Mapping -from datetime import UTC, datetime, timedelta -from pathlib import Path -from typing import Any -from urllib.parse import urlsplit - -from huggingface_hub import get_token - -try: - import psutil -except ModuleNotFoundError: # Base installs use the standard-library fallback. - psutil = None - -SCHEMA_VERSION = 1 -RETENTION_DAYS = 7 -MAX_QUEUED_BYTES = 100 * 1024 * 1024 -BATCH_SIZE = 100 -PRODUCTION_COLLECTOR_URL = ( - "https://microcosm-telemetry-389282473430.us-central1.run.app" -) - - -def _now() -> str: - return datetime.now(UTC).isoformat() - - -class _NoRedirectHandler(urllib.request.HTTPRedirectHandler): - """Keep bearer credentials on the explicitly configured origin.""" - - def redirect_request(self, req, fp, code, msg, headers, newurl): - return None - - -def _collector_origin(value: str, *, allow_loopback_http: bool = False) -> str: - parsed = urlsplit(value) - loopback = parsed.hostname in {"localhost", "127.0.0.1", "::1"} - valid_scheme = parsed.scheme == "https" or ( - allow_loopback_http and loopback and parsed.scheme == "http" - ) - if not parsed.hostname or not valid_scheme: - raise ValueError("collector URL must be an HTTPS origin") - if ( - parsed.username - or parsed.password - or parsed.query - or parsed.fragment - or parsed.path not in {"", "/"} - ): - raise ValueError( - "collector URL must be an origin without credentials or path data" - ) - return value.rstrip("/") - - -def _development_collector_url(value: str) -> str: - parsed = urlsplit(value) - if parsed.hostname not in {"localhost", "127.0.0.1", "::1"}: - raise ValueError("development collector URL must use a loopback address") - return _collector_origin(value, allow_loopback_http=True) - - -class EventSpool: - """Small SQLite queue shared by successive emitter service processes.""" - - def __init__(self, path: Path | str) -> None: - self.path = Path(path) - self.path.parent.mkdir(parents=True, exist_ok=True) - try: - self.path.parent.chmod(0o700) - except OSError: - pass - self._connection = sqlite3.connect( - self.path, timeout=5, check_same_thread=False - ) - self._connection.row_factory = sqlite3.Row - self._lock = threading.RLock() - self._last_prune_at = 0.0 - with self._lock, self._connection: - self._connection.execute("PRAGMA journal_mode=WAL") - self._connection.execute("PRAGMA synchronous=NORMAL") - self._connection.executescript( - """ - CREATE TABLE IF NOT EXISTS telemetry_runs ( - run_id TEXT NOT NULL, - producer_id TEXT NOT NULL, - registration_json TEXT NOT NULL, - next_sequence INTEGER NOT NULL DEFAULT 1, - upload_state TEXT NOT NULL DEFAULT 'pending', - local_only_reason TEXT, - updated_at TEXT NOT NULL, - PRIMARY KEY(run_id, producer_id) - ); - CREATE TABLE IF NOT EXISTS telemetry_events ( - event_id TEXT PRIMARY KEY, - run_id TEXT NOT NULL, - producer_id TEXT NOT NULL, - sequence INTEGER NOT NULL, - payload_json TEXT NOT NULL, - created_at TEXT NOT NULL, - UNIQUE(run_id, producer_id, sequence), - FOREIGN KEY(run_id, producer_id) - REFERENCES telemetry_runs(run_id, producer_id) - ); - CREATE INDEX IF NOT EXISTS telemetry_events_run_sequence - ON telemetry_events(run_id, producer_id, sequence); - """ - ) - columns = { - row["name"] - for row in self._connection.execute( - "PRAGMA table_info(telemetry_runs)" - ).fetchall() - } - if "upload_state" not in columns: - self._connection.execute( - "ALTER TABLE telemetry_runs ADD COLUMN " - "upload_state TEXT NOT NULL DEFAULT 'pending'" - ) - self._connection.execute( - "UPDATE telemetry_runs SET upload_state = 'local_only'" - ) - if "local_only_reason" not in columns: - self._connection.execute( - "ALTER TABLE telemetry_runs ADD COLUMN local_only_reason TEXT" - ) - self._connection.execute( - "UPDATE telemetry_runs SET local_only_reason = " - "'created_before_upload_eligibility' " - "WHERE upload_state = 'local_only'" - ) - self.prune() - - def register(self, registration: Mapping[str, Any]) -> None: - run_id = str(registration["run_id"]) - producer_id = str(registration["producer_id"]) - encoded = json.dumps(registration, separators=(",", ":"), sort_keys=True) - with self._lock, self._connection: - existing = self._connection.execute( - """ - SELECT registration_json FROM telemetry_runs - WHERE run_id = ? AND producer_id = ? - """, - (run_id, producer_id), - ).fetchone() - if existing is not None: - old = json.loads(existing["registration_json"]) - if old.get("producer_id") != registration.get("producer_id"): - raise ValueError(f"run_id {run_id!r} already has another producer") - self._connection.execute( - """ - INSERT INTO telemetry_runs ( - run_id, producer_id, registration_json, next_sequence, updated_at - ) VALUES (?, ?, ?, 1, ?) - ON CONFLICT(run_id, producer_id) DO UPDATE SET - registration_json = excluded.registration_json, - updated_at = excluded.updated_at - """, - (run_id, producer_id, encoded, _now()), - ) - - def append( - self, - registration: Mapping[str, Any], - event: Mapping[str, Any], - *, - resources: Mapping[str, Any] | None = None, - ) -> dict[str, Any]: - run_id = str(registration["run_id"]) - producer_id = str(registration["producer_id"]) - with self._lock, self._connection: - row = self._connection.execute( - """ - SELECT next_sequence FROM telemetry_runs - WHERE run_id = ? AND producer_id = ? - """, - (run_id, producer_id), - ).fetchone() - if row is None: - raise KeyError(run_id) - sequence = int(row["next_sequence"]) - event_id = uuid.uuid4().hex - payload = { - "schema_version": SCHEMA_VERSION, - "event_id": event_id, - "run_id": run_id, - "producer_id": producer_id, - "sequence": sequence, - "timestamp": event.get("timestamp") or _now(), - "event_type": event["event_type"], - "stage_id": event.get("stage_id"), - "status": event["status"], - "message": event.get("message"), - "details": event.get("details") or {}, - "resources": resources, - } - encoded = json.dumps(payload, separators=(",", ":"), allow_nan=False) - self._connection.execute( - """ - INSERT INTO telemetry_events ( - event_id, run_id, producer_id, sequence, - payload_json, created_at - ) VALUES (?, ?, ?, ?, ?, ?) - """, - (event_id, run_id, producer_id, sequence, encoded, _now()), - ) - self._connection.execute( - """ - UPDATE telemetry_runs - SET next_sequence = ?, updated_at = ? - WHERE run_id = ? AND producer_id = ? - """, - (sequence + 1, _now(), run_id, producer_id), - ) - if time.monotonic() - self._last_prune_at >= 60: - self.prune() - return payload - - def pending_runs(self) -> list[dict[str, Any]]: - with self._lock: - rows = self._connection.execute( - """ - SELECT DISTINCT r.registration_json - FROM telemetry_runs r - JOIN telemetry_events e - ON e.run_id = r.run_id AND e.producer_id = r.producer_id - WHERE r.upload_state = 'pending' - ORDER BY r.updated_at - """ - ).fetchall() - return [json.loads(row["registration_json"]) for row in rows] - - def make_local_only( - self, - run_id: str, - producer_id: str, - reason: str, - ) -> None: - """Permanently exclude one producer's queued events from upload.""" - - with self._lock, self._connection: - self._connection.execute( - """ - UPDATE telemetry_runs - SET upload_state = 'local_only', local_only_reason = ?, updated_at = ? - WHERE run_id = ? AND producer_id = ? - """, - (reason, _now(), run_id, producer_id), - ) - - def has_deliverable(self) -> bool: - with self._lock: - row = self._connection.execute( - """ - SELECT 1 - FROM telemetry_events e - JOIN telemetry_runs r - ON r.run_id = e.run_id AND r.producer_id = e.producer_id - WHERE r.upload_state = 'pending' - LIMIT 1 - """ - ).fetchone() - return row is not None - - def batch( - self, run_id: str, producer_id: str, limit: int = BATCH_SIZE - ) -> list[dict[str, Any]]: - with self._lock: - rows = self._connection.execute( - """ - SELECT payload_json FROM telemetry_events - WHERE run_id = ? AND producer_id = ? - ORDER BY sequence LIMIT ? - """, - (run_id, producer_id, limit), - ).fetchall() - return [json.loads(row["payload_json"]) for row in rows] - - def acknowledge(self, event_ids: list[str]) -> None: - if not event_ids: - return - placeholders = ",".join("?" for _ in event_ids) - with self._lock, self._connection: - self._connection.execute( - f"DELETE FROM telemetry_events WHERE event_id IN ({placeholders})", - event_ids, - ) - - def has_pending(self) -> bool: - with self._lock: - row = self._connection.execute( - "SELECT 1 FROM telemetry_events LIMIT 1" - ).fetchone() - return row is not None - - def prune(self) -> None: - cutoff = (datetime.now(UTC) - timedelta(days=RETENTION_DAYS)).isoformat() - with self._lock, self._connection: - self._connection.execute( - "DELETE FROM telemetry_events WHERE created_at < ?", (cutoff,) - ) - size_row = self._connection.execute( - "SELECT COALESCE(SUM(LENGTH(payload_json)), 0) AS bytes " - "FROM telemetry_events" - ).fetchone() - excess = int(size_row["bytes"]) - MAX_QUEUED_BYTES - if excess > 0: - rows = self._connection.execute( - """ - SELECT event_id, LENGTH(payload_json) AS bytes - FROM telemetry_events ORDER BY created_at, sequence - """ - ).fetchall() - removed = 0 - ids: list[str] = [] - for row in rows: - ids.append(row["event_id"]) - removed += int(row["bytes"]) - if removed >= excess: - break - if ids: - placeholders = ",".join("?" for _ in ids) - self._connection.execute( - f"DELETE FROM telemetry_events " - f"WHERE event_id IN ({placeholders})", - ids, - ) - self._connection.execute( - """ - DELETE FROM telemetry_runs - WHERE updated_at < ? AND NOT EXISTS ( - SELECT 1 FROM telemetry_events e - WHERE e.run_id = telemetry_runs.run_id - AND e.producer_id = telemetry_runs.producer_id - ) - """, - (cutoff,), - ) - self._last_prune_at = time.monotonic() - - -def _http_post( - url: str, - payload: Mapping[str, Any], - bearer_token: str, - *, - timeout: float = 5.0, -) -> tuple[int, dict[str, Any]]: - request = urllib.request.Request( - url, - data=json.dumps(payload, separators=(",", ":")).encode(), - headers={ - "Authorization": f"Bearer {bearer_token}", - "Content-Type": "application/json", - "User-Agent": "microcosm-telemetry-emitter/1", - }, - method="POST", - ) - try: - opener = urllib.request.build_opener(_NoRedirectHandler) - with opener.open(request, timeout=timeout) as response: - body = response.read(1_048_577) - if len(body) > 1_048_576: - raise OSError("collector response exceeds 1 MiB") - try: - payload = json.loads(body) if body else {} - except (json.JSONDecodeError, UnicodeDecodeError): - payload = {} - return response.status, payload - except urllib.error.HTTPError as error: - body = error.read(1_048_577) - if len(body) > 1_048_576: - return error.code, {} - try: - detail = json.loads(body) if body else {} - except (json.JSONDecodeError, UnicodeDecodeError): - detail = {} - return error.code, detail - - -def _huggingface_token() -> str | None: - return ( - os.environ.get("HF_TOKEN", "").strip() - or os.environ.get("HUGGINGFACE_TOKEN", "").strip() - or get_token() - ) - - -class CollectorDelivery: - """Authenticate queued runs and deliver idempotent event batches.""" - - def __init__( - self, - spool: EventSpool, - *, - development_collector_url: str | None = None, - ) -> None: - self.collector_url = ( - _development_collector_url(development_collector_url) - if development_collector_url is not None - else _collector_origin(PRODUCTION_COLLECTOR_URL) - ) - self.spool = spool - self._session_token: tuple[str, float] | None = None - self._registered: set[str] = set() - self._warned_no_token = False - self._warned_denied: set[str] = set() - self._next_attempt_at = 0.0 - self._retry_seconds = 1.0 - - def flush_once(self) -> bool: - if time.monotonic() < self._next_attempt_at: - return False - made_progress = False - for registration in self.spool.pending_runs(): - run_id = str(registration["run_id"]) - registration_key = f"{run_id}:{registration['producer_id']}" - token = self._collector_token(registration) - if token is None: - continue - if registration_key not in self._registered: - if not self._register(registration, token): - continue - events = self.spool.batch(run_id, registration["producer_id"]) - if not events: - continue - try: - status, _ = _http_post( - f"{self.collector_url}/v1/runs/{run_id}/events", - {"events": events}, - token, - ) - except (OSError, TimeoutError): - self._defer_retry() - continue - if status == 202: - self.spool.acknowledge([event["event_id"] for event in events]) - made_progress = True - self._retry_seconds = 1.0 - elif status == 401: - self._session_token = None - elif status == 403: - self._make_local_only(registration, "collector_authorization_rejected") - else: - self._defer_retry() - return made_progress - - def _collector_token(self, registration: Mapping[str, Any]) -> str | None: - if ( - self._session_token is not None - and self._session_token[1] > time.monotonic() + 30 - ): - return self._session_token[0] - hf_token = _huggingface_token() - if not hf_token: - if not self._warned_no_token: - print( - "Microcosm telemetry is local-only: no ambient Hugging Face " - "credential was found. The dataset build will continue.", - file=sys.stderr, - flush=True, - ) - self._warned_no_token = True - self._make_local_only(registration, "missing_huggingface_credential") - return None - try: - status, response = _http_post( - f"{self.collector_url}/v1/auth/huggingface/exchange", - {}, - hf_token, - ) - except (OSError, TimeoutError): - self._defer_retry() - return None - if status == 200 and isinstance(response.get("access_token"), str): - expires_in = max(60, int(response.get("expires_in", 3600))) - token = response["access_token"] - self._session_token = (token, time.monotonic() + expires_in) - return token - if status in {401, 403}: - self._make_local_only(registration, "huggingface_credential_rejected") - else: - self._defer_retry() - return None - - def _register(self, registration: Mapping[str, Any], token: str) -> bool: - registration_key = f"{registration['run_id']}:{registration['producer_id']}" - try: - status, _ = _http_post( - f"{self.collector_url}/v1/runs", - registration, - token, - ) - except (OSError, TimeoutError): - self._defer_retry() - return False - if status == 201: - self._registered.add(registration_key) - self._retry_seconds = 1.0 - return True - if status == 401: - self._session_token = None - elif status in {403, 409}: - self._make_local_only(registration, "run_registration_rejected") - else: - self._defer_retry() - return False - - def _make_local_only( - self, - registration: Mapping[str, Any], - reason: str, - ) -> None: - run_id = str(registration["run_id"]) - producer_id = str(registration["producer_id"]) - registration_key = f"{run_id}:{producer_id}" - self.spool.make_local_only(run_id, producer_id, reason) - if reason != "missing_huggingface_credential": - self._warn_denied(registration_key) - - def _defer_retry(self) -> None: - self._next_attempt_at = time.monotonic() + self._retry_seconds - self._retry_seconds = min(60.0, self._retry_seconds * 2) - - def _warn_denied(self, run_id: str) -> None: - if run_id in self._warned_denied: - return - print( - "Microcosm telemetry is local-only for this run: the ambient " - "Hugging Face credential was not accepted as a PolicyEngine " - "organization member. The dataset build will continue.", - file=sys.stderr, - flush=True, - ) - self._warned_denied.add(run_id) - - -class ProcessTreeSampler: - """Collect cumulative CPU and resident memory for a build process tree.""" - - def __init__(self, parent_pid: int) -> None: - self.parent_pid = parent_pid - self._peak_rss = 0 - self._parent_create_time = self._create_time() - - def _create_time(self) -> float | str | None: - if psutil is not None: - try: - return psutil.Process(self.parent_pid).create_time() - except (psutil.Error, OSError): - return None - stat_path = Path(f"/proc/{self.parent_pid}/stat") - try: - return stat_path.read_text().rsplit(")", 1)[1].split()[19] - except (OSError, IndexError): - return None - - def parent_alive(self) -> bool: - if psutil is not None: - try: - process = psutil.Process(self.parent_pid) - return ( - self._parent_create_time is not None - and process.create_time() == self._parent_create_time - and process.is_running() - and process.status() != psutil.STATUS_ZOMBIE - ) - except (psutil.Error, OSError): - return False - try: - os.kill(self.parent_pid, 0) - except OSError: - return False - current = self._create_time() - # /proc supplies a process-start tick that protects against PID reuse. - # On platforms without it, existence is the best standard-library test. - return self._parent_create_time is None or current == self._parent_create_time - - def sample(self) -> dict[str, Any]: - if psutil is None: - return self._fallback_sample() - processes = [] - try: - parent = psutil.Process(self.parent_pid) - processes = [parent, *parent.children(recursive=True)] - except (psutil.Error, OSError): - pass - user = 0.0 - system = 0.0 - rss = 0 - for process in processes: - try: - cpu = process.cpu_times() - user += float(cpu.user) + float(getattr(cpu, "children_user", 0.0)) - system += float(cpu.system) + float( - getattr(cpu, "children_system", 0.0) - ) - rss += int(process.memory_info().rss) - except (psutil.Error, OSError): - continue - self._peak_rss = max(self._peak_rss, rss) - return { - "cpu_user_seconds": user, - "cpu_system_seconds": system, - "rss_bytes": rss, - "peak_rss_bytes": self._peak_rss, - } - - def _fallback_sample(self) -> dict[str, Any]: - """Sample Linux process trees when the country engine is not installed.""" - - process_rows: dict[int, tuple[int, float, float, int]] = {} - try: - clock_ticks = float(os.sysconf("SC_CLK_TCK")) - page_size = int(os.sysconf("SC_PAGE_SIZE")) - except (AttributeError, OSError, ValueError): - return { - "cpu_user_seconds": 0.0, - "cpu_system_seconds": 0.0, - "rss_bytes": 0, - "peak_rss_bytes": self._peak_rss, - } - for stat_path in Path("/proc").glob("[0-9]*/stat"): - try: - pid = int(stat_path.parent.name) - fields = stat_path.read_text().rsplit(")", 1)[1].split() - process_rows[pid] = ( - int(fields[1]), - int(fields[11]) / clock_ticks, - int(fields[12]) / clock_ticks, - int(fields[21]) * page_size, - ) - except (OSError, ValueError, IndexError): - continue - selected = {self.parent_pid} - changed = True - while changed: - changed = False - for pid, (parent, *_rest) in process_rows.items(): - if parent in selected and pid not in selected: - selected.add(pid) - changed = True - user = sum(process_rows[pid][1] for pid in selected if pid in process_rows) - system = sum(process_rows[pid][2] for pid in selected if pid in process_rows) - rss = sum(process_rows[pid][3] for pid in selected if pid in process_rows) - self._peak_rss = max(self._peak_rss, rss) - return { - "cpu_user_seconds": user, - "cpu_system_seconds": system, - "rss_bytes": rss, - "peak_rss_bytes": self._peak_rss, - } - - -class EmitterService: - """Private local socket server with a concurrent delivery worker.""" - - def __init__( - self, - *, - socket_path: Path, - registration: Mapping[str, Any], - spool: EventSpool, - delivery: CollectorDelivery, - sampler: ProcessTreeSampler, - heartbeat_seconds: float, - drain_seconds: float = 15.0, - ) -> None: - self.socket_path = socket_path - self.registration = dict(registration) - self.spool = spool - self.delivery = delivery - self.sampler = sampler - self.heartbeat_seconds = max(1.0, heartbeat_seconds) - self.drain_seconds = max(0.0, drain_seconds) - self._stop = threading.Event() - self._closed_by_client = False - self._last_stage = "created" - - def run(self) -> None: - self.spool.register(self.registration) - self.socket_path.parent.mkdir(parents=True, exist_ok=True) - try: - self.socket_path.unlink(missing_ok=True) - with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as server: - server.bind(str(self.socket_path)) - os.chmod(self.socket_path, 0o600) - server.listen(16) - server.settimeout(0.5) - worker = threading.Thread(target=self._worker, daemon=True) - worker.start() - while not self._stop.is_set(): - try: - connection, _ = server.accept() - except TimeoutError: - continue - with connection: - connection.settimeout(0.25) - try: - message = self._read_message(connection) - self._handle(message) - response = b"ok\n" - except Exception: - response = b"error\n" - try: - connection.sendall(response) - except OSError: - pass - worker.join(timeout=self.drain_seconds + 1) - finally: - self.socket_path.unlink(missing_ok=True) - try: - self.socket_path.parent.rmdir() - except OSError: - pass - - @staticmethod - def _read_message(connection: socket.socket) -> dict[str, Any]: - chunks = bytearray() - while len(chunks) <= 1_048_576: - data = connection.recv(65536) - if not data: - break - chunks.extend(data) - if b"\n" in data: - break - if len(chunks) > 1_048_576: - raise ValueError("local telemetry message exceeds 1 MiB") - return json.loads(bytes(chunks).split(b"\n", 1)[0]) - - def _handle(self, message: Mapping[str, Any]) -> None: - action = message.get("action") - if action == "event": - event = message.get("event") - if not isinstance(event, Mapping): - raise ValueError("event must be an object") - stage_id = event.get("stage_id") - if isinstance(stage_id, str) and stage_id not in {"complete", "failed"}: - self._last_stage = stage_id - self.spool.append( - self.registration, - event, - resources=self.sampler.sample(), - ) - elif action == "close": - self._closed_by_client = True - self._stop.set() - elif action == "ping": - return - else: - raise ValueError("unsupported local telemetry action") - - def _worker(self) -> None: - next_heartbeat = time.monotonic() + self.heartbeat_seconds - while not self._stop.wait(1.0): - now = time.monotonic() - # Sample every worker iteration so short-lived build children are - # much less likely to disappear between stage and heartbeat events. - self.sampler.sample() - if now >= next_heartbeat: - self.spool.append( - self.registration, - { - "timestamp": _now(), - "event_type": "heartbeat", - "stage_id": self._last_stage, - "status": "progress", - "message": None, - "details": {}, - }, - resources=self.sampler.sample(), - ) - next_heartbeat = now + self.heartbeat_seconds - self.delivery.flush_once() - if not self.sampler.parent_alive(): - self.spool.append( - self.registration, - { - "timestamp": _now(), - "event_type": "run", - "stage_id": "failed", - "status": "failed", - "message": "The build process exited without reporting completion.", - "details": { - "failure_class": "unexpected_process_exit", - "failed_during": self._last_stage, - }, - }, - resources=self.sampler.sample(), - ) - self._stop.set() - break - deadline = time.monotonic() + self.drain_seconds - while self.spool.has_deliverable() and time.monotonic() < deadline: - if not self.delivery.flush_once(): - self._stop.wait(0.5) - - -def _parser() -> argparse.ArgumentParser: - parser = argparse.ArgumentParser() - parser.add_argument("--socket", type=Path, required=True) - parser.add_argument("--spool", type=Path, required=True) - parser.add_argument("--development-collector-url") - parser.add_argument("--registration-json", required=True) - parser.add_argument("--parent-pid", type=int, required=True) - parser.add_argument("--heartbeat-seconds", type=float, default=60.0) - return parser - - -def main(argv: list[str] | None = None) -> int: - args = _parser().parse_args(argv) - registration = json.loads(args.registration_json) - spool = EventSpool(args.spool) - service = EmitterService( - socket_path=args.socket, - registration=registration, - spool=spool, - delivery=CollectorDelivery( - spool, - development_collector_url=args.development_collector_url, - ), - sampler=ProcessTreeSampler(args.parent_pid), - heartbeat_seconds=args.heartbeat_seconds, - ) - service.run() - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__init__.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__init__.py new file mode 100644 index 000000000..ef5335561 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__init__.py @@ -0,0 +1,17 @@ +"""Local service that queues and delivers Microcosm telemetry.""" + +from microcosm.build.telemetry_emitter_service.collector import CollectorDelivery +from microcosm.build.telemetry_emitter_service.constants import ( + PRODUCTION_COLLECTOR_URL, +) +from microcosm.build.telemetry_emitter_service.resources import ProcessTreeSampler +from microcosm.build.telemetry_emitter_service.runtime import EmitterService +from microcosm.build.telemetry_emitter_service.spool import EventSpool + +__all__ = [ + "CollectorDelivery", + "EmitterService", + "EventSpool", + "PRODUCTION_COLLECTOR_URL", + "ProcessTreeSampler", +] diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__main__.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__main__.py new file mode 100644 index 000000000..298103831 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/__main__.py @@ -0,0 +1,5 @@ +"""Module execution entry point for the telemetry emitter service.""" + +from microcosm.build.telemetry_emitter_service.main import main + +raise SystemExit(main()) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/collector.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/collector.py new file mode 100644 index 000000000..795c61a17 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/collector.py @@ -0,0 +1,273 @@ +"""Collector authentication and best-effort event delivery.""" + +from __future__ import annotations + +import json +import os +import sys +import time +import urllib.error +import urllib.request +from collections.abc import Mapping +from http import HTTPStatus +from typing import Any +from urllib.parse import urlsplit + +from huggingface_hub import get_token + +from microcosm.build.telemetry_emitter_service.constants import ( + COLLECTOR_RESPONSE_TOO_LARGE_ERROR, + COLLECTOR_URL_HTTPS_ERROR, + COLLECTOR_URL_ORIGIN_ERROR, + DEFAULT_TOKEN_LIFETIME_SECONDS, + DEVELOPMENT_COLLECTOR_LOOPBACK_ERROR, + HTTP_TIMEOUT_SECONDS, + HTTP_USER_AGENT, + INITIAL_RETRY_SECONDS, + LOCAL_ONLY_MISSING_CREDENTIAL, + LOCAL_ONLY_REJECTED_COLLECTOR_AUTHORIZATION, + LOCAL_ONLY_REJECTED_CREDENTIAL, + LOCAL_ONLY_REJECTED_REGISTRATION, + LOOPBACK_HOSTS, + MAX_HTTP_RESPONSE_BYTES, + MAX_RETRY_SECONDS, + MINIMUM_TOKEN_LIFETIME_SECONDS, + NO_CREDENTIAL_MESSAGE, + PRODUCTION_COLLECTOR_URL, + REJECTED_CREDENTIAL_MESSAGE, + RUN_EVENTS_PATH_TEMPLATE, + RUN_REGISTRATION_PATH, + TOKEN_EXCHANGE_PATH, + TOKEN_REFRESH_MARGIN_SECONDS, +) +from microcosm.build.telemetry_emitter_service.spool import EventSpool + + +class _NoRedirectHandler(urllib.request.HTTPRedirectHandler): + """Keep bearer credentials on the explicitly configured origin.""" + + def redirect_request(self, req, fp, code, msg, headers, newurl): + return None + + +def _collector_origin(value: str, *, allow_loopback_http: bool = False) -> str: + parsed = urlsplit(value) + loopback = parsed.hostname in LOOPBACK_HOSTS + valid_scheme = parsed.scheme == "https" or ( + allow_loopback_http and loopback and parsed.scheme == "http" + ) + if not parsed.hostname or not valid_scheme: + raise ValueError(COLLECTOR_URL_HTTPS_ERROR) + if ( + parsed.username + or parsed.password + or parsed.query + or parsed.fragment + or parsed.path not in {"", "/"} + ): + raise ValueError(COLLECTOR_URL_ORIGIN_ERROR) + return value.rstrip("/") + + +def _development_collector_url(value: str) -> str: + parsed = urlsplit(value) + if parsed.hostname not in LOOPBACK_HOSTS: + raise ValueError(DEVELOPMENT_COLLECTOR_LOOPBACK_ERROR) + return _collector_origin(value, allow_loopback_http=True) + + +def _decode_response(body: bytes) -> dict[str, Any]: + try: + response = json.loads(body) if body else {} + except (json.JSONDecodeError, UnicodeDecodeError): + return {} + return response if isinstance(response, dict) else {} + + +def _http_post( + url: str, + payload: Mapping[str, Any], + bearer_token: str, + *, + timeout: float = HTTP_TIMEOUT_SECONDS, +) -> tuple[int, dict[str, Any]]: + request = urllib.request.Request( + url, + data=json.dumps(payload, separators=(",", ":")).encode(), + headers={ + "Authorization": f"Bearer {bearer_token}", + "Content-Type": "application/json", + "User-Agent": HTTP_USER_AGENT, + }, + method="POST", + ) + try: + opener = urllib.request.build_opener(_NoRedirectHandler) + with opener.open(request, timeout=timeout) as response: + body = response.read(MAX_HTTP_RESPONSE_BYTES + 1) + if len(body) > MAX_HTTP_RESPONSE_BYTES: + raise OSError(COLLECTOR_RESPONSE_TOO_LARGE_ERROR) + return response.status, _decode_response(body) + except urllib.error.HTTPError as error: + body = error.read(MAX_HTTP_RESPONSE_BYTES + 1) + if len(body) > MAX_HTTP_RESPONSE_BYTES: + return error.code, {} + return error.code, _decode_response(body) + + +def _huggingface_token() -> str | None: + return ( + os.environ.get("HF_TOKEN", "").strip() + or os.environ.get("HUGGINGFACE_TOKEN", "").strip() + or get_token() + ) + + +def _registration_key(registration: Mapping[str, Any]) -> str: + return f"{registration['run_id']}:{registration['producer_id']}" + + +class CollectorDelivery: + """Authenticate queued runs and deliver idempotent event batches.""" + + def __init__( + self, + spool: EventSpool, + *, + development_collector_url: str | None = None, + ) -> None: + self.collector_url = ( + _development_collector_url(development_collector_url) + if development_collector_url is not None + else _collector_origin(PRODUCTION_COLLECTOR_URL) + ) + self.spool = spool + self._session_token: tuple[str, float] | None = None + self._registered: set[str] = set() + self._warned_no_token = False + self._warned_denied: set[str] = set() + self._next_attempt_at = 0.0 + self._retry_seconds = INITIAL_RETRY_SECONDS + + def flush_once(self) -> bool: + """Attempt one delivery pass without waiting for retry deadlines.""" + + if time.monotonic() < self._next_attempt_at: + return False + made_progress = False + for registration in self.spool.pending_runs(): + run_id = str(registration["run_id"]) + registration_key = _registration_key(registration) + token = self._collector_token(registration) + if token is None: + continue + if registration_key not in self._registered: + if not self._register(registration, token): + continue + events = self.spool.batch(run_id, str(registration["producer_id"])) + if not events: + continue + try: + status, _ = _http_post( + self.collector_url + RUN_EVENTS_PATH_TEMPLATE.format(run_id=run_id), + {"events": events}, + token, + ) + except (OSError, TimeoutError): + self._defer_retry() + continue + if status == HTTPStatus.ACCEPTED: + self.spool.acknowledge([event["event_id"] for event in events]) + made_progress = True + self._retry_seconds = INITIAL_RETRY_SECONDS + elif status == HTTPStatus.UNAUTHORIZED: + self._session_token = None + elif status == HTTPStatus.FORBIDDEN: + self._make_local_only( + registration, + LOCAL_ONLY_REJECTED_COLLECTOR_AUTHORIZATION, + ) + else: + self._defer_retry() + return made_progress + + def _collector_token(self, registration: Mapping[str, Any]) -> str | None: + if ( + self._session_token is not None + and self._session_token[1] > time.monotonic() + TOKEN_REFRESH_MARGIN_SECONDS + ): + return self._session_token[0] + hf_token = _huggingface_token() + if not hf_token: + if not self._warned_no_token: + print(NO_CREDENTIAL_MESSAGE, file=sys.stderr, flush=True) + self._warned_no_token = True + self._make_local_only(registration, LOCAL_ONLY_MISSING_CREDENTIAL) + return None + try: + status, response = _http_post( + self.collector_url + TOKEN_EXCHANGE_PATH, + {}, + hf_token, + ) + except (OSError, TimeoutError): + self._defer_retry() + return None + if status == HTTPStatus.OK and isinstance(response.get("access_token"), str): + expires_in = max( + MINIMUM_TOKEN_LIFETIME_SECONDS, + int(response.get("expires_in", DEFAULT_TOKEN_LIFETIME_SECONDS)), + ) + token = response["access_token"] + self._session_token = (token, time.monotonic() + expires_in) + return token + if status in {HTTPStatus.UNAUTHORIZED, HTTPStatus.FORBIDDEN}: + self._make_local_only(registration, LOCAL_ONLY_REJECTED_CREDENTIAL) + else: + self._defer_retry() + return None + + def _register(self, registration: Mapping[str, Any], token: str) -> bool: + registration_key = _registration_key(registration) + try: + status, _ = _http_post( + self.collector_url + RUN_REGISTRATION_PATH, + registration, + token, + ) + except (OSError, TimeoutError): + self._defer_retry() + return False + if status == HTTPStatus.CREATED: + self._registered.add(registration_key) + self._retry_seconds = INITIAL_RETRY_SECONDS + return True + if status == HTTPStatus.UNAUTHORIZED: + self._session_token = None + elif status in {HTTPStatus.FORBIDDEN, HTTPStatus.CONFLICT}: + self._make_local_only(registration, LOCAL_ONLY_REJECTED_REGISTRATION) + else: + self._defer_retry() + return False + + def _make_local_only( + self, + registration: Mapping[str, Any], + reason: str, + ) -> None: + run_id = str(registration["run_id"]) + producer_id = str(registration["producer_id"]) + registration_key = _registration_key(registration) + self.spool.make_local_only(run_id, producer_id, reason) + if reason != LOCAL_ONLY_MISSING_CREDENTIAL: + self._warn_denied(registration_key) + + def _defer_retry(self) -> None: + self._next_attempt_at = time.monotonic() + self._retry_seconds + self._retry_seconds = min(MAX_RETRY_SECONDS, self._retry_seconds * 2) + + def _warn_denied(self, registration_key: str) -> None: + if registration_key in self._warned_denied: + return + print(REJECTED_CREDENTIAL_MESSAGE, file=sys.stderr, flush=True) + self._warned_denied.add(registration_key) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/constants.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/constants.py new file mode 100644 index 000000000..e4a55b7d7 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/constants.py @@ -0,0 +1,69 @@ +"""Configuration and protocol values for the telemetry emitter service.""" + +from __future__ import annotations + +from typing import Final + +RETENTION_DAYS: Final = 7 +MAX_QUEUED_BYTES: Final = 100 * 1024 * 1024 +BATCH_SIZE: Final = 100 +PRODUCTION_COLLECTOR_URL: Final = ( + "https://microcosm-telemetry-389282473430.us-central1.run.app" +) + +LOOPBACK_HOSTS: Final = frozenset({"localhost", "127.0.0.1", "::1"}) +TOKEN_EXCHANGE_PATH: Final = "/v1/auth/huggingface/exchange" +RUN_REGISTRATION_PATH: Final = "/v1/runs" +RUN_EVENTS_PATH_TEMPLATE: Final = "/v1/runs/{run_id}/events" +HTTP_USER_AGENT: Final = "microcosm-telemetry-emitter/1" +MAX_HTTP_RESPONSE_BYTES: Final = 1_048_576 +HTTP_TIMEOUT_SECONDS: Final = 5.0 + +UPLOAD_STATE_PENDING: Final = "pending" +UPLOAD_STATE_LOCAL_ONLY: Final = "local_only" +LOCAL_ONLY_MISSING_CREDENTIAL: Final = "missing_huggingface_credential" +LOCAL_ONLY_REJECTED_CREDENTIAL: Final = "huggingface_credential_rejected" +LOCAL_ONLY_REJECTED_REGISTRATION: Final = "run_registration_rejected" +LOCAL_ONLY_REJECTED_COLLECTOR_AUTHORIZATION: Final = "collector_authorization_rejected" +LOCAL_ONLY_PRE_ELIGIBILITY: Final = "created_before_upload_eligibility" + +NO_CREDENTIAL_MESSAGE: Final = ( + "Microcosm telemetry is local-only: no ambient Hugging Face credential was " + "found. The dataset build will continue." +) +REJECTED_CREDENTIAL_MESSAGE: Final = ( + "Microcosm telemetry is local-only for this run: the ambient Hugging Face " + "credential was not accepted as a PolicyEngine organization member. The " + "dataset build will continue." +) +COLLECTOR_URL_HTTPS_ERROR: Final = "collector URL must be an HTTPS origin" +COLLECTOR_URL_ORIGIN_ERROR: Final = ( + "collector URL must be an origin without credentials or path data" +) +DEVELOPMENT_COLLECTOR_LOOPBACK_ERROR: Final = ( + "development collector URL must use a loopback address" +) +COLLECTOR_RESPONSE_TOO_LARGE_ERROR: Final = "collector response exceeds 1 MiB" + +DEFAULT_TOKEN_LIFETIME_SECONDS: Final = 3_600 +MINIMUM_TOKEN_LIFETIME_SECONDS: Final = 60 +TOKEN_REFRESH_MARGIN_SECONDS: Final = 30 +INITIAL_RETRY_SECONDS: Final = 1.0 +MAX_RETRY_SECONDS: Final = 60.0 + +DATABASE_TIMEOUT_SECONDS: Final = 5 +PRUNE_INTERVAL_SECONDS: Final = 60.0 + +DEFAULT_HEARTBEAT_SECONDS: Final = 60.0 +DEFAULT_DRAIN_SECONDS: Final = 15.0 +MINIMUM_HEARTBEAT_SECONDS: Final = 1.0 +SOCKET_LISTEN_BACKLOG: Final = 16 +SOCKET_ACCEPT_TIMEOUT_SECONDS: Final = 0.5 +SOCKET_CONNECTION_TIMEOUT_SECONDS: Final = 0.25 +WORKER_INTERVAL_SECONDS: Final = 1.0 +DRAIN_RETRY_SECONDS: Final = 0.5 + +EVENT_OBJECT_ERROR: Final = "event must be an object" +UNSUPPORTED_ACTION_ERROR: Final = "unsupported local telemetry action" +LOCAL_MESSAGE_TOO_LARGE_ERROR: Final = "local telemetry message exceeds 1 MiB" +FAILURE_CLASS_UNEXPECTED_PROCESS_EXIT: Final = "unexpected_process_exit" diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/main.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/main.py new file mode 100644 index 000000000..2872bf91a --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/main.py @@ -0,0 +1,53 @@ +"""Command-line construction for the telemetry emitter service.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +from microcosm.build.telemetry_emitter_service.collector import CollectorDelivery +from microcosm.build.telemetry_emitter_service.constants import ( + DEFAULT_HEARTBEAT_SECONDS, +) +from microcosm.build.telemetry_emitter_service.resources import ProcessTreeSampler +from microcosm.build.telemetry_emitter_service.runtime import EmitterService +from microcosm.build.telemetry_emitter_service.spool import EventSpool + + +def build_parser() -> argparse.ArgumentParser: + """Create the service command-line parser.""" + + parser = argparse.ArgumentParser() + parser.add_argument("--socket", type=Path, required=True) + parser.add_argument("--spool", type=Path, required=True) + parser.add_argument("--development-collector-url") + parser.add_argument("--registration-json", required=True) + parser.add_argument("--parent-pid", type=int, required=True) + parser.add_argument( + "--heartbeat-seconds", + type=float, + default=DEFAULT_HEARTBEAT_SECONDS, + ) + return parser + + +def main(argv: list[str] | None = None) -> int: + """Run the telemetry emitter service until the build disconnects.""" + + args = build_parser().parse_args(argv) + registration = json.loads(args.registration_json) + spool = EventSpool(args.spool) + service = EmitterService( + socket_path=args.socket, + registration=registration, + spool=spool, + delivery=CollectorDelivery( + spool, + development_collector_url=args.development_collector_url, + ), + sampler=ProcessTreeSampler(args.parent_pid), + heartbeat_seconds=args.heartbeat_seconds, + ) + service.run() + return 0 diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/resources.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/resources.py new file mode 100644 index 000000000..d23d6f46c --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/resources.py @@ -0,0 +1,137 @@ +"""Process-tree resource sampling for telemetry events.""" + +from __future__ import annotations + +import os +from pathlib import Path +from typing import Any + +try: + import psutil +except ModuleNotFoundError: # Base installs use the standard-library fallback. + psutil = None + + +def _empty_sample(peak_rss: int) -> dict[str, Any]: + return { + "cpu_user_seconds": 0.0, + "cpu_system_seconds": 0.0, + "rss_bytes": 0, + "peak_rss_bytes": peak_rss, + } + + +class ProcessTreeSampler: + """Collect cumulative CPU and resident memory for a build process tree.""" + + def __init__(self, parent_pid: int) -> None: + self.parent_pid = parent_pid + self._peak_rss = 0 + self._parent_create_time = self._create_time() + + def _create_time(self) -> float | str | None: + if psutil is not None: + try: + return psutil.Process(self.parent_pid).create_time() + except (psutil.Error, OSError): + return None + stat_path = Path(f"/proc/{self.parent_pid}/stat") + try: + return stat_path.read_text().rsplit(")", 1)[1].split()[19] + except (OSError, IndexError): + return None + + def parent_alive(self) -> bool: + """Return whether the sampled process still has its original identity.""" + + if psutil is not None: + try: + process = psutil.Process(self.parent_pid) + return ( + self._parent_create_time is not None + and process.create_time() == self._parent_create_time + and process.is_running() + and process.status() != psutil.STATUS_ZOMBIE + ) + except (psutil.Error, OSError): + return False + try: + os.kill(self.parent_pid, 0) + except OSError: + return False + current = self._create_time() + # /proc supplies a process-start tick that protects against PID reuse. + # On platforms without it, existence is the best standard-library test. + return self._parent_create_time is None or current == self._parent_create_time + + def sample(self) -> dict[str, Any]: + """Sample the parent and every currently visible child process.""" + + if psutil is None: + return self._fallback_sample() + processes = [] + try: + parent = psutil.Process(self.parent_pid) + processes = [parent, *parent.children(recursive=True)] + except (psutil.Error, OSError): + pass + user = 0.0 + system = 0.0 + rss = 0 + for process in processes: + try: + cpu = process.cpu_times() + user += float(cpu.user) + float(getattr(cpu, "children_user", 0.0)) + system += float(cpu.system) + float( + getattr(cpu, "children_system", 0.0) + ) + rss += int(process.memory_info().rss) + except (psutil.Error, OSError): + continue + self._peak_rss = max(self._peak_rss, rss) + return { + "cpu_user_seconds": user, + "cpu_system_seconds": system, + "rss_bytes": rss, + "peak_rss_bytes": self._peak_rss, + } + + def _fallback_sample(self) -> dict[str, Any]: + """Sample Linux process trees when the country engine is not installed.""" + + process_rows: dict[int, tuple[int, float, float, int]] = {} + try: + clock_ticks = float(os.sysconf("SC_CLK_TCK")) + page_size = int(os.sysconf("SC_PAGE_SIZE")) + except (AttributeError, OSError, ValueError): + return _empty_sample(self._peak_rss) + for stat_path in Path("/proc").glob("[0-9]*/stat"): + try: + pid = int(stat_path.parent.name) + fields = stat_path.read_text().rsplit(")", 1)[1].split() + process_rows[pid] = ( + int(fields[1]), + int(fields[11]) / clock_ticks, + int(fields[12]) / clock_ticks, + int(fields[21]) * page_size, + ) + except (OSError, ValueError, IndexError): + continue + selected = {self.parent_pid} + changed = True + while changed: + changed = False + for pid, (parent, *_rest) in process_rows.items(): + if parent in selected and pid not in selected: + selected.add(pid) + changed = True + user = sum(process_rows[pid][1] for pid in selected if pid in process_rows) + system = sum(process_rows[pid][2] for pid in selected if pid in process_rows) + rss = sum(process_rows[pid][3] for pid in selected if pid in process_rows) + self._peak_rss = max(self._peak_rss, rss) + return { + "cpu_user_seconds": user, + "cpu_system_seconds": system, + "rss_bytes": rss, + "peak_rss_bytes": self._peak_rss, + } diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/runtime.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/runtime.py new file mode 100644 index 000000000..8f9e1c847 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/runtime.py @@ -0,0 +1,215 @@ +"""Unix-socket runtime for the telemetry emitter service.""" + +from __future__ import annotations + +import json +import os +import socket +import threading +import time +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +from microcosm.build.telemetry_emitter_service.collector import CollectorDelivery +from microcosm.build.telemetry_emitter_service.constants import ( + DEFAULT_DRAIN_SECONDS, + DRAIN_RETRY_SECONDS, + EVENT_OBJECT_ERROR, + FAILURE_CLASS_UNEXPECTED_PROCESS_EXIT, + LOCAL_MESSAGE_TOO_LARGE_ERROR, + MINIMUM_HEARTBEAT_SECONDS, + SOCKET_ACCEPT_TIMEOUT_SECONDS, + SOCKET_CONNECTION_TIMEOUT_SECONDS, + SOCKET_LISTEN_BACKLOG, + UNSUPPORTED_ACTION_ERROR, + WORKER_INTERVAL_SECONDS, +) +from microcosm.build.telemetry_emitter_service.resources import ProcessTreeSampler +from microcosm.build.telemetry_emitter_service.spool import EventSpool +from microcosm.build.telemetry_emitter_service.timestamps import utc_now +from microcosm.build.telemetry_protocol import ( + ACTION_CLOSE, + ACTION_EVENT, + ACTION_PING, + EVENT_TYPE_HEARTBEAT, + EVENT_TYPE_RUN, + LOCAL_ACKNOWLEDGEMENT_ERROR, + LOCAL_ACKNOWLEDGEMENT_OK, + LOCAL_MESSAGE_DELIMITER, + LOCAL_SOCKET_READ_BYTES, + MAX_LOCAL_MESSAGE_BYTES, + STAGE_COMPLETE, + STAGE_CREATED, + STAGE_FAILED, + STATUS_FAILED, + STATUS_PROGRESS, + UNEXPECTED_PROCESS_EXIT_MESSAGE, +) + + +def _heartbeat_event(stage_id: str) -> dict[str, Any]: + return { + "timestamp": utc_now(), + "event_type": EVENT_TYPE_HEARTBEAT, + "stage_id": stage_id, + "status": STATUS_PROGRESS, + "message": None, + "details": {}, + } + + +def _unexpected_exit_event(stage_id: str) -> dict[str, Any]: + return { + "timestamp": utc_now(), + "event_type": EVENT_TYPE_RUN, + "stage_id": STAGE_FAILED, + "status": STATUS_FAILED, + "message": UNEXPECTED_PROCESS_EXIT_MESSAGE, + "details": { + "failure_class": FAILURE_CLASS_UNEXPECTED_PROCESS_EXIT, + "failed_during": stage_id, + }, + } + + +class EmitterService: + """Private local socket server with a concurrent delivery worker.""" + + def __init__( + self, + *, + socket_path: Path, + registration: Mapping[str, Any], + spool: EventSpool, + delivery: CollectorDelivery, + sampler: ProcessTreeSampler, + heartbeat_seconds: float, + drain_seconds: float = DEFAULT_DRAIN_SECONDS, + ) -> None: + self.socket_path = socket_path + self.registration = dict(registration) + self.spool = spool + self.delivery = delivery + self.sampler = sampler + self.heartbeat_seconds = max( + MINIMUM_HEARTBEAT_SECONDS, + heartbeat_seconds, + ) + self.drain_seconds = max(0.0, drain_seconds) + self._stop = threading.Event() + self._last_stage = STAGE_CREATED + + def run(self) -> None: + """Serve local messages until the client closes or exits.""" + + self.spool.register(self.registration) + self.socket_path.parent.mkdir(parents=True, exist_ok=True) + try: + self.socket_path.unlink(missing_ok=True) + with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as server: + server.bind(str(self.socket_path)) + os.chmod(self.socket_path, 0o600) + server.listen(SOCKET_LISTEN_BACKLOG) + server.settimeout(SOCKET_ACCEPT_TIMEOUT_SECONDS) + worker = threading.Thread(target=self._worker, daemon=True) + worker.start() + self._serve(server) + worker.join(timeout=self.drain_seconds + WORKER_INTERVAL_SECONDS) + finally: + self.socket_path.unlink(missing_ok=True) + try: + self.socket_path.parent.rmdir() + except OSError: + pass + + def _serve(self, server: socket.socket) -> None: + while not self._stop.is_set(): + try: + connection, _ = server.accept() + except TimeoutError: + continue + with connection: + connection.settimeout(SOCKET_CONNECTION_TIMEOUT_SECONDS) + response = self._serve_connection(connection) + try: + connection.sendall(response) + except OSError: + pass + + def _serve_connection(self, connection: socket.socket) -> bytes: + try: + message = self._read_message(connection) + self._handle(message) + except Exception: + return LOCAL_ACKNOWLEDGEMENT_ERROR + return LOCAL_ACKNOWLEDGEMENT_OK + + @staticmethod + def _read_message(connection: socket.socket) -> dict[str, Any]: + chunks = bytearray() + while len(chunks) <= MAX_LOCAL_MESSAGE_BYTES: + data = connection.recv(LOCAL_SOCKET_READ_BYTES) + if not data: + break + chunks.extend(data) + if LOCAL_MESSAGE_DELIMITER in data: + break + if len(chunks) > MAX_LOCAL_MESSAGE_BYTES: + raise ValueError(LOCAL_MESSAGE_TOO_LARGE_ERROR) + return json.loads(bytes(chunks).split(LOCAL_MESSAGE_DELIMITER, 1)[0]) + + def _handle(self, message: Mapping[str, Any]) -> None: + action = message.get("action") + if action == ACTION_EVENT: + event = message.get("event") + if not isinstance(event, Mapping): + raise ValueError(EVENT_OBJECT_ERROR) + stage_id = event.get("stage_id") + if isinstance(stage_id, str) and stage_id not in { + STAGE_COMPLETE, + STAGE_FAILED, + }: + self._last_stage = stage_id + self.spool.append( + self.registration, + event, + resources=self.sampler.sample(), + ) + elif action == ACTION_CLOSE: + self._stop.set() + elif action == ACTION_PING: + return + else: + raise ValueError(UNSUPPORTED_ACTION_ERROR) + + def _worker(self) -> None: + next_heartbeat = time.monotonic() + self.heartbeat_seconds + while not self._stop.wait(WORKER_INTERVAL_SECONDS): + now = time.monotonic() + # Sample every worker iteration so short-lived build children are + # much less likely to disappear between stage and heartbeat events. + self.sampler.sample() + if now >= next_heartbeat: + self.spool.append( + self.registration, + _heartbeat_event(self._last_stage), + resources=self.sampler.sample(), + ) + next_heartbeat = now + self.heartbeat_seconds + self.delivery.flush_once() + if not self.sampler.parent_alive(): + self.spool.append( + self.registration, + _unexpected_exit_event(self._last_stage), + resources=self.sampler.sample(), + ) + self._stop.set() + break + self._drain() + + def _drain(self) -> None: + deadline = time.monotonic() + self.drain_seconds + while self.spool.has_deliverable() and time.monotonic() < deadline: + if not self.delivery.flush_once(): + self._stop.wait(DRAIN_RETRY_SECONDS) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py new file mode 100644 index 000000000..7474454a7 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py @@ -0,0 +1,342 @@ +"""Durable local queue for telemetry events.""" + +from __future__ import annotations + +import json +import sqlite3 +import threading +import time +import uuid +from collections.abc import Mapping +from datetime import UTC, datetime, timedelta +from pathlib import Path +from typing import Any, Final + +from microcosm.build.telemetry_emitter_service.constants import ( + BATCH_SIZE, + DATABASE_TIMEOUT_SECONDS, + LOCAL_ONLY_PRE_ELIGIBILITY, + MAX_QUEUED_BYTES, + PRUNE_INTERVAL_SECONDS, + RETENTION_DAYS, + UPLOAD_STATE_LOCAL_ONLY, + UPLOAD_STATE_PENDING, +) +from microcosm.build.telemetry_emitter_service.timestamps import utc_now +from microcosm.build.telemetry_protocol import TELEMETRY_SCHEMA_VERSION + +_DATABASE_SCHEMA: Final = f""" +CREATE TABLE IF NOT EXISTS telemetry_runs ( + run_id TEXT NOT NULL, + producer_id TEXT NOT NULL, + registration_json TEXT NOT NULL, + next_sequence INTEGER NOT NULL DEFAULT 1, + upload_state TEXT NOT NULL DEFAULT '{UPLOAD_STATE_PENDING}', + local_only_reason TEXT, + updated_at TEXT NOT NULL, + PRIMARY KEY(run_id, producer_id) +); +CREATE TABLE IF NOT EXISTS telemetry_events ( + event_id TEXT PRIMARY KEY, + run_id TEXT NOT NULL, + producer_id TEXT NOT NULL, + sequence INTEGER NOT NULL, + payload_json TEXT NOT NULL, + created_at TEXT NOT NULL, + UNIQUE(run_id, producer_id, sequence), + FOREIGN KEY(run_id, producer_id) + REFERENCES telemetry_runs(run_id, producer_id) +); +CREATE INDEX IF NOT EXISTS telemetry_events_run_sequence + ON telemetry_events(run_id, producer_id, sequence); +""" + + +class EventSpool: + """Small SQLite queue shared by successive emitter service processes.""" + + def __init__(self, path: Path | str) -> None: + self.path = Path(path) + self.path.parent.mkdir(parents=True, exist_ok=True) + try: + self.path.parent.chmod(0o700) + except OSError: + pass + self._connection = sqlite3.connect( + self.path, + timeout=DATABASE_TIMEOUT_SECONDS, + check_same_thread=False, + ) + self._connection.row_factory = sqlite3.Row + self._lock = threading.RLock() + self._last_prune_at = 0.0 + self._initialize_database() + self.prune() + + def _initialize_database(self) -> None: + with self._lock, self._connection: + self._connection.execute("PRAGMA journal_mode=WAL") + self._connection.execute("PRAGMA synchronous=NORMAL") + self._connection.executescript(_DATABASE_SCHEMA) + columns = { + row["name"] + for row in self._connection.execute( + "PRAGMA table_info(telemetry_runs)" + ).fetchall() + } + if "upload_state" not in columns: + self._connection.execute( + "ALTER TABLE telemetry_runs ADD COLUMN " + f"upload_state TEXT NOT NULL DEFAULT '{UPLOAD_STATE_PENDING}'" + ) + self._connection.execute( + "UPDATE telemetry_runs SET upload_state = ?", + (UPLOAD_STATE_LOCAL_ONLY,), + ) + if "local_only_reason" not in columns: + self._connection.execute( + "ALTER TABLE telemetry_runs ADD COLUMN local_only_reason TEXT" + ) + self._connection.execute( + "UPDATE telemetry_runs SET local_only_reason = ? " + "WHERE upload_state = ?", + (LOCAL_ONLY_PRE_ELIGIBILITY, UPLOAD_STATE_LOCAL_ONLY), + ) + + def register(self, registration: Mapping[str, Any]) -> None: + """Create or refresh a producer registration.""" + + run_id = str(registration["run_id"]) + producer_id = str(registration["producer_id"]) + encoded = json.dumps(registration, separators=(",", ":"), sort_keys=True) + with self._lock, self._connection: + existing = self._connection.execute( + """ + SELECT registration_json FROM telemetry_runs + WHERE run_id = ? AND producer_id = ? + """, + (run_id, producer_id), + ).fetchone() + if existing is not None: + old = json.loads(existing["registration_json"]) + if old.get("producer_id") != registration.get("producer_id"): + raise ValueError(f"run_id {run_id!r} already has another producer") + self._connection.execute( + """ + INSERT INTO telemetry_runs ( + run_id, producer_id, registration_json, next_sequence, updated_at + ) VALUES (?, ?, ?, 1, ?) + ON CONFLICT(run_id, producer_id) DO UPDATE SET + registration_json = excluded.registration_json, + updated_at = excluded.updated_at + """, + (run_id, producer_id, encoded, utc_now()), + ) + + def append( + self, + registration: Mapping[str, Any], + event: Mapping[str, Any], + *, + resources: Mapping[str, Any] | None = None, + ) -> dict[str, Any]: + """Append an event and assign its stable producer sequence.""" + + run_id = str(registration["run_id"]) + producer_id = str(registration["producer_id"]) + with self._lock, self._connection: + row = self._connection.execute( + """ + SELECT next_sequence FROM telemetry_runs + WHERE run_id = ? AND producer_id = ? + """, + (run_id, producer_id), + ).fetchone() + if row is None: + raise KeyError(run_id) + sequence = int(row["next_sequence"]) + event_id = uuid.uuid4().hex + payload = { + "schema_version": TELEMETRY_SCHEMA_VERSION, + "event_id": event_id, + "run_id": run_id, + "producer_id": producer_id, + "sequence": sequence, + "timestamp": event.get("timestamp") or utc_now(), + "event_type": event["event_type"], + "stage_id": event.get("stage_id"), + "status": event["status"], + "message": event.get("message"), + "details": event.get("details") or {}, + "resources": resources, + } + encoded = json.dumps(payload, separators=(",", ":"), allow_nan=False) + self._connection.execute( + """ + INSERT INTO telemetry_events ( + event_id, run_id, producer_id, sequence, + payload_json, created_at + ) VALUES (?, ?, ?, ?, ?, ?) + """, + (event_id, run_id, producer_id, sequence, encoded, utc_now()), + ) + self._connection.execute( + """ + UPDATE telemetry_runs + SET next_sequence = ?, updated_at = ? + WHERE run_id = ? AND producer_id = ? + """, + (sequence + 1, utc_now(), run_id, producer_id), + ) + if time.monotonic() - self._last_prune_at >= PRUNE_INTERVAL_SECONDS: + self.prune() + return payload + + def pending_runs(self) -> list[dict[str, Any]]: + """Return registrations that have events eligible for delivery.""" + + with self._lock: + rows = self._connection.execute( + """ + SELECT DISTINCT r.registration_json + FROM telemetry_runs r + JOIN telemetry_events e + ON e.run_id = r.run_id AND e.producer_id = r.producer_id + WHERE r.upload_state = ? + ORDER BY r.updated_at + """, + (UPLOAD_STATE_PENDING,), + ).fetchall() + return [json.loads(row["registration_json"]) for row in rows] + + def make_local_only( + self, + run_id: str, + producer_id: str, + reason: str, + ) -> None: + """Permanently exclude one producer's queued events from upload.""" + + with self._lock, self._connection: + self._connection.execute( + """ + UPDATE telemetry_runs + SET upload_state = ?, local_only_reason = ?, updated_at = ? + WHERE run_id = ? AND producer_id = ? + """, + ( + UPLOAD_STATE_LOCAL_ONLY, + reason, + utc_now(), + run_id, + producer_id, + ), + ) + + def has_deliverable(self) -> bool: + """Return whether any queued event remains eligible for delivery.""" + + with self._lock: + row = self._connection.execute( + """ + SELECT 1 + FROM telemetry_events e + JOIN telemetry_runs r + ON r.run_id = e.run_id AND r.producer_id = e.producer_id + WHERE r.upload_state = ? + LIMIT 1 + """, + (UPLOAD_STATE_PENDING,), + ).fetchone() + return row is not None + + def batch( + self, + run_id: str, + producer_id: str, + limit: int = BATCH_SIZE, + ) -> list[dict[str, Any]]: + """Return the next ordered batch for one producer.""" + + with self._lock: + rows = self._connection.execute( + """ + SELECT payload_json FROM telemetry_events + WHERE run_id = ? AND producer_id = ? + ORDER BY sequence LIMIT ? + """, + (run_id, producer_id, limit), + ).fetchall() + return [json.loads(row["payload_json"]) for row in rows] + + def acknowledge(self, event_ids: list[str]) -> None: + """Remove events acknowledged by the collector.""" + + if not event_ids: + return + placeholders = ",".join("?" for _ in event_ids) + with self._lock, self._connection: + self._connection.execute( + f"DELETE FROM telemetry_events WHERE event_id IN ({placeholders})", + event_ids, + ) + + def has_pending(self) -> bool: + """Return whether any event remains in local storage.""" + + with self._lock: + row = self._connection.execute( + "SELECT 1 FROM telemetry_events LIMIT 1" + ).fetchone() + return row is not None + + def prune(self) -> None: + """Enforce the age and total-size retention limits.""" + + cutoff = (datetime.now(UTC) - timedelta(days=RETENTION_DAYS)).isoformat() + with self._lock, self._connection: + self._connection.execute( + "DELETE FROM telemetry_events WHERE created_at < ?", + (cutoff,), + ) + size_row = self._connection.execute( + "SELECT COALESCE(SUM(LENGTH(payload_json)), 0) AS bytes " + "FROM telemetry_events" + ).fetchone() + excess = int(size_row["bytes"]) - MAX_QUEUED_BYTES + if excess > 0: + self._remove_oldest_bytes(excess) + self._connection.execute( + """ + DELETE FROM telemetry_runs + WHERE updated_at < ? AND NOT EXISTS ( + SELECT 1 FROM telemetry_events e + WHERE e.run_id = telemetry_runs.run_id + AND e.producer_id = telemetry_runs.producer_id + ) + """, + (cutoff,), + ) + self._last_prune_at = time.monotonic() + + def _remove_oldest_bytes(self, excess: int) -> None: + rows = self._connection.execute( + """ + SELECT event_id, LENGTH(payload_json) AS bytes + FROM telemetry_events ORDER BY created_at, sequence + """ + ).fetchall() + removed = 0 + event_ids: list[str] = [] + for row in rows: + event_ids.append(row["event_id"]) + removed += int(row["bytes"]) + if removed >= excess: + break + if not event_ids: + return + placeholders = ",".join("?" for _ in event_ids) + self._connection.execute( + f"DELETE FROM telemetry_events WHERE event_id IN ({placeholders})", + event_ids, + ) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/timestamps.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/timestamps.py new file mode 100644 index 000000000..3f1f39a91 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/timestamps.py @@ -0,0 +1,11 @@ +"""Timestamp helpers shared by the telemetry emitter service.""" + +from __future__ import annotations + +from datetime import UTC, datetime + + +def utc_now() -> str: + """Return the current UTC timestamp in ISO 8601 format.""" + + return datetime.now(UTC).isoformat() diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_identity.py b/packages/microcosm-build/src/microcosm/build/telemetry_identity.py new file mode 100644 index 000000000..0fa5e5b80 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_identity.py @@ -0,0 +1,53 @@ +"""Runtime identity attached to the first telemetry event.""" + +from __future__ import annotations + +import os +import subprocess +from importlib import metadata +from platform import platform +from typing import Any, Final + +_IDENTITY_DISTRIBUTIONS: Final = ( + "microcosm-build", + "microcosm-graph", + "policyengine-us", + "policyengine-uk", +) +_GIT_IDENTITY_TIMEOUT_SECONDS: Final = 1.0 + + +def runtime_identity() -> dict[str, Any]: + """Describe source, host capacity, and installed runtime versions.""" + + try: + commit = subprocess.run( + ["git", "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + timeout=_GIT_IDENTITY_TIMEOUT_SECONDS, + ).stdout.strip() + except (OSError, subprocess.SubprocessError): + commit = None + try: + memory_bytes = int(os.sysconf("SC_PHYS_PAGES")) * int( + os.sysconf("SC_PAGE_SIZE") + ) + except (AttributeError, OSError, ValueError): + memory_bytes = None + versions = {} + for distribution in _IDENTITY_DISTRIBUTIONS: + try: + versions[distribution] = metadata.version(distribution) + except metadata.PackageNotFoundError: + continue + return { + "git_commit": commit, + "host": { + "platform": platform(), + "cpu_count": os.cpu_count(), + "memory_bytes": memory_bytes, + }, + "runtime": versions, + } diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_protocol.py b/packages/microcosm-build/src/microcosm/build/telemetry_protocol.py new file mode 100644 index 000000000..1bc3f6839 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_protocol.py @@ -0,0 +1,66 @@ +"""Shared constants for local and hosted Microcosm telemetry messages.""" + +from __future__ import annotations + +from types import MappingProxyType +from typing import Final, Literal + +type TelemetryAction = Literal["event", "close", "ping"] +type TelemetryEventType = Literal[ + "run", "stage", "progress", "calibration", "heartbeat" +] +type TelemetryStatus = Literal["started", "progress", "completed", "failed"] + +TELEMETRY_SCHEMA_VERSION: Final = 1 + +ACTION_EVENT: Final = "event" +ACTION_CLOSE: Final = "close" +ACTION_PING: Final = "ping" + +EVENT_TYPE_RUN: Final = "run" +EVENT_TYPE_STAGE: Final = "stage" +EVENT_TYPE_PROGRESS: Final = "progress" +EVENT_TYPE_CALIBRATION: Final = "calibration" +EVENT_TYPE_HEARTBEAT: Final = "heartbeat" + +STATUS_STARTED: Final = "started" +STATUS_PROGRESS: Final = "progress" +STATUS_COMPLETED: Final = "completed" +STATUS_FAILED: Final = "failed" + +STAGE_CREATED: Final = "created" +STAGE_CALIBRATING: Final = "calibrating" +STAGE_COMPLETE: Final = "complete" +STAGE_FAILED: Final = "failed" + +CALIBRATION_EVENT_KIND: Final = "calibration_epoch" +BUILD_STARTED_MESSAGE: Final = "Microcosm build started." +BUILD_COMPLETED_MESSAGE: Final = "Microcosm build completed." +UNEXPECTED_PROCESS_EXIT_MESSAGE: Final = ( + "The build process exited without reporting completion." +) + +LOCAL_ACKNOWLEDGEMENT_OK: Final = b"ok\n" +LOCAL_ACKNOWLEDGEMENT_ERROR: Final = b"error\n" +LOCAL_MESSAGE_DELIMITER: Final = b"\n" +LOCAL_PING_MESSAGE: Final = b'{"action":"ping"}\n' + +MAX_TELEMETRY_TEXT_CHARS: Final = 2_000 +MAX_TELEMETRY_MESSAGE_CHARS: Final = 500 +MAX_TELEMETRY_DETAILS_DEPTH: Final = 6 +MAX_TELEMETRY_COLLECTION_ITEMS: Final = 200 +MAX_TELEMETRY_DETAILS_BYTES: Final = 8_192 +MAX_LOCAL_MESSAGE_BYTES: Final = 1_048_576 +LOCAL_SOCKET_READ_BYTES: Final = 65_536 +LOCAL_ACKNOWLEDGEMENT_READ_BYTES: Final = 16 + +SEQUENTIAL_STATUS_MAP: Final = MappingProxyType( + { + "failed": STATUS_FAILED, + "completed": STATUS_COMPLETED, + "passed": STATUS_COMPLETED, + "progress": STATUS_PROGRESS, + "started": STATUS_STARTED, + "running": STATUS_STARTED, + } +) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_sanitization.py b/packages/microcosm-build/src/microcosm/build/telemetry_sanitization.py new file mode 100644 index 000000000..cabd12f33 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_sanitization.py @@ -0,0 +1,94 @@ +"""Size bounds and credential redaction for telemetry payloads.""" + +from __future__ import annotations + +import json +import re +from collections.abc import Mapping +from pathlib import Path +from typing import Any, Final + +from microcosm.build.telemetry_protocol import ( + MAX_TELEMETRY_COLLECTION_ITEMS, + MAX_TELEMETRY_DETAILS_BYTES, + MAX_TELEMETRY_DETAILS_DEPTH, + MAX_TELEMETRY_TEXT_CHARS, +) + +_SENSITIVE_KEY_PARTS: Final = ( + "authorization", + "credential", + "password", + "secret", + "token", + "traceback", +) +_SECRET_TEXT: Final = re.compile( + r"(?i)(?:bearer\s+[^\s]+|(?:token|secret|password|credential)\s*[:=]\s*[^\s]+|hf_[A-Za-z0-9_-]{8,})" +) +_REDACTED_VALUE: Final = "[redacted]" +_MAXIMUM_DEPTH_VALUE: Final = "[maximum depth]" +_DETAILS_TRUNCATED_KEY: Final = "telemetry_details_truncated" + + +def sanitize_text(value: str, *, limit: int = MAX_TELEMETRY_TEXT_CHARS) -> str: + """Redact credential-like text and apply a character limit.""" + + return _SECRET_TEXT.sub(_REDACTED_VALUE, value)[:limit] + + +def sanitize_json(value: Any, *, depth: int = 0) -> Any: + """Return bounded JSON-compatible telemetry without model objects.""" + + if depth >= MAX_TELEMETRY_DETAILS_DEPTH: + return _MAXIMUM_DEPTH_VALUE + if isinstance(value, Mapping): + result = {} + for key, item in list(value.items())[:MAX_TELEMETRY_COLLECTION_ITEMS]: + name = str(key) + if any(part in name.lower() for part in _SENSITIVE_KEY_PARTS): + result[name] = _REDACTED_VALUE + else: + result[name] = sanitize_json(item, depth=depth + 1) + return result + if isinstance(value, (list, tuple)): + return [ + sanitize_json(item, depth=depth + 1) + for item in value[:MAX_TELEMETRY_COLLECTION_ITEMS] + ] + if isinstance(value, Path): + return str(value) + if isinstance(value, float): + return value if value == value and abs(value) != float("inf") else None + if isinstance(value, str): + return sanitize_text(value) + if isinstance(value, (int, bool)) or value is None: + return value + item = getattr(value, "item", None) + if callable(item): + return sanitize_json(item()) + return sanitize_text(str(value)) + + +def sanitize_details(value: Mapping[str, Any]) -> dict[str, Any]: + """Return a redacted details object within the wire-size limit.""" + + sanitized = sanitize_json(value) + assert isinstance(sanitized, dict) + if ( + len(json.dumps(sanitized, separators=(",", ":")).encode()) + <= MAX_TELEMETRY_DETAILS_BYTES + ): + return sanitized + compact: dict[str, Any] = {_DETAILS_TRUNCATED_KEY: True} + for key, item in sanitized.items(): + if not (isinstance(item, (str, int, float, bool)) or item is None): + continue + candidate = {**compact, key: item} + if ( + len(json.dumps(candidate, separators=(",", ":")).encode()) + > MAX_TELEMETRY_DETAILS_BYTES + ): + break + compact[key] = item + return compact diff --git a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py index 75ce48358..8d1cb5745 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py @@ -9,19 +9,30 @@ from pathlib import Path from types import SimpleNamespace -import microcosm.build.telemetry_emitter_service as service_module from microcosm.build.telemetry_emitter import ( LocalTelemetryEmitter, TelemetryRun, - _safe_details, - _safe_json, - _safe_text, ) from microcosm.build.telemetry_emitter_service import ( CollectorDelivery, EmitterService, EventSpool, ) +from microcosm.build.telemetry_emitter_service import collector as collector_module +from microcosm.build.telemetry_emitter_service import resources as resources_module +from microcosm.build.telemetry_emitter_service.constants import ( + PRODUCTION_COLLECTOR_URL, +) +from microcosm.build.telemetry_protocol import ( + BUILD_COMPLETED_MESSAGE, + BUILD_STARTED_MESSAGE, + MAX_TELEMETRY_DETAILS_BYTES, +) +from microcosm.build.telemetry_sanitization import ( + sanitize_details, + sanitize_json, + sanitize_text, +) def _registration(run_id: str = "run-a") -> dict[str, object]: @@ -46,12 +57,12 @@ def _event(stage_id: str = "compile_targets") -> dict[str, object]: def test_development_collector_must_be_on_loopback() -> None: - assert service_module._development_collector_url("http://127.0.0.1:8080") == ( + assert collector_module._development_collector_url("http://127.0.0.1:8080") == ( "http://127.0.0.1:8080" ) for value in ("https://collector.example", "http://192.0.2.1:8080"): try: - service_module._development_collector_url(value) + collector_module._development_collector_url(value) except ValueError as error: assert "loopback" in str(error) else: @@ -68,7 +79,7 @@ def test_production_collector_cannot_be_replaced_by_environment( delivery = CollectorDelivery(EventSpool(tmp_path / "events.sqlite3")) - assert delivery.collector_url == service_module.PRODUCTION_COLLECTOR_URL + assert delivery.collector_url == PRODUCTION_COLLECTOR_URL def test_token_bearing_http_post_does_not_follow_redirects() -> None: @@ -89,7 +100,7 @@ def log_message(self, format, *args): server_thread = threading.Thread(target=server.serve_forever) server_thread.start() try: - status, _ = service_module._http_post( + status, _ = collector_module._http_post( f"http://127.0.0.1:{server.server_port}/exchange", {"run_id": "run-a"}, "hf-private-token", @@ -104,8 +115,8 @@ def log_message(self, format, *args): def test_outbound_payload_redacts_credentials_and_tracebacks() -> None: - assert _safe_text("Bearer hf_abcdefghijk") == "[redacted]" - assert _safe_json( + assert sanitize_text("Bearer hf_abcdefghijk") == "[redacted]" + assert sanitize_json( { "HF_TOKEN": "hf_abcdefghijk", "traceback": "private stack", @@ -116,12 +127,12 @@ def test_outbound_payload_redacts_credentials_and_tracebacks() -> None: "traceback": "[redacted]", "message": "[redacted]", } - bounded = _safe_details( + bounded = sanitize_details( {"done": 2, **{f"large_{index}": "x" * 2_000 for index in range(10)}} ) assert bounded["done"] == 2 assert bounded["telemetry_details_truncated"] is True - assert len(json.dumps(bounded).encode()) <= 8_192 + assert len(json.dumps(bounded).encode()) <= MAX_TELEMETRY_DETAILS_BYTES def test_event_spool_assigns_stable_sequences_and_acknowledges(tmp_path) -> None: @@ -248,7 +259,7 @@ def test_collector_delivery_exchanges_hf_token_then_flushes( queued = spool.append(registration, _event()) requests: list[tuple[str, dict[str, object], str]] = [] - monkeypatch.setattr(service_module, "_huggingface_token", lambda: "hf-secret") + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: "hf-secret") def fake_post(url, payload, token, *, timeout=5.0): requests.append((url, payload, token)) @@ -258,7 +269,7 @@ def fake_post(url, payload, token, *, timeout=5.0): return 201, {"registered": True} return 202, {"accepted": 1, "duplicates": 0} - monkeypatch.setattr(service_module, "_http_post", fake_post) + monkeypatch.setattr(collector_module, "_http_post", fake_post) delivery = CollectorDelivery( spool, @@ -283,13 +294,13 @@ def test_non_org_credential_keeps_event_local(tmp_path, monkeypatch, capsys) -> spool.register(registration) spool.append(registration, _event()) requests: list[dict[str, object]] = [] - monkeypatch.setattr(service_module, "_huggingface_token", lambda: "hf-outsider") + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: "hf-outsider") def reject(url, payload, token, *, timeout=5.0): requests.append(payload) return 403, {"detail": "not a member"} - monkeypatch.setattr(service_module, "_http_post", reject) + monkeypatch.setattr(collector_module, "_http_post", reject) delivery = CollectorDelivery( spool, development_collector_url="http://127.0.0.1:8080", @@ -310,9 +321,9 @@ def test_identity_provider_outage_keeps_events_eligible_for_retry( registration = _registration() spool.register(registration) spool.append(registration, _event()) - monkeypatch.setattr(service_module, "_huggingface_token", lambda: "hf-member") + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: "hf-member") monkeypatch.setattr( - service_module, + collector_module, "_http_post", lambda *args, **kwargs: (503, {"detail": "temporarily unavailable"}), ) @@ -329,9 +340,9 @@ def test_missing_credential_never_contacts_collector_and_stays_local_only( registration = _registration() spool.register(registration) spool.append(registration, _event()) - monkeypatch.setattr(service_module, "_huggingface_token", lambda: None) + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: None) monkeypatch.setattr( - service_module, + collector_module, "_http_post", lambda *args, **kwargs: (_ for _ in ()).throw( AssertionError("network request") @@ -343,13 +354,13 @@ def test_missing_credential_never_contacts_collector_and_stays_local_only( assert spool.pending_runs() == [] reopened = EventSpool(spool_path) - monkeypatch.setattr(service_module, "_huggingface_token", lambda: "hf-later") + monkeypatch.setattr(collector_module, "_huggingface_token", lambda: "hf-later") assert reopened.pending_runs() == [] def test_resource_sampler_has_a_base_install_fallback(monkeypatch) -> None: - monkeypatch.setattr(service_module, "psutil", None) - sampler = service_module.ProcessTreeSampler(os.getpid()) + monkeypatch.setattr(resources_module, "psutil", None) + sampler = resources_module.ProcessTreeSampler(os.getpid()) assert sampler.parent_alive() sample = sampler.sample() @@ -397,11 +408,11 @@ def memory_info(self): return SimpleNamespace(rss=100) monkeypatch.setattr( - service_module, + resources_module, "psutil", SimpleNamespace(Process=FakeProcess, Error=OSError, STATUS_ZOMBIE="zombie"), ) - sampler = service_module.ProcessTreeSampler(10) + sampler = resources_module.ProcessTreeSampler(10) before = sampler.sample() state["child_alive"] = False @@ -577,4 +588,6 @@ def log_message(self, format, *args): identity = events[0]["details"]["identity"] assert identity["host"]["cpu_count"] is not None assert "runtime" in identity + assert events[0]["message"] == BUILD_STARTED_MESSAGE + assert events[-1]["message"] == BUILD_COMPLETED_MESSAGE assert events[-1]["status"] == "completed" From 094dc7a1d3845341f26fb030fc0158d11b49cc95 Mon Sep 17 00:00:00 2001 From: Anthony Volk <14987227+anth-volk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:55:28 +0400 Subject: [PATCH 8/8] Migrate telemetry spool to SQLAlchemy --- .../always-on-telemetry-emitter.added.md | 2 +- packages/microcosm-build/pyproject.toml | 2 + .../telemetry_emitter_service/alembic/env.py | 28 ++ .../alembic/script.py.mako | 31 ++ .../versions/20261007_01_initial_spool.py | 140 +++++++ .../telemetry_emitter_service/database.py | 34 ++ .../telemetry_emitter_service/migrations.py | 60 +++ .../build/telemetry_emitter_service/models.py | 130 +++++++ .../build/telemetry_emitter_service/spool.py | 346 +++++++----------- .../shared/test_telemetry_emitter.py | 179 ++++++--- uv.lock | 132 ++++++- 11 files changed, 805 insertions(+), 279 deletions(-) create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/env.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/script.py.mako create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/versions/20261007_01_initial_spool.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/database.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/migrations.py create mode 100644 packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/models.py diff --git a/changelog.d/always-on-telemetry-emitter.added.md b/changelog.d/always-on-telemetry-emitter.added.md index 392d39d3b..3081aaefc 100644 --- a/changelog.d/always-on-telemetry-emitter.added.md +++ b/changelog.d/always-on-telemetry-emitter.added.md @@ -1 +1 @@ -Microcosm builds now start a local telemetry emitter service that reports authenticated live progress, process-tree CPU and memory, heartbeats, US engine-batch progress, and UK graph-node progress without making collector network requests in the build process. PolicyEngine members use their existing Hugging Face login automatically, while builds without an accepted organization credential remain local-only. The staging JSON writers are now named as run-bundle writers and operate independently from the hosted event emitter. +Microcosm builds now start a local telemetry emitter service that reports authenticated live progress, process-tree CPU and memory, heartbeats, US engine-batch progress, and UK graph-node progress without making collector network requests in the build process. PolicyEngine members use their existing Hugging Face login automatically, while builds without an accepted organization credential remain local-only. The staging JSON writers are now named as run-bundle writers and operate independently from the hosted event emitter. SQLAlchemy ORM sessions now manage the local event spool, and Alembic applies and records its schema migrations. diff --git a/packages/microcosm-build/pyproject.toml b/packages/microcosm-build/pyproject.toml index 51d6b154e..68ac4ce97 100644 --- a/packages/microcosm-build/pyproject.toml +++ b/packages/microcosm-build/pyproject.toml @@ -7,6 +7,7 @@ requires-python = ">=3.13" license = { text = "MIT" } authors = [{ name = "PolicyEngine" }] dependencies = [ + "alembic>=1.13.3,<2", "numpy>=1.26", "pandas>=2", "scipy>=1.13", @@ -24,6 +25,7 @@ dependencies = [ "pyyaml>=6", "jsonschema>=4.23,<5", "referencing>=0.35,<1", + "sqlalchemy>=2,<3", ] [project.optional-dependencies] diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/env.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/env.py new file mode 100644 index 000000000..2cb2edcc2 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/env.py @@ -0,0 +1,28 @@ +"""Alembic environment for the local telemetry spool.""" + +from alembic import context +from sqlalchemy.engine import Connection + +from microcosm.build.telemetry_emitter_service.models import SpoolModel + +_MISSING_CONNECTION_ERROR = ( + "telemetry spool migrations require a programmatically supplied connection" +) + + +def run_migrations() -> None: + """Run migrations on the connection supplied by the emitter service.""" + + connection = context.config.attributes.get("connection") + if not isinstance(connection, Connection): + raise RuntimeError(_MISSING_CONNECTION_ERROR) + context.configure( + connection=connection, + target_metadata=SpoolModel.metadata, + render_as_batch=True, + ) + with context.begin_transaction(): + context.run_migrations() + + +run_migrations() diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/script.py.mako b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/script.py.mako new file mode 100644 index 000000000..f059f7642 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/script.py.mako @@ -0,0 +1,31 @@ +"""${message} + +Revision ID: ${up_revision} +Revises: ${down_revision | comma,n} +Create Date: ${create_date} +""" + +from __future__ import annotations + +from collections.abc import Sequence + +import sqlalchemy as sa +from alembic import op +${imports if imports else ""} + +revision: str = ${repr(up_revision)} +down_revision: str | None = ${repr(down_revision)} +branch_labels: str | Sequence[str] | None = ${repr(branch_labels)} +depends_on: str | Sequence[str] | None = ${repr(depends_on)} + + +def upgrade() -> None: + """Apply this schema revision.""" + + ${upgrades if upgrades else "pass"} + + +def downgrade() -> None: + """Reverse this schema revision.""" + + ${downgrades if downgrades else "pass"} diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/versions/20261007_01_initial_spool.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/versions/20261007_01_initial_spool.py new file mode 100644 index 000000000..87290fd65 --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/alembic/versions/20261007_01_initial_spool.py @@ -0,0 +1,140 @@ +"""Create or adopt the telemetry spool schema. + +Revision ID: 20261007_01 +Revises: +""" + +from __future__ import annotations + +from collections.abc import Sequence + +import sqlalchemy as sa +from alembic import op + +revision: str = "20261007_01" +down_revision: str | None = None +branch_labels: str | Sequence[str] | None = None +depends_on: str | Sequence[str] | None = None + +_RUNS_TABLE = "telemetry_runs" +_EVENTS_TABLE = "telemetry_events" +_EVENTS_INDEX = "telemetry_events_run_sequence" +_PENDING_UPLOAD_STATE = "pending" +_LOCAL_ONLY_UPLOAD_STATE = "local_only" +_PRE_ELIGIBILITY_REASON = "created_before_upload_eligibility" + + +def _create_runs_table() -> None: + op.create_table( + _RUNS_TABLE, + sa.Column("run_id", sa.Text(), nullable=False), + sa.Column("producer_id", sa.Text(), nullable=False), + sa.Column("registration_json", sa.Text(), nullable=False), + sa.Column( + "next_sequence", + sa.Integer(), + nullable=False, + server_default="1", + ), + sa.Column( + "upload_state", + sa.Text(), + nullable=False, + server_default=_PENDING_UPLOAD_STATE, + ), + sa.Column("local_only_reason", sa.Text(), nullable=True), + sa.Column("updated_at", sa.Text(), nullable=False), + sa.PrimaryKeyConstraint("run_id", "producer_id"), + ) + + +def _adopt_runs_table(inspector: sa.Inspector) -> None: + column_names = {column["name"] for column in inspector.get_columns(_RUNS_TABLE)} + added_upload_state = "upload_state" not in column_names + if added_upload_state: + op.add_column( + _RUNS_TABLE, + sa.Column( + "upload_state", + sa.Text(), + nullable=False, + server_default=_PENDING_UPLOAD_STATE, + ), + ) + added_local_only_reason = "local_only_reason" not in column_names + if added_local_only_reason: + op.add_column( + _RUNS_TABLE, + sa.Column("local_only_reason", sa.Text(), nullable=True), + ) + + runs = sa.table( + _RUNS_TABLE, + sa.column("upload_state", sa.Text()), + sa.column("local_only_reason", sa.Text()), + ) + if added_upload_state: + op.execute(sa.update(runs).values(upload_state=_LOCAL_ONLY_UPLOAD_STATE)) + if added_upload_state or added_local_only_reason: + op.execute( + sa.update(runs) + .where(runs.c.upload_state == _LOCAL_ONLY_UPLOAD_STATE) + .values(local_only_reason=_PRE_ELIGIBILITY_REASON) + ) + + +def _create_events_table() -> None: + op.create_table( + _EVENTS_TABLE, + sa.Column("event_id", sa.Text(), nullable=False), + sa.Column("run_id", sa.Text(), nullable=False), + sa.Column("producer_id", sa.Text(), nullable=False), + sa.Column("sequence", sa.Integer(), nullable=False), + sa.Column("payload_json", sa.Text(), nullable=False), + sa.Column("created_at", sa.Text(), nullable=False), + sa.ForeignKeyConstraint( + ["run_id", "producer_id"], + [f"{_RUNS_TABLE}.run_id", f"{_RUNS_TABLE}.producer_id"], + name="fk_telemetry_events_run", + ), + sa.PrimaryKeyConstraint("event_id"), + sa.UniqueConstraint( + "run_id", + "producer_id", + "sequence", + name="uq_telemetry_events_run_producer_sequence", + ), + ) + + +def upgrade() -> None: + """Create the current schema or adopt a pre-Alembic spool.""" + + inspector = sa.inspect(op.get_bind()) + table_names = set(inspector.get_table_names()) + if _RUNS_TABLE not in table_names: + _create_runs_table() + else: + _adopt_runs_table(inspector) + + inspector = sa.inspect(op.get_bind()) + table_names = set(inspector.get_table_names()) + if _EVENTS_TABLE not in table_names: + _create_events_table() + + inspector = sa.inspect(op.get_bind()) + index_names = {index["name"] for index in inspector.get_indexes(_EVENTS_TABLE)} + if _EVENTS_INDEX not in index_names: + op.create_index( + _EVENTS_INDEX, + _EVENTS_TABLE, + ["run_id", "producer_id", "sequence"], + ) + + +def downgrade() -> None: + """Remove the local spool schema.""" + + op.drop_index(_EVENTS_INDEX, table_name=_EVENTS_TABLE) + op.drop_table(_EVENTS_TABLE) + op.drop_table(_RUNS_TABLE) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/database.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/database.py new file mode 100644 index 000000000..75d32d96b --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/database.py @@ -0,0 +1,34 @@ +"""SQLAlchemy engine and session construction for the telemetry spool.""" + +from __future__ import annotations + +from pathlib import Path + +from sqlalchemy import URL, create_engine +from sqlalchemy.engine import Engine +from sqlalchemy.orm import Session, sessionmaker + +from microcosm.build.telemetry_emitter_service.constants import ( + DATABASE_TIMEOUT_SECONDS, +) + + +def sqlite_database_url(path: Path | str) -> URL: + """Return a SQLAlchemy URL for a local SQLite spool path.""" + + return URL.create("sqlite+pysqlite", database=str(Path(path))) + + +def create_spool_engine(path: Path | str) -> Engine: + """Create the SQLAlchemy engine used by one emitter service process.""" + + return create_engine( + sqlite_database_url(path), + connect_args={"timeout": DATABASE_TIMEOUT_SECONDS}, + ) + + +def create_spool_session_factory(engine: Engine) -> sessionmaker[Session]: + """Create short-lived ORM sessions bound to the spool engine.""" + + return sessionmaker(engine, expire_on_commit=False) diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/migrations.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/migrations.py new file mode 100644 index 000000000..c1c69c61c --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/migrations.py @@ -0,0 +1,60 @@ +"""Programmatic Alembic migration entry points for the telemetry spool.""" + +from __future__ import annotations + +from collections.abc import Iterator +from contextlib import contextmanager +from importlib.resources import as_file, files +from pathlib import Path + +from alembic import command +from alembic.config import Config +from alembic.migration import MigrationContext +from alembic.script import ScriptDirectory +from sqlalchemy.engine import Connection, Engine + +from microcosm.build.telemetry_emitter_service.database import create_spool_engine + +_MIGRATION_TARGET = "head" + + +@contextmanager +def alembic_config( + *, + connection: Connection | None = None, +) -> Iterator[Config]: + """Yield Alembic configuration with migrations on the filesystem.""" + + migration_resources = files(__package__).joinpath("alembic") + with as_file(migration_resources) as migration_directory: + config = Config() + config.set_main_option("script_location", str(migration_directory)) + if connection is not None: + config.attributes["connection"] = connection + yield config + + +def upgrade_spool_database(engine: Engine) -> None: + """Apply every committed spool migration to an engine.""" + + with engine.begin() as connection: + with alembic_config(connection=connection) as config: + command.upgrade(config, _MIGRATION_TARGET) + + +def current_database_revision(path: Path | str) -> str | None: + """Return the Alembic revision recorded by one spool database.""" + + engine = create_spool_engine(path) + try: + with engine.connect() as connection: + return MigrationContext.configure(connection).get_current_revision() + finally: + engine.dispose() + + +def migration_head_revision() -> str | None: + """Return the single current head from the packaged migration history.""" + + with alembic_config() as config: + return ScriptDirectory.from_config(config).get_current_head() diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/models.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/models.py new file mode 100644 index 000000000..87f1c678d --- /dev/null +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/models.py @@ -0,0 +1,130 @@ +"""SQLAlchemy ORM models for the local telemetry spool.""" + +from __future__ import annotations + +import json +from typing import Any + +from sqlalchemy import ( + ForeignKeyConstraint, + Index, + Integer, + Text, + UniqueConstraint, +) +from sqlalchemy.orm import DeclarativeBase, Mapped, mapped_column, relationship +from sqlalchemy.types import TypeDecorator + +from microcosm.build.telemetry_emitter_service.constants import ( + UPLOAD_STATE_PENDING, +) + + +class JsonObjectText(TypeDecorator[dict[str, Any]]): + """Store JSON objects in the spool's existing text-column format.""" + + impl = Text + cache_ok = True + + def process_bind_param( + self, + value: dict[str, Any] | None, + dialect, + ) -> str | None: + if value is None: + return None + return _serialize_json_object(value) + + def process_result_value( + self, + value: str | None, + dialect, + ) -> dict[str, Any] | None: + if value is None: + return None + decoded = json.loads(value) + if not isinstance(decoded, dict): + raise ValueError("telemetry spool JSON must contain an object") + return decoded + + +def serialized_json_length(value: dict[str, Any]) -> int: + """Return the stored character count for one mapped JSON object.""" + + return len(_serialize_json_object(value)) + + +def _serialize_json_object(value: dict[str, Any]) -> str: + return json.dumps( + value, + separators=(",", ":"), + sort_keys=True, + allow_nan=False, + ) + + +class SpoolModel(DeclarativeBase): + """Declarative base for the local telemetry spool.""" + + +class TelemetryRunRecord(SpoolModel): + """One telemetry producer registered for one build run.""" + + __tablename__ = "telemetry_runs" + + run_id: Mapped[str] = mapped_column(Text, primary_key=True) + producer_id: Mapped[str] = mapped_column(Text, primary_key=True) + registration: Mapped[dict[str, Any]] = mapped_column( + "registration_json", + JsonObjectText, + nullable=False, + ) + next_sequence: Mapped[int] = mapped_column(Integer, nullable=False, default=1) + upload_state: Mapped[str] = mapped_column( + Text, + nullable=False, + default=UPLOAD_STATE_PENDING, + ) + local_only_reason: Mapped[str | None] = mapped_column(Text) + updated_at: Mapped[str] = mapped_column(Text, nullable=False) + events: Mapped[list[TelemetryEventRecord]] = relationship( + back_populates="run", + cascade="all, delete-orphan", + ) + + +class TelemetryEventRecord(SpoolModel): + """One ordered telemetry event awaiting collector acknowledgement.""" + + __tablename__ = "telemetry_events" + __table_args__ = ( + UniqueConstraint( + "run_id", + "producer_id", + "sequence", + name="uq_telemetry_events_run_producer_sequence", + ), + ForeignKeyConstraint( + ["run_id", "producer_id"], + ["telemetry_runs.run_id", "telemetry_runs.producer_id"], + name="fk_telemetry_events_run", + ), + Index( + "telemetry_events_run_sequence", + "run_id", + "producer_id", + "sequence", + ), + ) + + event_id: Mapped[str] = mapped_column(Text, primary_key=True) + run_id: Mapped[str] = mapped_column(Text, nullable=False) + producer_id: Mapped[str] = mapped_column(Text, nullable=False) + sequence: Mapped[int] = mapped_column(Integer, nullable=False) + payload: Mapped[dict[str, Any]] = mapped_column( + "payload_json", + JsonObjectText, + nullable=False, + ) + created_at: Mapped[str] = mapped_column(Text, nullable=False) + run: Mapped[TelemetryRunRecord] = relationship(back_populates="events") diff --git a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py index 7474454a7..1ef43c526 100644 --- a/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py +++ b/packages/microcosm-build/src/microcosm/build/telemetry_emitter_service/spool.py @@ -2,55 +2,40 @@ from __future__ import annotations -import json -import sqlite3 import threading import time import uuid from collections.abc import Mapping from datetime import UTC, datetime, timedelta from pathlib import Path -from typing import Any, Final +from typing import Any + +from sqlalchemy import func, select +from sqlalchemy.orm import Session from microcosm.build.telemetry_emitter_service.constants import ( BATCH_SIZE, - DATABASE_TIMEOUT_SECONDS, - LOCAL_ONLY_PRE_ELIGIBILITY, MAX_QUEUED_BYTES, PRUNE_INTERVAL_SECONDS, RETENTION_DAYS, UPLOAD_STATE_LOCAL_ONLY, UPLOAD_STATE_PENDING, ) +from microcosm.build.telemetry_emitter_service.database import ( + create_spool_engine, + create_spool_session_factory, +) +from microcosm.build.telemetry_emitter_service.migrations import ( + upgrade_spool_database, +) +from microcosm.build.telemetry_emitter_service.models import ( + TelemetryEventRecord, + TelemetryRunRecord, + serialized_json_length, +) from microcosm.build.telemetry_emitter_service.timestamps import utc_now from microcosm.build.telemetry_protocol import TELEMETRY_SCHEMA_VERSION -_DATABASE_SCHEMA: Final = f""" -CREATE TABLE IF NOT EXISTS telemetry_runs ( - run_id TEXT NOT NULL, - producer_id TEXT NOT NULL, - registration_json TEXT NOT NULL, - next_sequence INTEGER NOT NULL DEFAULT 1, - upload_state TEXT NOT NULL DEFAULT '{UPLOAD_STATE_PENDING}', - local_only_reason TEXT, - updated_at TEXT NOT NULL, - PRIMARY KEY(run_id, producer_id) -); -CREATE TABLE IF NOT EXISTS telemetry_events ( - event_id TEXT PRIMARY KEY, - run_id TEXT NOT NULL, - producer_id TEXT NOT NULL, - sequence INTEGER NOT NULL, - payload_json TEXT NOT NULL, - created_at TEXT NOT NULL, - UNIQUE(run_id, producer_id, sequence), - FOREIGN KEY(run_id, producer_id) - REFERENCES telemetry_runs(run_id, producer_id) -); -CREATE INDEX IF NOT EXISTS telemetry_events_run_sequence - ON telemetry_events(run_id, producer_id, sequence); -""" - class EventSpool: """Small SQLite queue shared by successive emitter service processes.""" @@ -62,75 +47,38 @@ def __init__(self, path: Path | str) -> None: self.path.parent.chmod(0o700) except OSError: pass - self._connection = sqlite3.connect( - self.path, - timeout=DATABASE_TIMEOUT_SECONDS, - check_same_thread=False, - ) - self._connection.row_factory = sqlite3.Row + self._engine = create_spool_engine(self.path) + upgrade_spool_database(self._engine) + self._session_factory = create_spool_session_factory(self._engine) self._lock = threading.RLock() self._last_prune_at = 0.0 - self._initialize_database() self.prune() - def _initialize_database(self) -> None: - with self._lock, self._connection: - self._connection.execute("PRAGMA journal_mode=WAL") - self._connection.execute("PRAGMA synchronous=NORMAL") - self._connection.executescript(_DATABASE_SCHEMA) - columns = { - row["name"] - for row in self._connection.execute( - "PRAGMA table_info(telemetry_runs)" - ).fetchall() - } - if "upload_state" not in columns: - self._connection.execute( - "ALTER TABLE telemetry_runs ADD COLUMN " - f"upload_state TEXT NOT NULL DEFAULT '{UPLOAD_STATE_PENDING}'" - ) - self._connection.execute( - "UPDATE telemetry_runs SET upload_state = ?", - (UPLOAD_STATE_LOCAL_ONLY,), - ) - if "local_only_reason" not in columns: - self._connection.execute( - "ALTER TABLE telemetry_runs ADD COLUMN local_only_reason TEXT" - ) - self._connection.execute( - "UPDATE telemetry_runs SET local_only_reason = ? " - "WHERE upload_state = ?", - (LOCAL_ONLY_PRE_ELIGIBILITY, UPLOAD_STATE_LOCAL_ONLY), - ) - def register(self, registration: Mapping[str, Any]) -> None: """Create or refresh a producer registration.""" run_id = str(registration["run_id"]) producer_id = str(registration["producer_id"]) - encoded = json.dumps(registration, separators=(",", ":"), sort_keys=True) - with self._lock, self._connection: - existing = self._connection.execute( - """ - SELECT registration_json FROM telemetry_runs - WHERE run_id = ? AND producer_id = ? - """, - (run_id, producer_id), - ).fetchone() + with self._lock, self._session_factory.begin() as session: + existing = session.get(TelemetryRunRecord, (run_id, producer_id)) if existing is not None: - old = json.loads(existing["registration_json"]) - if old.get("producer_id") != registration.get("producer_id"): + if existing.registration.get("producer_id") != registration.get( + "producer_id" + ): raise ValueError(f"run_id {run_id!r} already has another producer") - self._connection.execute( - """ - INSERT INTO telemetry_runs ( - run_id, producer_id, registration_json, next_sequence, updated_at - ) VALUES (?, ?, ?, 1, ?) - ON CONFLICT(run_id, producer_id) DO UPDATE SET - registration_json = excluded.registration_json, - updated_at = excluded.updated_at - """, - (run_id, producer_id, encoded, utc_now()), + existing.registration = dict(registration) + existing.updated_at = utc_now() + return + session.add( + TelemetryRunRecord( + run_id=run_id, + producer_id=producer_id, + registration=dict(registration), + next_sequence=1, + upload_state=UPLOAD_STATE_PENDING, + local_only_reason=None, + updated_at=utc_now(), + ) ) def append( @@ -144,17 +92,11 @@ def append( run_id = str(registration["run_id"]) producer_id = str(registration["producer_id"]) - with self._lock, self._connection: - row = self._connection.execute( - """ - SELECT next_sequence FROM telemetry_runs - WHERE run_id = ? AND producer_id = ? - """, - (run_id, producer_id), - ).fetchone() - if row is None: + with self._lock, self._session_factory.begin() as session: + run = session.get(TelemetryRunRecord, (run_id, producer_id)) + if run is None: raise KeyError(run_id) - sequence = int(row["next_sequence"]) + sequence = run.next_sequence event_id = uuid.uuid4().hex payload = { "schema_version": TELEMETRY_SCHEMA_VERSION, @@ -170,24 +112,18 @@ def append( "details": event.get("details") or {}, "resources": resources, } - encoded = json.dumps(payload, separators=(",", ":"), allow_nan=False) - self._connection.execute( - """ - INSERT INTO telemetry_events ( - event_id, run_id, producer_id, sequence, - payload_json, created_at - ) VALUES (?, ?, ?, ?, ?, ?) - """, - (event_id, run_id, producer_id, sequence, encoded, utc_now()), - ) - self._connection.execute( - """ - UPDATE telemetry_runs - SET next_sequence = ?, updated_at = ? - WHERE run_id = ? AND producer_id = ? - """, - (sequence + 1, utc_now(), run_id, producer_id), + session.add( + TelemetryEventRecord( + event_id=event_id, + run_id=run_id, + producer_id=producer_id, + sequence=sequence, + payload=payload, + created_at=utc_now(), + ) ) + run.next_sequence = sequence + 1 + run.updated_at = utc_now() if time.monotonic() - self._last_prune_at >= PRUNE_INTERVAL_SECONDS: self.prune() return payload @@ -195,19 +131,16 @@ def append( def pending_runs(self) -> list[dict[str, Any]]: """Return registrations that have events eligible for delivery.""" - with self._lock: - rows = self._connection.execute( - """ - SELECT DISTINCT r.registration_json - FROM telemetry_runs r - JOIN telemetry_events e - ON e.run_id = r.run_id AND e.producer_id = r.producer_id - WHERE r.upload_state = ? - ORDER BY r.updated_at - """, - (UPLOAD_STATE_PENDING,), - ).fetchall() - return [json.loads(row["registration_json"]) for row in rows] + statement = ( + select(TelemetryRunRecord) + .join(TelemetryRunRecord.events) + .where(TelemetryRunRecord.upload_state == UPLOAD_STATE_PENDING) + .order_by(TelemetryRunRecord.updated_at) + .distinct() + ) + with self._lock, self._session_factory() as session: + runs = session.scalars(statement).all() + return [dict(run.registration) for run in runs] def make_local_only( self, @@ -217,38 +150,25 @@ def make_local_only( ) -> None: """Permanently exclude one producer's queued events from upload.""" - with self._lock, self._connection: - self._connection.execute( - """ - UPDATE telemetry_runs - SET upload_state = ?, local_only_reason = ?, updated_at = ? - WHERE run_id = ? AND producer_id = ? - """, - ( - UPLOAD_STATE_LOCAL_ONLY, - reason, - utc_now(), - run_id, - producer_id, - ), - ) + with self._lock, self._session_factory.begin() as session: + run = session.get(TelemetryRunRecord, (run_id, producer_id)) + if run is None: + return + run.upload_state = UPLOAD_STATE_LOCAL_ONLY + run.local_only_reason = reason + run.updated_at = utc_now() def has_deliverable(self) -> bool: """Return whether any queued event remains eligible for delivery.""" - with self._lock: - row = self._connection.execute( - """ - SELECT 1 - FROM telemetry_events e - JOIN telemetry_runs r - ON r.run_id = e.run_id AND r.producer_id = e.producer_id - WHERE r.upload_state = ? - LIMIT 1 - """, - (UPLOAD_STATE_PENDING,), - ).fetchone() - return row is not None + statement = ( + select(TelemetryEventRecord) + .join(TelemetryEventRecord.run) + .where(TelemetryRunRecord.upload_state == UPLOAD_STATE_PENDING) + .limit(1) + ) + with self._lock, self._session_factory() as session: + return session.scalar(statement) is not None def batch( self, @@ -258,85 +178,81 @@ def batch( ) -> list[dict[str, Any]]: """Return the next ordered batch for one producer.""" - with self._lock: - rows = self._connection.execute( - """ - SELECT payload_json FROM telemetry_events - WHERE run_id = ? AND producer_id = ? - ORDER BY sequence LIMIT ? - """, - (run_id, producer_id, limit), - ).fetchall() - return [json.loads(row["payload_json"]) for row in rows] + statement = ( + select(TelemetryEventRecord) + .where( + TelemetryEventRecord.run_id == run_id, + TelemetryEventRecord.producer_id == producer_id, + ) + .order_by(TelemetryEventRecord.sequence) + .limit(limit) + ) + with self._lock, self._session_factory() as session: + events = session.scalars(statement).all() + return [dict(event.payload) for event in events] def acknowledge(self, event_ids: list[str]) -> None: """Remove events acknowledged by the collector.""" if not event_ids: return - placeholders = ",".join("?" for _ in event_ids) - with self._lock, self._connection: - self._connection.execute( - f"DELETE FROM telemetry_events WHERE event_id IN ({placeholders})", - event_ids, - ) + statement = select(TelemetryEventRecord).where( + TelemetryEventRecord.event_id.in_(event_ids) + ) + with self._lock, self._session_factory.begin() as session: + for event in session.scalars(statement): + session.delete(event) def has_pending(self) -> bool: """Return whether any event remains in local storage.""" - with self._lock: - row = self._connection.execute( - "SELECT 1 FROM telemetry_events LIMIT 1" - ).fetchone() - return row is not None + statement = select(TelemetryEventRecord).limit(1) + with self._lock, self._session_factory() as session: + return session.scalar(statement) is not None def prune(self) -> None: """Enforce the age and total-size retention limits.""" cutoff = (datetime.now(UTC) - timedelta(days=RETENTION_DAYS)).isoformat() - with self._lock, self._connection: - self._connection.execute( - "DELETE FROM telemetry_events WHERE created_at < ?", - (cutoff,), + with self._lock, self._session_factory.begin() as session: + expired_events = session.scalars( + select(TelemetryEventRecord).where( + TelemetryEventRecord.created_at < cutoff + ) + ) + for event in expired_events: + session.delete(event) + session.flush() + stored_characters = session.scalar( + select( + func.coalesce( + func.sum(func.length(TelemetryEventRecord.payload)), + 0, + ) + ) ) - size_row = self._connection.execute( - "SELECT COALESCE(SUM(LENGTH(payload_json)), 0) AS bytes " - "FROM telemetry_events" - ).fetchone() - excess = int(size_row["bytes"]) - MAX_QUEUED_BYTES + excess = int(stored_characters or 0) - MAX_QUEUED_BYTES if excess > 0: - self._remove_oldest_bytes(excess) - self._connection.execute( - """ - DELETE FROM telemetry_runs - WHERE updated_at < ? AND NOT EXISTS ( - SELECT 1 FROM telemetry_events e - WHERE e.run_id = telemetry_runs.run_id - AND e.producer_id = telemetry_runs.producer_id + self._remove_oldest_bytes(session, excess) + expired_runs = session.scalars( + select(TelemetryRunRecord).where( + TelemetryRunRecord.updated_at < cutoff, + ~TelemetryRunRecord.events.any(), ) - """, - (cutoff,), ) + for run in expired_runs: + session.delete(run) self._last_prune_at = time.monotonic() - def _remove_oldest_bytes(self, excess: int) -> None: - rows = self._connection.execute( - """ - SELECT event_id, LENGTH(payload_json) AS bytes - FROM telemetry_events ORDER BY created_at, sequence - """ - ).fetchall() + @staticmethod + def _remove_oldest_bytes(session: Session, excess: int) -> None: + statement = select(TelemetryEventRecord).order_by( + TelemetryEventRecord.created_at, + TelemetryEventRecord.sequence, + ) removed = 0 - event_ids: list[str] = [] - for row in rows: - event_ids.append(row["event_id"]) - removed += int(row["bytes"]) + for event in session.scalars(statement): + session.delete(event) + removed += serialized_json_length(event.payload) if removed >= excess: break - if not event_ids: - return - placeholders = ",".join("?" for _ in event_ids) - self._connection.execute( - f"DELETE FROM telemetry_events WHERE event_id IN ({placeholders})", - event_ids, - ) diff --git a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py index 8d1cb5745..bb85c366c 100644 --- a/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py +++ b/packages/microcosm-build/tests/engine_free/shared/test_telemetry_emitter.py @@ -1,7 +1,6 @@ import json import os import socket -import sqlite3 import tempfile import threading import time @@ -9,6 +8,10 @@ from pathlib import Path from types import SimpleNamespace +from alembic import command +from sqlalchemy import Column, Integer, MetaData, String, Table, create_engine +from sqlalchemy.orm import Session + from microcosm.build.telemetry_emitter import ( LocalTelemetryEmitter, TelemetryRun, @@ -20,9 +23,25 @@ ) from microcosm.build.telemetry_emitter_service import collector as collector_module from microcosm.build.telemetry_emitter_service import resources as resources_module +from microcosm.build.telemetry_emitter_service import spool as spool_module from microcosm.build.telemetry_emitter_service.constants import ( PRODUCTION_COLLECTOR_URL, ) +from microcosm.build.telemetry_emitter_service.database import ( + create_spool_engine, + sqlite_database_url, +) +from microcosm.build.telemetry_emitter_service.migrations import ( + alembic_config, + current_database_revision, + migration_head_revision, +) +from microcosm.build.telemetry_emitter_service.models import ( + SpoolModel, + TelemetryEventRecord, + TelemetryRunRecord, + serialized_json_length, +) from microcosm.build.telemetry_protocol import ( BUILD_COMPLETED_MESSAGE, BUILD_STARTED_MESSAGE, @@ -136,7 +155,8 @@ def test_outbound_payload_redacts_credentials_and_tracebacks() -> None: def test_event_spool_assigns_stable_sequences_and_acknowledges(tmp_path) -> None: - spool = EventSpool(tmp_path / "events.sqlite3") + spool_path = tmp_path / "events.sqlite3" + spool = EventSpool(spool_path) registration = _registration() spool.register(registration) @@ -153,6 +173,19 @@ def test_event_spool_assigns_stable_sequences_and_acknowledges(tmp_path) -> None spool.acknowledge([first["event_id"]]) assert [event["sequence"] for event in spool.batch("run-a", "producer-a")] == [2] + assert current_database_revision(spool_path) == migration_head_revision() + + +def test_alembic_schema_matches_sqlalchemy_models(tmp_path) -> None: + spool_path = tmp_path / "events.sqlite3" + EventSpool(spool_path) + engine = create_spool_engine(spool_path) + try: + with engine.begin() as connection: + with alembic_config(connection=connection) as config: + command.check(config) + finally: + engine.dispose() def test_sequential_stage_updates_close_the_previous_stage() -> None: @@ -200,54 +233,114 @@ def test_event_spool_keeps_repeated_run_producers_separate(tmp_path) -> None: def test_pre_eligibility_spool_is_not_uploaded_after_upgrade(tmp_path) -> None: path = tmp_path / "events.sqlite3" registration = _registration() - connection = sqlite3.connect(path) - connection.executescript( - """ - CREATE TABLE telemetry_runs ( - run_id TEXT NOT NULL, - producer_id TEXT NOT NULL, - registration_json TEXT NOT NULL, - next_sequence INTEGER NOT NULL DEFAULT 1, - updated_at TEXT NOT NULL, - PRIMARY KEY(run_id, producer_id) - ); - CREATE TABLE telemetry_events ( - event_id TEXT PRIMARY KEY, - run_id TEXT NOT NULL, - producer_id TEXT NOT NULL, - sequence INTEGER NOT NULL, - payload_json TEXT NOT NULL, - created_at TEXT NOT NULL, - UNIQUE(run_id, producer_id, sequence) - ); - """ + metadata = MetaData() + runs = Table( + "telemetry_runs", + metadata, + Column("run_id", String, primary_key=True), + Column("producer_id", String, primary_key=True), + Column("registration_json", String, nullable=False), + Column("next_sequence", Integer, nullable=False, default=1), + Column("updated_at", String, nullable=False), ) - connection.execute( - "INSERT INTO telemetry_runs VALUES (?, ?, ?, 2, ?)", - ( - "run-a", - "producer-a", - json.dumps(registration), - "2026-10-02T10:00:00+00:00", - ), + events = Table( + "telemetry_events", + metadata, + Column("event_id", String, primary_key=True), + Column("run_id", String, nullable=False), + Column("producer_id", String, nullable=False), + Column("sequence", Integer, nullable=False), + Column("payload_json", String, nullable=False), + Column("created_at", String, nullable=False), ) - connection.execute( - "INSERT INTO telemetry_events VALUES (?, ?, ?, 1, ?, ?)", - ( - "old-event", - "run-a", - "producer-a", - json.dumps({"event_id": "old-event"}), - "2026-10-02T10:00:00+00:00", - ), - ) - connection.commit() - connection.close() + engine = create_engine(sqlite_database_url(path)) + metadata.create_all(engine) + with engine.begin() as connection: + connection.execute( + runs.insert().values( + run_id="run-a", + producer_id="producer-a", + registration_json=json.dumps(registration), + next_sequence=2, + updated_at="2026-10-02T10:00:00+00:00", + ) + ) + connection.execute( + events.insert().values( + event_id="old-event", + run_id="run-a", + producer_id="producer-a", + sequence=1, + payload_json=json.dumps({"event_id": "old-event"}), + created_at="2026-10-02T10:00:00+00:00", + ) + ) + engine.dispose() spool = EventSpool(path) assert spool.has_pending() assert spool.pending_runs() == [] + assert current_database_revision(path) == migration_head_revision() + + +def test_current_pre_alembic_spool_is_adopted_without_losing_events(tmp_path) -> None: + path = tmp_path / "events.sqlite3" + registration = _registration() + engine = create_engine(sqlite_database_url(path)) + SpoolModel.metadata.create_all(engine) + with Session(engine) as session, session.begin(): + session.add( + TelemetryRunRecord( + run_id="run-a", + producer_id="producer-a", + registration=registration, + next_sequence=2, + upload_state="pending", + local_only_reason=None, + updated_at="2026-10-02T10:00:00+00:00", + ) + ) + session.add( + TelemetryEventRecord( + event_id="existing-event", + run_id="run-a", + producer_id="producer-a", + sequence=1, + payload={"event_id": "existing-event", "sequence": 1}, + created_at="2026-10-02T10:00:00+00:00", + ) + ) + engine.dispose() + + spool = EventSpool(path) + + assert spool.pending_runs() == [registration] + assert spool.batch("run-a", "producer-a") == [ + {"event_id": "existing-event", "sequence": 1} + ] + assert current_database_revision(path) == migration_head_revision() + + +def test_event_spool_prunes_oldest_events_to_size_limit( + tmp_path, + monkeypatch, +) -> None: + spool = EventSpool(tmp_path / "events.sqlite3") + registration = _registration() + spool.register(registration) + first = spool.append(registration, _event("first")) + second = spool.append(registration, _event("second")) + monkeypatch.setattr( + spool_module, + "MAX_QUEUED_BYTES", + serialized_json_length(second), + ) + + spool.prune() + + assert first["event_id"] != second["event_id"] + assert spool.batch("run-a", "producer-a") == [second] def test_collector_delivery_exchanges_hf_token_then_flushes( diff --git a/uv.lock b/uv.lock index 37d480654..774fbc6d8 100644 --- a/uv.lock +++ b/uv.lock @@ -1,5 +1,5 @@ version = 1 -revision = 3 +revision = 5 requires-python = ">=3.13" resolution-markers = [ "python_full_version >= '3.14' and sys_platform == 'win32'", @@ -22,6 +22,20 @@ members = [ "microcosm-workspace", ] +[[package]] +name = "alembic" +version = "1.20.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "mako" }, + { name = "sqlalchemy" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ed/aa/02910bdb8e2f1444f6654d5b296cd827d126f82209050ee7b1000f92ac4b/alembic-1.20.0.tar.gz", hash = "sha256:db505480647bc60386c5369402f4a57a506b7539c9e9ef5e270d45cbbe4939bf", size = 2093272, upload-time = "2026-09-11T19:09:11.126Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/3f/27/78a89b55b0904d222183164e079b4ca56208e94eff1d35ad1f1ad5be9b06/alembic-1.20.0-py3-none-any.whl", hash = "sha256:77eb101048d95f982c0353e9233404889dcd7a6fc244c107836c0e2fc9cf7d9d", size = 268719, upload-time = "2026-09-11T19:09:12.88Z" }, +] + [[package]] name = "annotated-doc" version = "0.0.4" @@ -206,7 +220,7 @@ name = "cuda-bindings" version = "13.3.1" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "cuda-pathfinder", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "cuda-pathfinder" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/cc/6e/2394f8163360f8391f8f1b7e72d300a82724edb81a7b7084c799fbd4c91f/cuda_bindings-13.3.1-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9efb21c1ee64981e184b9e0ba5eb3179e5ba3d4b51665a6cb52b8ef3d01a7cbf", size = 5920504, upload-time = "2026-05-29T23:11:56.883Z" }, @@ -235,34 +249,34 @@ wheels = [ [package.optional-dependencies] cudart = [ - { name = "nvidia-cuda-runtime", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cuda-runtime" }, ] cufft = [ - { name = "nvidia-cufft", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cufft" }, ] cufile = [ - { name = "nvidia-cufile", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cufile" }, ] cupti = [ - { name = "nvidia-cuda-cupti", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cuda-cupti" }, ] curand = [ - { name = "nvidia-curand", marker = "sys_platform == 'linux'" }, + { name = "nvidia-curand" }, ] cusolver = [ - { name = "nvidia-cusolver", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cusolver" }, ] cusparse = [ - { name = "nvidia-cusparse", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cusparse" }, ] nvjitlink = [ - { name = "nvidia-nvjitlink", marker = "sys_platform == 'linux'" }, + { name = "nvidia-nvjitlink" }, ] nvrtc = [ - { name = "nvidia-cuda-nvrtc", marker = "sys_platform == 'linux'" }, + { name = "nvidia-cuda-nvrtc" }, ] nvtx = [ - { name = "nvidia-nvtx", marker = "sys_platform == 'linux'" }, + { name = "nvidia-nvtx" }, ] [[package]] @@ -653,6 +667,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/41/45/1a4ed80516f02155c51f51e8cedb3c1902296743db0bbc66608a0db2814f/jsonschema_specifications-2025.9.1-py3-none-any.whl", hash = "sha256:98802fee3a11ee76ecaca44429fda8a41bff98b00a0f2838151b113f210cc6fe", size = 18437, upload-time = "2025-09-08T01:34:57.871Z" }, ] +[[package]] +name = "mako" +version = "1.4.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "markupsafe" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/5a/09/e07c4b5579a79f4b16f8d4f29f6c54514ac787c4ad506b8c4f28a0e6b0bf/mako-1.4.3.tar.gz", hash = "sha256:cd6537fe88d5fec315c55c2f8529bc4ce7a9a352ad7db3eeaa6a66e2dd4ec37a", size = 412799, upload-time = "2026-09-22T20:54:31.509Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/6d/a0/053d6af3e8f871e0073b4a36732d9e65be77a72e5434c31b94f6af78a6bb/mako-1.4.3-py3-none-any.whl", hash = "sha256:723296007c870bfd6b3f0c3230dba7198096e5269297ebf5e4eff9e7ffa39d4f", size = 80164, upload-time = "2026-09-22T20:54:33.128Z" }, +] + [[package]] name = "markdown-it-py" version = "4.2.0" @@ -743,6 +769,7 @@ name = "microcosm-build" version = "0.1.0" source = { editable = "packages/microcosm-build" } dependencies = [ + { name = "alembic" }, { name = "huggingface-hub" }, { name = "jsonschema" }, { name = "microcosm-calibrate" }, @@ -756,6 +783,7 @@ dependencies = [ { name = "pyyaml" }, { name = "referencing" }, { name = "scipy" }, + { name = "sqlalchemy" }, ] [package.optional-dependencies] @@ -781,6 +809,7 @@ dev = [ [package.metadata] requires-dist = [ + { name = "alembic", specifier = ">=1.13.3,<2" }, { name = "h5py", marker = "extra == 'uk'", specifier = ">=3" }, { name = "h5py", marker = "extra == 'us'", specifier = ">=3" }, { name = "huggingface-hub", specifier = ">=0.20" }, @@ -802,6 +831,7 @@ requires-dist = [ { name = "pyyaml", specifier = ">=6" }, { name = "referencing", specifier = ">=0.35,<1" }, { name = "scipy", specifier = ">=1.13" }, + { name = "sqlalchemy", specifier = ">=2,<3" }, { name = "tables", marker = "extra == 'uk'", specifier = ">=3" }, { name = "tables", marker = "extra == 'us'", specifier = ">=3" }, ] @@ -1251,7 +1281,7 @@ name = "nvidia-cublas" version = "13.1.1.3" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cuda-nvrtc", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-cuda-nvrtc" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/a7/a1/0bd24ee8c8d03adac032fd2909426a00c88f8c57961b1277ded97f91119f/nvidia_cublas-13.1.1.3-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:b7a210458267ac818974c53038fbec2e969d5c99f305ab15c72522fa9f001dd5", size = 542848918, upload-time = "2026-04-08T18:46:22.985Z" }, @@ -1290,7 +1320,7 @@ name = "nvidia-cudnn-cu13" version = "9.20.0.48" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cublas", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-cublas" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/56/c5/83384d846b2fd17c44bd499b36c75a45ed4f095fbbb2252294e89cea5c5c/nvidia_cudnn_cu13-9.20.0.48-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:e31454ae00094b0c55319d9d15b6fa2fc50a9e1c0f5c8c80fb75258234e731e1", size = 444574296, upload-time = "2026-03-09T19:28:27.751Z" }, @@ -1302,7 +1332,7 @@ name = "nvidia-cufft" version = "12.0.0.61" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-nvjitlink", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-nvjitlink" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/8b/ae/f417a75c0259e85c1d2f83ca4e960289a5f814ed0cea74d18c353d3e989d/nvidia_cufft-12.0.0.61-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:2708c852ef8cd89d1d2068bdbece0aa188813a0c934db3779b9b1faa8442e5f5", size = 214053554, upload-time = "2025-09-04T08:31:38.196Z" }, @@ -1332,9 +1362,9 @@ name = "nvidia-cusolver" version = "12.0.4.66" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cublas", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, - { name = "nvidia-cusparse", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, - { name = "nvidia-nvjitlink", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-cublas" }, + { name = "nvidia-cusparse" }, + { name = "nvidia-nvjitlink" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/c8/c3/b30c9e935fc01e3da443ec0116ed1b2a009bb867f5324d3f2d7e533e776b/nvidia_cusolver-12.0.4.66-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:02c2457eaa9e39de20f880f4bd8820e6a1cfb9f9a34f820eb12a155aa5bc92d2", size = 223467760, upload-time = "2025-09-04T08:33:04.222Z" }, @@ -1346,7 +1376,7 @@ name = "nvidia-cusparse" version = "12.6.3.3" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-nvjitlink", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-nvjitlink" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/f8/94/5c26f33738ae35276672f12615a64bd008ed5be6d1ebcb23579285d960a9/nvidia_cusparse-12.6.3.3-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:80bcc4662f23f1054ee334a15c72b8940402975e0eab63178fc7e670aa59472c", size = 162155568, upload-time = "2025-09-04T08:33:42.864Z" }, @@ -1477,7 +1507,7 @@ name = "pexpect" version = "4.9.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "ptyprocess", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "ptyprocess" }, ] sdist = { url = "https://files.pythonhosted.org/packages/42/92/cc564bf6381ff43ce1f4d06852fc19a2f11d180f23dc32d9588bee2f149d/pexpect-4.9.0.tar.gz", hash = "sha256:ee7d41123f3c9911050ea2c2dac107568dc43b2d3b0c7557a33212c398ead30f", size = 166450, upload-time = "2023-11-25T09:07:26.339Z" } wheels = [ @@ -2131,6 +2161,68 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/19/a0/c484f69a0ebf88a9b9fd0ac28176f8df66550f46714f1b0c5ec0b360824e/spm_calculator-1.0.0-py3-none-any.whl", hash = "sha256:e354937a5e1a4045d4966ed594a528d8b02866fabaac9bb5672017004b627305", size = 7395384, upload-time = "2026-09-11T15:45:10.492Z" }, ] +[[package]] +name = "sqlalchemy" +version = "2.1.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/02/4b/81d972a46c9f1d978af1795e2abc988855c711453552cb45d75f6e4abbf5/sqlalchemy-2.1.3.tar.gz", hash = "sha256:ade5281df06038c6394f532590d592d3381ee5d11b9e0006a2570aef41145118", size = 10525778, upload-time = "2026-10-02T22:57:00.308Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/1a/71/68c29931fb2c5178241f7d0c63dbdd800ad65685eac70d7f2117145b5e3c/sqlalchemy-2.1.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:260bdd8baa1dc631d76d14b26ad009ebbb665b4469666afef8dcaa1f964016cb", size = 2457479, upload-time = "2026-10-03T02:39:19.741Z" }, + { url = "https://files.pythonhosted.org/packages/94/21/5dbd91816b2cdb367bca3195f300ffeca916762e088d6924fc33a18f50ad/sqlalchemy-2.1.3-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:5fc01a8047d6a4f4e997441994f07af79fdd470fc80669751e32f91153d6b491", size = 4582199, upload-time = "2026-10-02T23:37:44.508Z" }, + { url = "https://files.pythonhosted.org/packages/32/d1/79962ce06d6cd34516def113f5170184fa9401ebddc59fd6cb18f6e919f3/sqlalchemy-2.1.3-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b1b3ff10eba0f9290142375e667351c99d24ee7fcff6e0f85bd8be9d39b50a17", size = 4632923, upload-time = "2026-10-02T23:35:55.318Z" }, + { url = "https://files.pythonhosted.org/packages/3c/94/03e98031dc9a1738c580c3bf01beeafe695f247eb647f4e8dda3fab2454f/sqlalchemy-2.1.3-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:efb1ebf695e80d9a69f5bbff9d812d2491fd168b27d6b42ad4c8a4d3d6faaef1", size = 4300224, upload-time = "2026-10-02T23:46:45.495Z" }, + { url = "https://files.pythonhosted.org/packages/3a/1f/388c87607eb6c7abee4866355eeebb6221b3bcb413a64ce4a5cfc3edd22f/sqlalchemy-2.1.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:a95754928d47b37ea5c4b5180101db2afabef1da0dd5cd0c573b547d5d036ff0", size = 4507170, upload-time = "2026-10-02T23:38:29.697Z" }, + { url = "https://files.pythonhosted.org/packages/57/73/5515a3f555d31748564fbeb0b9e0b6ab7a647bf2ea0cdb6d233f20f710e3/sqlalchemy-2.1.3-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:7606b413730ac6002930faa96f52195ed6969040b78071a3336370661f5e9e09", size = 4301439, upload-time = "2026-10-02T23:46:47.845Z" }, + { url = "https://files.pythonhosted.org/packages/a3/a0/0eae9b96f6f8a09f3e5d388f6fc5d6fe8d5ff2aced2a5c75bea7333d445e/sqlalchemy-2.1.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:76c3d5c7ada0925ff12e8c2395c547f88fcfccc13368ff8e8dd55ab5e28dcdab", size = 4594441, upload-time = "2026-10-03T02:46:16.928Z" }, + { url = "https://files.pythonhosted.org/packages/f5/b1/b6d681015414ce9048c07ef3ac49b693c1b8373de4716ee4bdeff4ec19a0/sqlalchemy-2.1.3-cp313-cp313-win32.whl", hash = "sha256:cb6fc5c856e8f1c0228bf74b636c0a03a6c12dd502ced7aa82ced21dcdc7f547", size = 2371547, upload-time = "2026-10-03T02:49:23.549Z" }, + { url = "https://files.pythonhosted.org/packages/56/ad/35aa4d737016640c6da7573a4bcd9d91b59a87359bc6df075627963a4781/sqlalchemy-2.1.3-cp313-cp313-win_amd64.whl", hash = "sha256:82d4c63d338737d593f1d29c4dae0a124577a9a76186d90b5c0f9ec9cbeeab35", size = 2421558, upload-time = "2026-10-03T02:49:25.448Z" }, + { url = "https://files.pythonhosted.org/packages/fc/42/0dcc7cd92a3f4773376a5488b6ce0d6d75d2d029ad7e7f9e0ecc5fb48fcd/sqlalchemy-2.1.3-cp313-cp313-win_arm64.whl", hash = "sha256:0ab6f42761cc5c48d2f3e8a70810d10630d8222a51a7111cf9b3a48f22c3f148", size = 2382507, upload-time = "2026-10-03T02:42:42.812Z" }, + { url = "https://files.pythonhosted.org/packages/11/55/d626c933b7d034619c57120fafb47c0d44b2553ad2b8f74e904a40c01258/sqlalchemy-2.1.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:660c6e16d254568874d4b77fd593aa5426f55cf40d0acc8b5364a5e61cc1044d", size = 2460626, upload-time = "2026-10-03T02:39:21.751Z" }, + { url = "https://files.pythonhosted.org/packages/35/90/9b276c8eed59694d5c2291588fae9c51f25a5c63efb00ef9a7c6e754a2cd/sqlalchemy-2.1.3-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7277ba31ffb636666ed207d2aaf74de3cd2ee78436da7552bd4eeadf1f67740a", size = 4576390, upload-time = "2026-10-02T23:38:31.943Z" }, + { url = "https://files.pythonhosted.org/packages/60/21/818e89fe73ba4600abf5090e0d2356d92de9ca3a8aa9222d64801e510bd2/sqlalchemy-2.1.3-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b728c406b1b202e8998e8f9adacb56773bfeea67da74252ad3025bcb8f5863f1", size = 4606177, upload-time = "2026-10-03T02:46:19.291Z" }, + { url = "https://files.pythonhosted.org/packages/88/75/ab96bb0e27153d123340d6f1b14fac3713c03b7e3ca3f74f4c0715345775/sqlalchemy-2.1.3-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:85623c57c3696dcbc4fc909bbf335a2f0c27ae36f762d83810067fb49abc040e", size = 4300438, upload-time = "2026-10-02T23:46:50.03Z" }, + { url = "https://files.pythonhosted.org/packages/94/49/c7e0223fe5553ff1c31853c07f158f75245f8d435435755a789249f5121a/sqlalchemy-2.1.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:ac0de6dc326cc99ac31206359e9fc2c13345f667edeb6ef556585fedac4ff698", size = 4506089, upload-time = "2026-10-02T23:38:34.209Z" }, + { url = "https://files.pythonhosted.org/packages/10/83/d6aad5e1a5b3a3e9257f932971b7f2105be6669205d4d9946f5f87d73216/sqlalchemy-2.1.3-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:65e002b794efa1bacbbde9fa973e5340d6c3c29b53e592b863a281802829319d", size = 4301730, upload-time = "2026-10-02T23:46:52.115Z" }, + { url = "https://files.pythonhosted.org/packages/04/54/310b9fd03ba7de6359185a0914bcd0cf45a76c6314e49d1ec814a50ae13e/sqlalchemy-2.1.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:65e4813c78eaa679c9369b289e67fdba924db6be14067700c5ccbda37953af9a", size = 4565315, upload-time = "2026-10-03T02:46:21.64Z" }, + { url = "https://files.pythonhosted.org/packages/d4/8b/f6f7d46bc4e7bdc16fa78ec9a390f01852f2b080b7c75e83db4947b707c6/sqlalchemy-2.1.3-cp314-cp314-win32.whl", hash = "sha256:726987e7f8500614016be056ba5c9715b760ab465295741e2087b29753d72632", size = 2378402, upload-time = "2026-10-03T02:49:27.524Z" }, + { url = "https://files.pythonhosted.org/packages/9a/5f/dabd1ea48dd6c45383526c7f2e453fea15dcb53235edbadab49857fd7bfd/sqlalchemy-2.1.3-cp314-cp314-win_amd64.whl", hash = "sha256:0560458a775917bcd7a47eccf4f9e0b135fd9b72d7b819cb1bd60581de946f9a", size = 2428828, upload-time = "2026-10-03T02:49:29.481Z" }, + { url = "https://files.pythonhosted.org/packages/23/aa/ecba6a6141d24dbd4b002b451e0c410b7c6cb82a8a873f915f89314af830/sqlalchemy-2.1.3-cp314-cp314-win_arm64.whl", hash = "sha256:63c9289ed81991ba4d500611ce201531711430f5d23d7ae9a61a8842639b85ea", size = 2393599, upload-time = "2026-10-03T02:42:44.416Z" }, + { url = "https://files.pythonhosted.org/packages/33/41/f2c5eb9371cc1352a16001642702dfc2d73537995eea5cdeec7738f90a9d/sqlalchemy-2.1.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:08b2f345db8b0742e7814c49f9a210d5f5c8fb9d80d12c8a63bba762fa277477", size = 2500084, upload-time = "2026-10-02T23:35:35.774Z" }, + { url = "https://files.pythonhosted.org/packages/73/f1/fabd748dd8a1700579906649ac05414d65eca5bca44a3ac3c22856ed8d25/sqlalchemy-2.1.3-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:493b632f426579c4b0e97a5610f76c8c3d401cede442d462bf700b537035c91e", size = 4904105, upload-time = "2026-10-03T02:49:06.101Z" }, + { url = "https://files.pythonhosted.org/packages/ad/6b/a7a3d5c77c16386e16eb9dc8ba82e0cc6b03d6ee0ef6620921655df82413/sqlalchemy-2.1.3-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:55e06933580f4faa5a7b1d2c9ad04f365fe01c40465eb982e38fda8758c1e95c", size = 4814161, upload-time = "2026-10-02T23:34:39.253Z" }, + { url = "https://files.pythonhosted.org/packages/63/72/275dac552b869a21774c08a7084bad16bd99f28384b8c95ffcded1a18884/sqlalchemy-2.1.3-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:fdcee9536cb3281b8810787ba651f4e5f22bd687b177de89fba2155b055b07b0", size = 4494471, upload-time = "2026-10-02T23:59:11.642Z" }, + { url = "https://files.pythonhosted.org/packages/e8/3e/41d8071c58a20e2ba9fabc422938a219461f0b32aa537ab69afe06daa0e5/sqlalchemy-2.1.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:d2cb28779dbf54067655030d7ab3a6726c7957c56bde5ff6745ed86b5cebff41", size = 4762070, upload-time = "2026-10-03T02:49:08.052Z" }, + { url = "https://files.pythonhosted.org/packages/75/73/b747233b8ad935467eb1f2aa1c466af69831693cab3495c88842bf1bd3ac/sqlalchemy-2.1.3-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:01a90cac3f8d93fcbda0df119ef56c4d58aad5601f6b7ee5a38bfabbd5d119b8", size = 4494691, upload-time = "2026-10-02T23:59:13.733Z" }, + { url = "https://files.pythonhosted.org/packages/71/aa/ba11b725b9529879aff489e36cdaccc3fd69620fb30b89186792523bc262/sqlalchemy-2.1.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:edbc28049ab8da07387ea2b85ceee747b36b25444ffa4aaba9f6e1461d5b2634", size = 4752692, upload-time = "2026-10-03T02:54:01.04Z" }, + { url = "https://files.pythonhosted.org/packages/b6/b8/86ec625db10e285ed8459e39ab3e56b7630554bc733e64a190e248eeea25/sqlalchemy-2.1.3-cp314-cp314t-win32.whl", hash = "sha256:724f587ced076d89291032b35476d6bdb401cce1dd680c1e3418c01d9bd55ad4", size = 2436746, upload-time = "2026-10-03T02:56:47.297Z" }, + { url = "https://files.pythonhosted.org/packages/18/be/1309b0b3acfcd6e0060c815762012c66d23eece8c9eb8091da5c850a1289/sqlalchemy-2.1.3-cp314-cp314t-win_amd64.whl", hash = "sha256:935085a5b8849d324c22bc3cc09fd07b3669cc22ba5bdd3ec208829e535cdef9", size = 2504160, upload-time = "2026-10-03T02:56:49.705Z" }, + { url = "https://files.pythonhosted.org/packages/9b/b2/ac9067f4d6c43887570df54e6c340ae9c591a8e75a5fd3029a9d7deba163/sqlalchemy-2.1.3-cp314-cp314t-win_arm64.whl", hash = "sha256:2ebfff82cece947fea1ed6337d4f0613b9799f6db621c9729790fdd41edfb5ea", size = 2422266, upload-time = "2026-10-02T23:35:30.802Z" }, + { url = "https://files.pythonhosted.org/packages/a0/e5/54a863bf8e12fa97cd06bb2e012aeca53b5535bc07aabe0192a4ed77ed44/sqlalchemy-2.1.3-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:47a052dcd935a62326216bc5f0ed10ad941c77c3acc9200385e0c03c4f31d5a9", size = 2460525, upload-time = "2026-10-03T02:38:02.872Z" }, + { url = "https://files.pythonhosted.org/packages/f9/63/f6f8a8d8e8f8be456bea70ad2b9af90a60a7f98d349b178b9eaaea86431b/sqlalchemy-2.1.3-cp315-cp315-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1e5916b43a9a8f346026f663583f203d0c43febad504ff2fed504dec6e078f20", size = 4579659, upload-time = "2026-10-02T23:35:35.789Z" }, + { url = "https://files.pythonhosted.org/packages/fa/7f/9f1064ef0f8f15e3c1cb8fa7ad5e4fc7e1e311ad0e1e46bc2e77940585bb/sqlalchemy-2.1.3-cp315-cp315-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b9af78a85af63114e2650f019ddf8fffb871e659e92bffb7d0bce5753f706caf", size = 4615129, upload-time = "2026-10-02T23:38:44.119Z" }, + { url = "https://files.pythonhosted.org/packages/c4/4a/f1a002235cc785d219738e096c7c65f68e9b7ed30c810997236567c50af9/sqlalchemy-2.1.3-cp315-cp315-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:d218698c4f05728a941d6ad7c769b6491f03058e102cf811b7e4900f78fffa07", size = 4316528, upload-time = "2026-10-02T23:39:13.275Z" }, + { url = "https://files.pythonhosted.org/packages/a8/9c/6ee4e5ab55558408c16fe1713db415e45024f42cdb1c573ebceacc09d01c/sqlalchemy-2.1.3-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:80e6ccf111bc3f57530935f5541c6c1da50ac9713c418378840683532a45dd2d", size = 4509727, upload-time = "2026-10-03T02:41:25.314Z" }, + { url = "https://files.pythonhosted.org/packages/64/8c/a6bdf2874518c628fdedf24144194afa88cceb70339e5050459cdc2723cf/sqlalchemy-2.1.3-cp315-cp315-musllinux_1_2_riscv64.whl", hash = "sha256:9a3ce1ff88dfb9f9f4305da07f4fab87720b609e264d134e47ae84a52df130bd", size = 4316490, upload-time = "2026-10-02T23:39:15.342Z" }, + { url = "https://files.pythonhosted.org/packages/10/3b/d859b638495a63ef7c155a8185369fa86ba2b860154d777e46eba922324f/sqlalchemy-2.1.3-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:77a365f7cb81bbd738c948b64e0b683f8f1b68efe51ca2815e8ad4baf6287ed9", size = 4576491, upload-time = "2026-10-02T23:38:45.97Z" }, + { url = "https://files.pythonhosted.org/packages/23/0b/c560177600613df0cfe27e27c346bac956b631cd5236178b81144e783e9e/sqlalchemy-2.1.3-cp315-cp315-win32.whl", hash = "sha256:83757320391b96cde715af0772675941d3e345da9ced3696c030f7f55ffa8eab", size = 2378116, upload-time = "2026-10-02T23:14:57.441Z" }, + { url = "https://files.pythonhosted.org/packages/d3/72/d39f60d0bc873653c0fec76aa088c969a2394ea7737f18d812fbe0e8c0c5/sqlalchemy-2.1.3-cp315-cp315-win_amd64.whl", hash = "sha256:f87533295b84c9e19ca2a72dc26a76e0192f0a2424f8fd575a82e75e7d58f9e1", size = 2428865, upload-time = "2026-10-02T23:14:58.959Z" }, + { url = "https://files.pythonhosted.org/packages/c5/03/953cf93136de44ee9d6cebee14666dcc2c4a4285fe159b2e459326631e49/sqlalchemy-2.1.3-cp315-cp315-win_arm64.whl", hash = "sha256:3e46a27b0c74af74aa6c08c983ca51db88d22dd86c316d72a634e989b7caef58", size = 2393289, upload-time = "2026-10-02T23:08:31.958Z" }, + { url = "https://files.pythonhosted.org/packages/a2/1e/5b5af9c0fe4e6f713d2ab684b88f10fd1cbe55bd34cca4d676fb930c9da3/sqlalchemy-2.1.3-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:06ef75e3e0c72d774b41f659bfc4f63d5e207d6c8418abe2b80843a4a25e2fa6", size = 2496326, upload-time = "2026-10-03T02:48:37.363Z" }, + { url = "https://files.pythonhosted.org/packages/2b/6c/6192d30fe24edc7c9e98c315341b7342ba7168d1729867c5d87e56285658/sqlalchemy-2.1.3-cp315-cp315t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:70e05ae07dd9be29d4723bd0c880df622387eb51efb46a7b8efd49295248ffd7", size = 4880284, upload-time = "2026-10-03T02:49:10.221Z" }, + { url = "https://files.pythonhosted.org/packages/40/11/a758ae0116ab663f3520970cd870da5616cbb09c369b843239cce50696d7/sqlalchemy-2.1.3-cp315-cp315t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:10fdff11782842696172489bd4861ae63c212e62ee7c4fc61fdda90961abd1f3", size = 4830599, upload-time = "2026-10-03T02:54:03.353Z" }, + { url = "https://files.pythonhosted.org/packages/8a/e8/faa500255fd90f421ef9e918d522aa4f02212b53859c35dadbafb3532bb2/sqlalchemy-2.1.3-cp315-cp315t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:edd3e6690f350c89c616e78e9b58833c3d8925607ac5faf3530f38f053f58e4d", size = 4481272, upload-time = "2026-10-02T23:59:15.592Z" }, + { url = "https://files.pythonhosted.org/packages/3c/4d/26fc9adb3fbc1d5fb8b92043a1639fd111761b8c52a2ce61df8b00ff405d/sqlalchemy-2.1.3-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:e3c165df26eca3d2612f60ea91af02648cb1bd492eadfa64dba15c9a3a0fb09c", size = 4737553, upload-time = "2026-10-03T02:49:12.413Z" }, + { url = "https://files.pythonhosted.org/packages/08/41/3639fc1d9fd7ea201ed4941215c14b87ce3c3b5f5f0190eb93321d673e48/sqlalchemy-2.1.3-cp315-cp315t-musllinux_1_2_riscv64.whl", hash = "sha256:70c073d28d043766eb1725010792f302f59b14636ffef478793045471efe92dd", size = 4479761, upload-time = "2026-10-02T23:59:17.572Z" }, + { url = "https://files.pythonhosted.org/packages/47/c8/541035b19b34dd581fb24b6fcf065db6fb4f210b9fbf93e1b2bf5f8cd5c0/sqlalchemy-2.1.3-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:bf5f63229512f14e7bffc9a0041386318f592e76bbeaf9b750b39056094d91f3", size = 4757307, upload-time = "2026-10-03T02:54:05.524Z" }, + { url = "https://files.pythonhosted.org/packages/40/a3/bb51886c02a60f1b7fb5461a25d6a3e57ce0557b916032dc3c7a72fae226/sqlalchemy-2.1.3-cp315-cp315t-win32.whl", hash = "sha256:697942762c40e31eb0fa39bf0e58e1ac1b5b1ff9022da62b74f7cdc53096a482", size = 2433533, upload-time = "2026-10-03T02:56:52.298Z" }, + { url = "https://files.pythonhosted.org/packages/e5/8e/1739b378a2b40bfc5541355c9d944b96dffcbe86aa1ed3aff31ad13f5d9b/sqlalchemy-2.1.3-cp315-cp315t-win_amd64.whl", hash = "sha256:a9c33c0d081a9d2c94229b9ae1834cb2ef81dd81253df4763e32111287b64c31", size = 2499211, upload-time = "2026-10-03T02:56:54.801Z" }, + { url = "https://files.pythonhosted.org/packages/5d/9a/ce4ea58551258203381017bc4724369cc8948ac5a46212671266b13424c4/sqlalchemy-2.1.3-cp315-cp315t-win_arm64.whl", hash = "sha256:c36ff11196dacf50312a2cc51a798bb41611331c174871f065be651a838a5361", size = 2418086, upload-time = "2026-10-03T02:44:11.611Z" }, + { url = "https://files.pythonhosted.org/packages/e5/12/23d8fb7ed90e5b674006ace415ede1c60cdfc38ace50968b59ef42c22754/sqlalchemy-2.1.3-py3-none-any.whl", hash = "sha256:6f0cf4debd86cb5623a50310784bd005c215761ca7f618d0d4f46193d033675a", size = 2052603, upload-time = "2026-10-03T02:34:19.115Z" }, +] + [[package]] name = "stack-data" version = "0.6.3"