From 9ca3fbda643155858255c03fa691f0882c2b8ba6 Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Mon, 28 Sep 2026 10:30:24 -0700 Subject: [PATCH 1/8] Honor recipe resources and stop commands on cancellation --- CLAUDE.md | 50 ++- docs/api/compute.md | 48 ++- docs/api/index.md | 1 + docs/api/materialize.md | 15 +- docs/api/plan.md | 14 +- docs/api/sandbox.md | 17 +- docs/api/worker.md | 24 +- docs/architecture.md | 25 +- docs/cli/materialize.md | 18 +- docs/cli/run.md | 12 +- docs/user/cluster.md | 77 +++- src/lightcone/cli/commands.py | 25 +- src/lightcone/engine/compute/local.py | 36 +- src/lightcone/engine/compute/local_runtime.py | 34 +- src/lightcone/engine/compute/output.py | 7 +- .../engine/compute/slurm_bootstrap.py | 1 + src/lightcone/engine/execution.py | 227 +++++++++++ src/lightcone/engine/execution_resources.py | 155 +++++++ src/lightcone/engine/materialize.py | 220 +++++----- src/lightcone/engine/plan.py | 9 +- src/lightcone/engine/run.py | 15 +- src/lightcone/engine/sandbox/boundary.py | 71 ++-- src/lightcone/engine/sandbox/oci.py | 4 +- src/lightcone/engine/sandbox/processes.py | 339 ++++++++++++++++ src/lightcone/engine/worker.py | 44 +- tests/conftest.py | 10 +- tests/test_cli.py | 12 +- tests/test_compute_local.py | 5 +- tests/test_compute_output.py | 76 +++- tests/test_container_smoke.py | 45 +- tests/test_execution.py | 384 ++++++++++++++++++ tests/test_execution_processes.py | 229 +++++++++++ tests/test_execution_resources.py | 129 ++++++ tests/test_materialize.py | 191 ++++++++- tests/test_plan.py | 26 +- tests/test_sandbox_oci.py | 28 +- 36 files changed, 2330 insertions(+), 293 deletions(-) create mode 100644 src/lightcone/engine/execution.py create mode 100644 src/lightcone/engine/execution_resources.py create mode 100644 src/lightcone/engine/sandbox/processes.py create mode 100644 tests/test_execution.py create mode 100644 tests/test_execution_processes.py create mode 100644 tests/test_execution_resources.py diff --git a/CLAUDE.md b/CLAUDE.md index 961880e6..e54bc81b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -996,14 +996,16 @@ same run created, and whether it did would depend on whether a recipe finished before or after the previous save. Nondeterminism in a provenance field is worse than either answer. -**The worker never raises, and that is enforced at the unit boundary.** +**Ordinary recipe failures are results; execution-safety failures propagate.** It returns `ok`, `current`, `behind`, `failed`, or `blocked`. A task whose upstream did not report — failed, or never finished at all — returns `blocked` without running. Raising would make Dask re-raise in the driver and abort every task in flight, and reporting all independent failures in one run is most of what owning the loop buys. `worker.materialize` wraps the -whole unit, so the contract holds for failure modes nobody enumerated; -the one inner guard that remains exists because "your recipe failed" and +whole unit, so ordinary failures keep that contract. `ExecutionCancelled` and +`ExecutionUncertain` bypass it: reporting an uncertain writer as an ordinary +failure would let the driver restore files underneath it. The +one inner guard that remains exists because "your recipe failed" and "your recipe worked and we could not record it" deserve different words. **`data_version` is computed in the worker, before anything is staged.** @@ -1195,12 +1197,12 @@ crash, or a Ctrl-C would otherwise leave tracked files deleted or half-written — and the next run's refusal would tell the user to commit truncated, manifest-less garbage into `results/`, destroying the one property the layer exists for. So `ok` → `dataset.save`, and `failed` or -`blocked` → `dataset.restore`. The one exception is an interrupted run: a -task that never reported may still have a recipe writing on the cluster, so -its files are left in place rather than restored underneath it, and the -dirty-tree refusal's `results/` block says to stop the allocation before -discarding them. This is what makes the refusal survivable rather than a -trap. +`blocked` → `dataset.restore`. On interruption or driver failure, the invocation +first revokes admission and drains its claimed tasks. Restore unconsumed outputs +only with positive `Invocation.stopped` confirmation, independent of exception +type: a later context's error may mask uncertain cleanup. Otherwise leave files +in place and require native verification before repair. No cross-invocation +checkout lock is provided; run one execution invocation per project at a time. **The run record names declared paths, never resolved ones.** Every declared input under `data/` is an annex symlink, so a `Path.resolve()` @@ -1766,18 +1768,30 @@ Walltime follows Slurm's native overrun and termination-grace policy; Lightcone does not independently guarantee a finite termination deadline for Slurm jobs. **Execution borrows a client and leaves the allocation alive.** Validate native -identity and scheduler readiness. The driver keeps git and convergence. Use unique -invocation task keys. Interrupted unreported outputs remain in place because a -client disconnect does not prove remote subprocess termination. Comprehensive -cancellation/fencing and simultaneous writers are deferred by explicit user decision. -Local containerized processes can outlive process-group shutdown; do not claim -that `down` or walltime proves an external runtime's containers have stopped. +identity and scheduler readiness. The driver keeps git and convergence. Unique +invocation task keys plus claims and completion receipts in the existing Dask +scheduler prevent uncertain automatic recipe replay; missing state refuses work. +The driver renews authorization, then revokes and drains it before detaching. +Only positively confirmed cleanup permits restoring unconsumed outputs. A command +supervisor watches worker liveness through a pipe, drains the command group on +timeout/cancellation, and verifies native OCI container termination by immutable ID. +Recipes must not daemonize into other sessions. A hard-killed supervisor can leave +external containers alive; do not claim allocation termination proves otherwise. +Cross-invocation checkout locking remains deferred by explicit user decision. Read-only project validation precedes cluster connection, and a run with no tasks never connects: it only converges the crate. Populate the declared input-hash memo on the driver before serializing it to independent worker tasks. -Any driver failure while tasks are outstanding (a failed commit included, not -only a cluster error) carries `compute.UNSTOPPED`, the one wording for "the -allocation was not stopped and unreported tasks may still be running". +Any driver failure while tasks are outstanding (a failed commit included) first +drains the invocation. Unconfirmed cleanup raises `ExecutionUncertain` and retains +partial outputs; completing cleanup does not terminate the reusable allocation. + +**Recipe resources use standard Dask admission.** Preserve ASTRA `recipe.resources` +in `plan.Task` as validated `TaskResources`: whole CPUs, memory bytes, and optional +command walltime. Validate the whole selected graph before preparation or submission. +Workers advertise CPU/MEMORY; tasks reserve their declarations, with omitted RAM +reserving a whole worker's memory and probes reserving both whole-worker budgets. +Thread slots remain a separate concurrency cap. Reservations are cooperative, not +per-command OS CPU/RAM limits; unsupported GPU/disk requests fail explicitly. **One catalog selector, `LC_COMPUTE_CONFIG` (2026-09).** `lc compute --config` was removed: `run` and `materialize` resolve clusters through the catalog too, diff --git a/docs/api/compute.md b/docs/api/compute.md index c6fa9429..ea0dd865 100644 --- a/docs/api/compute.md +++ b/docs/api/compute.md @@ -113,15 +113,49 @@ Dask chooses the workers and handles dependencies; invocation-specific keys prev unintended reuse across commands. There is no worker-selection layer, per-worker preflight orchestration, source fingerprinting, or login-node guard. Driver-side preparation and the existing task runtime/sandbox checks remain in their owners. + +Workers advertise standard Dask `CPU` and `MEMORY` resources; memory is measured +in bytes. `engine.execution_resources.TaskResources` validates ASTRA's +`recipe.resources` into whole CPUs, bytes, and optional walltime seconds, and +`plan.Task` carries that request. `requirements(workers)` checks that one worker +can satisfy it and returns the resource dictionary used by `Client.submit`. +An omitted memory request reserves the full homogeneous worker budget; +`whole_worker=True` reserves CPU and memory for a probe. Unsupported GPU/disk +requests and fractional CPUs fail before execution. + +The materialize scheduler validates every selected task before preparation or +submission, preventing earlier tasks from starting before a later impossible +request is discovered. Standard Dask scheduling accounts for concurrent CPU and +memory reservations; Dask execution-thread counts remain a separate concurrency +cap. Reservations do not impose hard limits on recipe subprocesses. Task walltime +uses the subprocess boundary's timeout and teardown, independent of the native +allocation's lifetime. + `output.py` transports byte chunks through standard Dask events so detached workers' output reaches the invoking CLI. It uses the borrowed client's event topic, which the schedulers lc launches drop as soon as the client disconnects (`runtime.SCHEDULER_CONFIG`), rather than retaining a separate topic for every -command. A driver that exits before every task reports says so with -`UNSTOPPED`: closing a client cannot prove that a remote subprocess stopped. Probes preserve both streams; +command. Output-delivery errors cannot replace an execution-safety exception. +Probes preserve both streams; materialization sends recipe output to stderr to leave stdout for its report. -Local teardown drains the allocation's validated process group rather than +`engine.execution.invocation` owns a short renewable authorization in the existing +scheduler. Each task claims its logical key before touching files. Completion +receipts preserve the original result if Dask recomputes a lost result; a running +or uncertain claim refuses replay and revokes the invocation. Missing state also +refuses execution. This uses ordinary tasks and `run_on_scheduler`, without a +custom worker, service, project lock, or persistent execution registry. + +On exit the invocation revokes admission, cancels pending futures, and waits for +claimed tasks to acknowledge cleanup. Dask cancellation alone is insufficient: +running tasks poll authorization and the subprocess boundary stops their commands. +Only a positive `Invocation.stopped` flag permits restoring unconsumed outputs; +an exception from closing another context cannot manufacture that confirmation. +Scheduler loss or an unacknowledged attempt raises `ExecutionUncertain` and retains +outputs. Receipts are removed after confirmed cleanup; uncertain records remain +until the allocation ends. They are not a recovery log for a later invocation. + +Local teardown drains the allocation's validated process session rather than assuming the owner's exit proves every child stopped. Boot UUID, UID, process session and the exact command containing a random allocation token establish identity without depending on hostname or wall-clock creation time. Discovery @@ -131,9 +165,11 @@ are cleaned up, and incomplete locator directories do not hide healthy allocatio An allocation verified as ended, by `down` or by discovery, is retired: its TLS material, scheduler files and scratch are removed, and a marker lets discovery skip it unread. Its identity record stays, so a full ID still reports `ended`. -Cancellation and concurrent project writers are not made safe by allocation -management; callers must respect the documented execution limits. Containers -managed outside that process group can survive local teardown. +Concurrent invocations writing the same project remain unsupported. Command +cleanup covers process groups and native OCI container identities; recipes must +not daemonize into new sessions. Killing the command supervisor can leave an +external runtime's container alive, so an uncertain execution requires native +verification before output repair. Tests cover deterministic selection, malformed identities and catalogs, partial native failures, acceptance ambiguity, PID reuse, detached local lifetime, standard diff --git a/docs/api/index.md b/docs/api/index.md index 8674ea0b..2702c160 100644 --- a/docs/api/index.md +++ b/docs/api/index.md @@ -18,6 +18,7 @@ is responsibility and contract, not every signature. | [`worker`](worker.md) | Making one output; the rerun entry point | impure | | [`materialize`](materialize.md) | The driver: gates, scheduling, the save/restore loop, status | impure | | [`compute`](compute.md) | Resource requests, native allocation lifecycle, borrowed Dask clients | impure | +| [`execution` & `execution_resources`](compute.md) | Invocation claims, completion receipts, cleanup confirmation, task resource admission | mixed | | [`sandbox`](sandbox.md) | The exec boundary: policy, backends, attestation, denials | mixed | | [`image` & `container`](container.md) | The container hatch: declaration → image → archive → runtime | pure / impure | | [`crate`](crate.md) | The publication view: the repo as an RO-Crate | pure | diff --git a/docs/api/materialize.md b/docs/api/materialize.md index 39450bcb..47fe87a9 100644 --- a/docs/api/materialize.md +++ b/docs/api/materialize.md @@ -18,7 +18,7 @@ driver's stderr, independently of success or failure, leaving stdout for the rep | `check(root, targets, *, refresh)` | The same classification without executing, committing, or fetching. Exempt from the dirty refusal. | | `status(root)` | The report: every output's state and provenance commit, plus the mode/image/sandbox header facts. | | `MaterializeReport` / `StatusReport` | The JSON surfaces; `ok` and `up_to_date` first. | -| `cluster_for_run(cluster_id)` | Borrow the cluster; the submit/completed scheduler seam (`submit`, `completed`). | +| `cluster_for_run(cluster_id)` | Borrow the cluster; expose resource validation, submission, completion, and positive cleanup confirmation. | | `run_record(...)` / `datalad_run_subject(...)` | The commit message `datalad rerun` replays, and the one spelling of its subject line — shared with the foreign-write comparator, because two strings here would drift. | | `_engine_requirement()` | How a record pins its engine: by version for a release, by source commit (hatch-vcs) for a dev build. | @@ -27,7 +27,8 @@ driver's stderr, independently of success or failure, leaving stdout for the rep 1. **Read-only project checks before connecting** — tool, committer, dirty-tree, spec and lock errors do not require a reachable cluster to report. 2. **Explicit cluster before preparing the environment** — validate native - allocation identity and connect before fetching inputs or building an image. + allocation identity, connect, and validate every selected task's CPU/memory + request before fetching inputs or building an image. The dirty refusal has already run: in containerized mode the converge can commit an image archive, and `dataset.save` commits the whole index; on a dirty tree the user's @@ -44,9 +45,11 @@ driver's stderr, independently of success or failure, leaving stdout for the rep worse than either answer. The populated input-hash memo travels with each task; independent worker processes do not rehash shared inputs. Unreadable inputs still fail only the tasks that need them. -6. **Save on `ok`, restore reported failures** — unreported outputs are retained - after interruption because their tasks may still be writing. Allocation - management does not provide concurrent-writer or cancellation guarantees. +6. **Save on `ok`, restore reported failures** — on interruption or a driver + error, first revoke and drain the invocation. Restore submitted, unconsumed + outputs only when the scheduler seam reports positive cleanup confirmation. + Otherwise retain them: an exception is not evidence that a writer stopped. + Separate invocations must still not write the same project concurrently. ## What must stay true @@ -74,6 +77,6 @@ driver's stderr, independently of success or failure, leaving stdout for the rep ## Tests `tests/test_materialize.py` — real repositories, real recipes, a real -`LocalCluster` through the seam exactly once, real `datalad rerun` for +`LocalCluster` for scheduling and lifecycle checks, real `datalad rerun` for the record's whole claim. `cluster_for_run` is the one monkeypatch point for allocation-free tests. diff --git a/docs/api/plan.md b/docs/api/plan.md index 93cf8e35..9a67117a 100644 --- a/docs/api/plan.md +++ b/docs/api/plan.md @@ -3,7 +3,8 @@ The spec, read as a graph of tasks. `astra.yaml` × `universes/*.yaml` gives one task per `(universe, output)` pair that has a recipe; a task carries everything executing it needs — the rendered command, where its -bytes go, what it reads, its decisions, its `definition_version` — and +bytes go, what it reads, its decisions, its `definition_version`, and its +resource requirements — and nothing about *how* it will be executed. Source: `src/lightcone/engine/plan.py`. @@ -14,7 +15,8 @@ Source: `src/lightcone/engine/plan.py`. |---|---| | `build(root)` | Validate the spec with ASTRA's own validators, resolve every universe, return the `Graph`. | | `Graph` | Tasks keyed on `(universe_id, output_id)`; `order()` for the read-only topological walk, `resolve(targets)` for what a user typed, `closure(keys)` to narrow a run. | -| `Task` | One output in one universe, frozen. | +| `Task` | One output in one universe, frozen, including its parsed `resources`. | +| `TaskResources` | Frozen Pydantic request from `engine.execution_resources`: positive whole CPUs, optional memory bytes, and optional walltime seconds. | | `declared_path(root, path)` | The one rule that names a path: project-relative inside the tree, absolute outside, never resolved. | ## What must stay true @@ -31,6 +33,12 @@ Source: `src/lightcone/engine/plan.py`. schema, file, and universe validators before resolving anything — resolution answers what a *valid* spec means and does not re-check that it is one. +- **Resource declarations survive resolution.** `build` reads + `recipe.resources` from ASTRA's resolved output definition and validates + units and supported requirements through `TaskResources.parse`. It rejects + unsupported GPU/disk requests and fractional CPUs with the output's name. + Cluster capacity is checked later, before materialize prepares the project + or submits any task. No worker placement belongs in this module. - **The layout is flat and path-addressed.** `results//.`, and the path in a rendered recipe *is* the path on disk — no staging, no relocation. @@ -50,7 +58,7 @@ Source: `src/lightcone/engine/plan.py`. ## Tests `tests/test_plan.py` — pure; tests what lc *adds* (directories, edges, -versions, the validation gate), never what a spec means — that +versions, resource requirements, the validation gate), never what a spec means — that coverage lives in astra-tools' own suite, and re-asserting it here would recreate the second implementation this module deleted. Every fixture must be a spec `astra validate` accepts; the gate enforces it diff --git a/docs/api/sandbox.md b/docs/api/sandbox.md index afd87882..40d5d2e9 100644 --- a/docs/api/sandbox.md +++ b/docs/api/sandbox.md @@ -7,7 +7,7 @@ one, runs it, and reports what was actually enforced. `run.py` (the `lc run` engine) and the worker are the two consumers. Source: `src/lightcone/engine/sandbox/` — `model.py`, `policy.py`, -`boundary.py`, `landlock.py`, `seatbelt.py`, `oci.py`, `denial.py` — +`boundary.py`, `processes.py`, `landlock.py`, `seatbelt.py`, `oci.py`, `denial.py` — plus `lightcone/_sandbox_exec.py`, the Landlock shim. ## Key symbols @@ -26,6 +26,21 @@ An optional output receiver gets stdout/stderr byte chunks. Capturing output nev decodes or normalizes stdout; only the retained stderr tail is decoded for denial classification. Without a receiver, stdout remains inherited. +`processes.Command` starts a small supervisor outside the sandbox. The supervisor +owns the wrapped command's process group and watches a control pipe: worker death +closes the pipe and triggers cleanup even when the worker cannot run `finally`. +Timeouts and cancellation use the same TERM/KILL cleanup. A command that exits +while leaving background processes is failed after those processes are stopped. +The unreaped leader pins the process-group ID until cleanup completes. + +OCI commands use a private `--cidfile`. Cleanup inspects that immutable container +ID, stops or kills a running container, verifies it stopped, then removes it. +An absent or unverifiable identity during interruption is uncertain, not success. +Only confirmed cleanup allows an ordinary result or `ExecutionCancelled`; +unconfirmed cleanup raises `ExecutionUncertain` and retains the temporary home. +Recipes must not detach into other process sessions. A hard-killed supervisor +cannot guarantee cleanup of containers managed by an external runtime. + ## What must stay true - **`wrap` stays pure** — no temp files, no FDs, no global state diff --git a/docs/api/worker.md b/docs/api/worker.md index 3d6e255d..5e943f04 100644 --- a/docs/api/worker.md +++ b/docs/api/worker.md @@ -18,23 +18,33 @@ Source: `src/lightcone/engine/worker.py`. Cluster execution supplies an output receiver to `materialize`/`execute`, which passes byte chunks from the sandbox back to the invocation. Standalone reruns -retain direct terminal output. +retain direct terminal output. The driver submits each cluster task with its +CPU and memory reservations; `execute` applies the task's walltime limit through +the sandbox boundary. Standalone reruns also apply that time limit, but do not +perform Dask resource admission. ## Key symbols | Symbol | Role | |---|---| -| `materialize(task, versions, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns a `TaskResult`, always. | -| `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, and the attestation. `.usable` is what dependents check. | +| `materialize(root, task, context, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns ordinary failures as `TaskResult`; propagates execution safety exceptions. | +| `execute(root, task, input_versions, context)` | Run a recipe unconditionally with its time limit and cancellation checks, then record its payload and manifest. | +| `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, reason, and diagnostic notes. `.usable` is what dependents check. | | `main(argv)` | The rerun entry point: guards, converges the project environment from the commit's own lock, resolves its own HEAD and runtime, executes. | | `lc_version()` | The engine version every manifest records. | ## What must stay true -- **The worker never raises** — enforced at the unit boundary, so the - contract holds for failure modes nobody enumerated. Raising would - make Dask abort every task in flight; reporting all independent - failures in one run is most of what owning the loop buys. +- **Ordinary recipe failures are results.** Independent tasks continue so + the driver can report all their failures. `ExecutionCancelled` and + `ExecutionUncertain` instead propagate and abort the invocation. They must + not enter the ordinary failed-output restore path: cleanup first establishes + that writers have stopped, and uncertainty retains partial outputs. +- **Task completion includes subprocess teardown.** The boundary owns process + and container cleanup, applies `task.resources.time_seconds`, and reports + uncertain teardown as an exception. A time limit that stops the recipe + becomes an ordinary failed result. CPU and memory reservations are standard + Dask scheduling constraints, not OS limits imposed by this module. - **`data_version` is computed here, before anything is staged** — the dependent's argument *is* this return value, so the digest must exist while the files are still unannexed. Deriving it from diff --git a/docs/architecture.md b/docs/architecture.md index 5e55fb62..66b2f411 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -41,10 +41,11 @@ lc materialize "$CLUSTER" │ plan: astra validate + resolve → Graph of Tasks │ (no tasks → converge the crate and stop; nothing connects) │ connect: native identity + Dask readiness + │ admit: each task's CPU and memory request fits a worker │ fetch: git annex get (declared inputs not in this clone) │ converge: uv.lock ⇄ .venv (and the image, containerized) ├─► workers: reset output file → sandbox → recipe → hash → manifest - │ (never raise; return ok/current/behind/failed/blocked) + │ (ordinary failures return failed/blocked) └─ driver: consume results in one thread ok → dataset.save (commit + run record) failed → dataset.restore (tree as clean as it started) @@ -60,10 +61,15 @@ The division of labor is strict and load-bearing: - **Dask owns the ordering.** Every task is submitted with its upstream futures as arguments; there is no ready-set loop or hand-rolled topological sort on the execution path. -- **The worker never raises.** A recipe failure, a gate failure, an - unreadable manifest — all come back as a state, so one failure - doesn't abort every task in flight, and a run reports *all* its - independent failures. +- **Ordinary failures are task results.** A recipe failure, a gate failure, + or an unreadable manifest returns a state so independent tasks can continue. + Cancellation and uncertain execution propagate instead, aborting the + invocation. Partial outputs are restored only after writers are confirmed + stopped; uncertainty retains those files for inspection. +- **Dask accounts for task resources.** Workers advertise CPU and memory + budgets; submissions reserve the recipe's requirements. The subprocess + boundary applies time limits and owns process teardown. CPU and memory + reservations coordinate scheduling rather than imposing per-recipe OS limits. - **Values are resolved once and handed down.** HEAD, the container runtime, and the foreign-write facts are read by the driver and passed to workers as values — a worker that asked git itself could @@ -166,9 +172,12 @@ material, not a registry. `compute.connect(CLUSTER_ID)` borrows a standard Dask client and closes only that client on exit. Both execution commands require a cluster ID. The materialization -scheduler keeps its `submit`/`completed` seam. Driver preparation and existing -task runtime/sandbox checks remain unchanged. Tasks use ordinary Dask scheduling; -there is no separate worker-selection or preflight layer, or site-marker guard. No execution command implicitly allocates compute. +scheduler validates resource requests, then keeps its `submit`/`completed` seam; +ordinary Dask scheduling places the tasks. Driver preparation and task runtime +checks remain in their existing owners. `engine.execution` holds invocation claims +and receipts in the scheduler, revokes admission on cancellation, and waits for +command cleanup before allowing output restoration. No execution command +implicitly allocates compute, and there is no separate execution service. See [compute internals](api/compute.md) and [deployment limits](user/cluster.md). ## The publication view diff --git a/docs/cli/materialize.md b/docs/cli/materialize.md index e75a72b5..9e78a049 100644 --- a/docs/cli/materialize.md +++ b/docs/cli/materialize.md @@ -48,14 +48,16 @@ never touched, under any flag. ## The run's contract - **Starts clean.** A dirty tree is a refusal. A recipe that returns a - failure has its partial work restored. After a cluster interruption, - unreported outputs are retained because tasks may still be running. The - same holds when a commit fails while other recipes are still running: the - error says so. - Stop the allocation with `lc compute down` and its full ID (a name can - already belong to a newer allocation), and confirm its recipes have - stopped before cleaning results. Local containers may need separate - termination through their runtime; see [execution limits](../user/cluster.md#execution-requirements-and-limits). + failure has its partial work restored. On interruption or a driver failure, + lc cancels outstanding tasks and restores their uncommitted outputs only after + cleanup is positively confirmed. Completed commits remain. If cleanup is + uncertain, outputs stay in place: stop the allocation by its full ID and verify + its commands and containers have stopped before repairing results. See + [execution limits](../user/cluster.md#execution-requirements-and-limits). +- **Honors recipe resources.** CPU and memory requests must fit one worker and + are reserved through standard Dask scheduling; time limits stop overrunning + commands. The whole selected graph is checked before preparation or submission. + See [recipe resource requirements](../user/cluster.md#recipe-resource-requirements). - **Fetches what it needs.** Declared inputs whose annexed content is not in this clone are fetched before anything hashes. - **Commits as it goes.** Each output lands in its own commit, written diff --git a/docs/cli/run.md b/docs/cli/run.md index 1e21c53e..63d9bd3a 100644 --- a/docs/cli/run.md +++ b/docs/cli/run.md @@ -40,12 +40,12 @@ that variable to an existing cluster. Set command-specific values inside the command, for example `lc run "$CLUSTER" -- env NAME=value python script.py`. Containerized commands use the image's environment and the sandbox overlays. -Interrupting the CLI detaches its client; the remote command may still be running. -Stop the allocation with `lc compute down` and its full ID (a name can already -belong to a newer allocation) before working with files the interrupted command -could still be writing. Confirm that the command has -stopped; local containers may require separate termination through their runtime -(see [execution limits](../user/cluster.md#execution-requirements-and-limits)). +The command reserves one worker's full CPU and memory budgets for its duration. +Interrupting the CLI requests cancellation and waits for command cleanup; the +cluster remains available. If cleanup cannot be confirmed, the error says so. +Stop the allocation using `lc compute down` with its full ID (names can be reused) +and verify its commands and containers have stopped before repairing outputs. +See [execution limits](../user/cluster.md#execution-requirements-and-limits). ## What it does diff --git a/docs/user/cluster.md b/docs/user/cluster.md index 0250908d..aac69d9e 100644 --- a/docs/user/cluster.md +++ b/docs/user/cluster.md @@ -44,9 +44,10 @@ Local resources are cooperative limits, not an exclusive CPU/RAM reservation. An allocation owns a detached process session and standard `LocalCluster`: one worker process with `task_slots_per_node` threads, and a scheduler that listens on `127.0.0.1` over TLS. Its own logs are discarded; a startup failure is kept -and shown as the reason by `lc compute status`. At its time limit the whole -process session is killed with SIGKILL, so a recipe still running stops mid-write. -`down` sends SIGTERM, waits three seconds, then sends SIGKILL. +and shown as the reason by `lc compute status`. At its time limit, or on `down`, +the allocation stops its process session with SIGTERM, then SIGKILL if needed. +Shutdown normally allows three seconds; active command supervisors get up to +sixteen seconds to stop their commands and containers before escalation. Private process locators are checked against the native boot UUID, UID, process session, and exact command containing the allocation's random token before attachment or termination. Hostname changes and clock adjustments do not change @@ -308,6 +309,46 @@ has as long to start. A failed check or timeout logs `Slurm Dask startup failed: …` to the submission log and exits nonzero. Look there when a job is active but never becomes ready. +## Recipe resource requirements + +Declare each recipe's needs in `astra.yaml`: + +```yaml +recipe: + command: python src/fit.py {output} + resources: + cpus: 4 + memory: 8Gi + time_limit: 1h30m +``` + +Each recipe runs on one worker. Its CPU and memory request must fit that +worker, even when the cluster has several nodes. Dask reserves both budgets +while the task runs, so recipes can run together only when their combined +requests fit. `task_slots_per_node` also caps concurrent tasks; it does not +limit how many CPUs a single recipe may request. + +CPUs must be positive whole numbers and default to one. Memory needs units: +`512Mi` and `8Gi` are binary sizes; `8GB` is decimal. Without a memory +declaration, a recipe reserves the worker's entire memory budget, so only +one such recipe runs per worker. `lc run` reserves an entire worker's CPU and +memory budgets because its arbitrary command has no recipe declaration. +Time limits accept combinations such as `1h30m` or `45s`; exceeding the limit +stops the recipe and reports failure. Fractional CPUs, GPUs, and disk requests +are rejected rather than ignored. + +`lc materialize` validates the complete selected graph against the cluster +before fetching inputs, preparing the environment, or starting a recipe. This +also validates currently complete outputs, which workers may need to rebuild +after an upstream change. Use `lc materialize --check` to inspect currency +without allocation. + +These are scheduling reservations, not per-recipe CPU or RAM enforcement. +Recipes must respect their declarations; a subprocess can otherwise exceed +its request. Slurm enforces the overall allocation, while local execution +uses cooperative budgets. Leave capacity for the scheduler, workers, and other +overhead when declaring recipe requirements. + ## Execution requirements and limits Driver and workers must see the same project, prepared environment, and inputs @@ -332,13 +373,23 @@ directories and credential files still reject symlinks, retain ownership and ancestor-permission checks, and require modes `0700` and `0600`, respectively. The catalog's location is independent of the private connection files. -Use one execution invocation per project at a time. Concurrent writers, -comprehensive cancellation, task fencing, and recovery after client/worker loss -are not guaranteed. A lost client does not prove its subprocesses stopped. -Unreported partial outputs are retained after interruption rather than restored -while a task may still write them. End the allocation and establish that work has -stopped before inspecting or repairing that project's outputs. -For local containerized execution, `down` and walltime expiry stop the managed -process group but do not guarantee termination of containers managed by an -external runtime. A Podman container that ignores SIGTERM can survive. Inspect -and stop such containers through the container runtime before cleaning results. +Ctrl-C revokes the invocation and waits for its commands to stop. Materialize +keeps completed commits and restores uncommitted outputs only after cleanup is +confirmed. If cleanup cannot be confirmed, partial outputs stay in place and the +error asks you to stop the allocation and verify its commands and containers +before retrying. The allocation remains available after ordinary cancellation. + +The existing Dask scheduler holds invocation claims and completion receipts. +After a worker disappears, a replacement task cannot rerun a recipe whose result +is uncertain. A completed task returns its original receipt. Loss of the client +or its heartbeat revokes further work; a small command supervisor also stops the +command if its worker dies. There is no automatic recovery or replay after an +uncertain execution, and no additional server or checkout state directory. + +Use one execution invocation per project at a time: there is no checkout lock +across invocations. Recipes must finish all their work before returning and must +not detach daemon processes into new sessions. Cleanup covers each command's +process group and its OCI container, whose immutable runtime ID is checked. +If a supervisor itself is forcibly killed, a container managed by an external +runtime can survive; native allocation termination alone cannot prove it stopped. +Inspect and stop such containers before repairing outputs. diff --git a/src/lightcone/cli/commands.py b/src/lightcone/cli/commands.py index e9346c34..2d959d3d 100644 --- a/src/lightcone/cli/commands.py +++ b/src/lightcone/cli/commands.py @@ -188,8 +188,13 @@ def run(cluster_id: str, command: tuple[str, ...]) -> None: raise click.UsageError("A command is required; no interactive shell is opened.") try: outcome = engine_run.probe(current_project(), command, cluster_id=cluster_id) - except KeyboardInterrupt: - _interrupted(cluster_id, "the remote command may still be running") + except KeyboardInterrupt as exc: + if getattr(exc, "execution_stopped", False): + click.echo( + "Interrupted; the command has stopped. The cluster remains available.", err=True, + ) + else: + _interrupted(cluster_id, "the remote command may still be running") raise if outcome.notes: click.echo("\n".join(["", *outcome.notes]), err=True) @@ -377,11 +382,17 @@ def materialize( else: try: report = engine.materialize(root, targets, cluster_id=cluster_id, refresh=refresh) - except KeyboardInterrupt: - _interrupted( - cluster_id, "remote recipes may still be writing results", - " and confirm they have stopped before cleaning results/", - ) + except KeyboardInterrupt as exc: + if getattr(exc, "execution_stopped", False): + click.echo( + "Interrupted; recipes have stopped and uncommitted outputs were restored. " + "The cluster remains available.", err=True, + ) + else: + _interrupted( + cluster_id, "remote recipes may still be writing results", + " and confirm they have stopped before cleaning results/", + ) raise if as_json: diff --git a/src/lightcone/engine/compute/local.py b/src/lightcone/engine/compute/local.py index 0474d146..7ae1d3b6 100644 --- a/src/lightcone/engine/compute/local.py +++ b/src/lightcone/engine/compute/local.py @@ -382,7 +382,7 @@ def connect(self, identity: Identity, *, timeout: float = 10) -> Iterator[Any]: client.close(timeout=min(timeout, 5)) def terminate(self, identity: Identity) -> None: - """Terminate the validated allocation process group even if Dask is wedged.""" + """Terminate the validated allocation session even if Dask is wedged.""" directory, record = self._record(identity) self._stop(identity, directory, record) self._retire(directory) @@ -396,7 +396,6 @@ def _stop(self, identity: Identity, directory: Path, record: dict[str, Any]) -> try: if ( member.uids().real == os.getuid() - and os.getpgid(member.pid) == process.pid and os.getsid(member.pid) == process.pid ): # Capture each birth identity while the owner still proves @@ -405,20 +404,16 @@ def _stop(self, identity: Identity, directory: Path, record: dict[str, Any]) -> members.append(member) except (psutil.NoSuchProcess, psutil.AccessDenied, ProcessLookupError): continue - process = self._process(identity, directory, record) - if process is not None: + for member in members: try: - os.killpg(process.pid, signal.SIGTERM) - except ProcessLookupError: + member.terminate() + except psutil.NoSuchProcess: pass - else: - for member in members: - try: - member.terminate() - except psutil.NoSuchProcess: - pass for escalation in (False, True): - deadline = time.monotonic() + _STOP_GRACE + from lightcone.engine.sandbox.processes import has_custodian + + grace = max(_STOP_GRACE, 16) if has_custodian(members) else _STOP_GRACE + deadline = time.monotonic() + grace while members and time.monotonic() < deadline: living = [] for member in members: @@ -439,12 +434,15 @@ def _stop(self, identity: Identity, directory: Path, record: dict[str, Any]) -> ) process = self._process(identity, directory, record) if process is not None: - try: - # A live, verified owner also covers children created while - # stopping. Its finalizer provides the same group-wide kill. - os.killpg(process.pid, signal.SIGKILL) - except ProcessLookupError: - pass + # Commands have separate groups inside this session. The + # live owner establishes custody of newly created members. + from lightcone.engine.sandbox.processes import members as session_members + + for member in session_members(session=process.pid): + try: + member.kill() + except psutil.NoSuchProcess: + pass for member in members: try: # The owner may have exited first. Never signal its old PGID diff --git a/src/lightcone/engine/compute/local_runtime.py b/src/lightcone/engine/compute/local_runtime.py index dbb360e6..202b2e8b 100644 --- a/src/lightcone/engine/compute/local_runtime.py +++ b/src/lightcone/engine/compute/local_runtime.py @@ -20,6 +20,34 @@ ) +def _stop_session() -> None: + """Give command custodians time to drain, then kill the allocation session.""" + import psutil + + from lightcone.engine.sandbox.processes import has_custodian, members + + owner = os.getpid() + remaining = [member for member in members(session=owner) if member.pid != owner] + for member in remaining: + try: + member.terminate() + except psutil.NoSuchProcess: + pass + started = time.monotonic() + while remaining: + grace = 15 if has_custodian(remaining) else 2.5 + if time.monotonic() - started >= grace: + break + time.sleep(0.05) + remaining = [member for member in members(session=owner) if member.pid != owner] + for member in remaining: + try: + member.kill() + except psutil.NoSuchProcess: + pass + os.kill(owner, signal.SIGKILL) + + def main() -> None: """Run the detached allocation owner until shutdown or its walltime expires.""" os.umask(0o077) @@ -34,12 +62,12 @@ def stop(_signum: int, _frame: FrameType | None) -> None: def expire(_signum: int, _frame: FrameType | None) -> None: # This bound does not depend on the scheduler loop or graceful Dask close. - os.killpg(os.getpgrp(), signal.SIGKILL) + _stop_session() # Register before importing Dask so its multiprocessing finalizers run first. # A recipe can ignore SIGTERM and outlive its worker: keep custody of the # session and walltime timer until every member has been sent SIGKILL. - atexit.register(os.killpg, os.getpgrp(), signal.SIGKILL) + atexit.register(_stop_session) signal.signal(signal.SIGTERM, stop) signal.signal(signal.SIGINT, stop) signal.signal(signal.SIGALRM, expire) @@ -54,6 +82,7 @@ def expire(_signum: int, _frame: FrameType | None) -> None: from distributed import LocalCluster security = create_security(directory) + allocation = read_private_json(directory / "identity.json") with dask.config.set(SCHEDULER_CONFIG), LocalCluster( # type: ignore[no-untyped-call] n_workers=1, threads_per_worker=int(launch["task_slots"]), @@ -73,6 +102,7 @@ def expire(_signum: int, _frame: FrameType | None) -> None: # Recipes use subprocesses: Dask's Python-process RSS cannot enforce # their RAM envelope. Local resource limits are explicitly cooperative. memory_limit=0, + resources={"CPU": int(allocation["cpus"]), "MEMORY": int(allocation["memory"])}, silence_logs=50, ) as cluster: write_private_json( diff --git a/src/lightcone/engine/compute/output.py b/src/lightcone/engine/compute/output.py index c92bbcb9..3fadd122 100644 --- a/src/lightcone/engine/compute/output.py +++ b/src/lightcone/engine/compute/output.py @@ -65,4 +65,9 @@ def output(stream: str, data: bytes) -> None: try: return function(*args, output=output) finally: - worker.log_event(topic, {"done": task}) + # Output delivery must not mask an execution-safety exception. A lost + # final marker is reported by the driver's bounded output wait. + try: + worker.log_event(topic, {"done": task}) + except Exception: + pass diff --git a/src/lightcone/engine/compute/slurm_bootstrap.py b/src/lightcone/engine/compute/slurm_bootstrap.py index 6e35cdf0..e184d728 100644 --- a/src/lightcone/engine/compute/slurm_bootstrap.py +++ b/src/lightcone/engine/compute/slurm_bootstrap.py @@ -101,6 +101,7 @@ async def run(args: argparse.Namespace) -> None: **address, "nthreads": args.task_slots, "memory_limit": 0, + "resources": {"CPU": args.cpus, "MEMORY": args.memory_bytes}, "local_directory": str(scratch), "dashboard_address": "127.0.0.1:0", "dashboard": False, diff --git a/src/lightcone/engine/execution.py b/src/lightcone/engine/execution.py new file mode 100644 index 00000000..728e49e2 --- /dev/null +++ b/src/lightcone/engine/execution.py @@ -0,0 +1,227 @@ +"""Bound one invocation's ordinary Dask tasks to their command lifetimes. + +The existing scheduler keeps small claims and completion receipts. Tasks never +create missing invocations: losing the scheduler therefore fails closed, rather +than starting a recipe again. This is not a lock on a project checkout. +""" + +from __future__ import annotations + +import threading +import time +from collections.abc import Callable, Iterator +from contextlib import contextmanager +from contextvars import ContextVar +from dataclasses import dataclass, field +from typing import Any +from uuid import uuid4 + +from lightcone.engine.project import ProjectError + +_HEARTBEAT = 2.0 +_LEASE = 15.0 +_RPC_TIMEOUT = 5.0 +_STOP_TIMEOUT = 40.0 +_CANCELLED: ContextVar[Callable[[], bool]] = ContextVar( + "execution_cancelled", default=lambda: False, +) + + +class ExecutionUncertain(ProjectError): # noqa: N818 + """Execution may still own writers; its partial outputs must be retained.""" + + +class ExecutionCancelled(ProjectError): # noqa: N818 + """Execution was revoked and its command has stopped.""" + + +def cancelled() -> bool: + """Check the current task's authorization without blocking its subprocess loop.""" + return _CANCELLED.get()() + + +def check_cancelled() -> None: + """Refuse further task mutations once cancellation has been observed.""" + if cancelled(): + raise ExecutionCancelled("execution was cancelled") + + +def _state( + invocation: str, operation: str, task: str = "", value: Any = None, + *, dask_scheduler: Any, +) -> Any: + # Scheduler callbacks run serially on its event loop. Registration is a + # driver-only operation, never retried implicitly by a task or heartbeat. + records = dask_scheduler.extensions.setdefault("lightcone-executions", {}) + now = time.monotonic() + if operation == "register": + if invocation in records: + raise ExecutionUncertain("execution is already registered") + records[invocation] = {"client": value, "deadline": now + _LEASE, + "active": True, "tasks": {}} + return None + record = records.get(invocation) + if record is None: + raise ExecutionUncertain("execution is no longer registered; refusing task replay") + if now > record["deadline"] or record["client"] not in dask_scheduler.clients: + record["active"] = False + if operation == "heartbeat": + if record["active"]: + record["deadline"] = now + _LEASE + return record["active"] + if operation == "active": + return record["active"] + if operation == "claim": + if not record["active"]: + raise ExecutionCancelled("execution is no longer active") + previous = record["tasks"].get(task) + if previous is not None: + if previous["state"] == "finished": + return False, previous["result"] + record["active"] = False + raise ExecutionUncertain(f"{task}: a previous attempt has no confirmed result") + record["tasks"][task] = {"state": "running", "attempt": value} + return True, None + if operation in {"finished", "stopped", "uncertain"}: + attempt, result = value + if record["tasks"].get(task, {}).get("attempt") != attempt: + raise ExecutionUncertain(f"{task}: completion belongs to another attempt") + record["tasks"][task] = {"state": operation, "attempt": attempt, "result": result} + if operation == "uncertain": + record["active"] = False + return None + if operation == "revoke": + record["active"] = False + if operation in {"revoke", "pending"}: + return [name for name, item in record["tasks"].items() + if item["state"] in {"running", "uncertain"}] + if operation == "forget": + del records[invocation] + return None + raise ValueError(f"unknown execution operation: {operation}") + + +async def _request(client: Any, invocation: str, operation: str, task: str, value: Any) -> Any: + return await client.run_on_scheduler(_state, invocation, operation, task, value) + + +def _rpc(client: Any, invocation: str, operation: str, task: str = "", value: Any = None) -> Any: + return client.sync( + _request, client, invocation, operation, task, value, callback_timeout=_RPC_TIMEOUT, + ) + + +def _call(invocation: str, task: str, function: Callable[..., Any], *args: Any) -> Any: + from distributed import get_client + + client = get_client() + # A lost claim reply is ambiguous. Do not execute unless it was received. + attempt = uuid4().hex + claimed, result = _rpc(client, invocation, "claim", task, attempt) + if not claimed: + return result + stopped = threading.Event() + revoked = threading.Event() + + def monitor() -> None: + while not stopped.wait(_HEARTBEAT): + try: + active = _rpc(client, invocation, "active") + except Exception: + active = False + if not active: + revoked.set() + return + + watcher = threading.Thread(target=monitor, daemon=True) + watcher.start() + token = _CANCELLED.set(revoked.is_set) + try: + result = function(*args) + check_cancelled() + except BaseException as exc: + state = "uncertain" if isinstance(exc, ExecutionUncertain) else "stopped" + try: + _rpc(client, invocation, state, task, (attempt, None)) + except Exception: + pass + raise + else: + try: + _rpc(client, invocation, "finished", task, (attempt, result)) + except Exception as exc: + raise ExecutionUncertain( + f"{task}: could not record completion; outputs retained, refusing replay" + ) from exc + return result + finally: + _CANCELLED.reset(token) + stopped.set() + + +@dataclass +class Invocation: + """Submit tasks whose side effects must not be replayed automatically.""" + + client: Any + id: str = field(default_factory=lambda: uuid4().hex) + futures: list[Any] = field(default_factory=list) + stopped: bool = False + + def submit( + self, function: Callable[..., Any], *args: Any, key: str, resources: dict[str, float], + ) -> Any: + """Claim each logical task inside its worker before it can mutate files.""" + future = self.client.submit( + _call, self.id, key, function, *args, + key=f"lc-{self.id}-{key}", pure=False, retries=0, resources=resources, + ) + self.futures.append(future) + return future + + +@contextmanager +def invocation(client: Any) -> Iterator[Invocation]: + """Own authorization and wait for running commands to stop before detaching.""" + run = Invocation(client) + _rpc(client, run.id, "register", value=client.id) + stopped = threading.Event() + + def heartbeat() -> None: + while not stopped.wait(_HEARTBEAT): + try: + if not _rpc(client, run.id, "heartbeat"): + return + except Exception: + return + + threading.Thread(target=heartbeat, daemon=True).start() + interruption: KeyboardInterrupt | None = None + try: + yield run + except KeyboardInterrupt as exc: + interruption = exc + raise + finally: + stopped.set() + try: + pending = _rpc(client, run.id, "revoke") + unfinished = [future for future in run.futures if not future.done()] + if unfinished: + client.sync(client.cancel, unfinished, callback_timeout=_RPC_TIMEOUT) + deadline = time.monotonic() + _STOP_TIMEOUT + while pending and time.monotonic() < deadline: + time.sleep(0.1) + pending = _rpc(client, run.id, "pending") + if pending: + raise ExecutionUncertain("unconfirmed tasks: " + ", ".join(pending)) + # A late dispatch cannot recreate this record: missing is a refusal. + _rpc(client, run.id, "forget") + run.stopped = True + if interruption is not None: + interruption.execution_stopped = True # type: ignore[attr-defined] + except Exception as exc: + raise ExecutionUncertain( + f"could not confirm execution stopped: {exc}; partial outputs were retained. " + "Stop the allocation and verify its commands/containers have ended before retrying" + ) from exc diff --git a/src/lightcone/engine/execution_resources.py b/src/lightcone/engine/execution_resources.py new file mode 100644 index 00000000..256a8797 --- /dev/null +++ b/src/lightcone/engine/execution_resources.py @@ -0,0 +1,155 @@ +"""Recipe resource requests and admission to stock Dask workers.""" + +from __future__ import annotations + +import math +import re +from decimal import Decimal +from typing import Any, Self + +from pydantic import BaseModel, ConfigDict, Field, ValidationError, field_validator + +from lightcone.engine.project import ProjectError + + +class TaskResources(BaseModel): + """Reserve CPUs and memory for a recipe, with an optional walltime limit.""" + + model_config = ConfigDict(frozen=True, strict=True, extra="forbid") + + cpus: int = Field(default=1, gt=0) + memory_bytes: int | None = Field(default=None, gt=0) + time_seconds: int | None = Field(default=None, gt=0) + + @field_validator("cpus", mode="before") + @classmethod + def _whole_cpus(cls, value: object) -> object: + # ASTRA permits fractional CPUs. This executor reserves whole CPUs; + # accepting 4.0 is exact, whereas rounding 0.5 would hide a policy change. + if isinstance(value, float): + if not math.isfinite(value) or not value.is_integer(): + raise ValueError("fractional CPUs are not supported; request whole CPUs") + return int(value) + return value + + @classmethod + def parse(cls, value: object) -> Self: + """Parse ASTRA's ``recipe.resources`` into explicit execution units. + + Args: + value: The recipe resource mapping, or ``None`` when omitted. + + Returns: + A validated CPU, byte, and second request. + + Raises: + ProjectError: A requirement is invalid or cannot be honored. + """ + if value is None: + return cls() + if not isinstance(value, dict): + raise ProjectError("recipe.resources must be a mapping") + if extra := value.keys() - {"cpus", "memory", "time_limit"}: + names = ", ".join(sorted(map(str, extra))) + raise ProjectError(f"unsupported recipe resource requirements: {names}") + parsed = {"cpus": value.get("cpus", 1)} + if "memory" in value: + parsed["memory_bytes"] = _memory(value["memory"]) + if "time_limit" in value: + parsed["time_seconds"] = _duration(value["time_limit"]) + try: + return cls.model_validate(parsed) + except ValidationError as exc: + detail = "; ".join( + f"{'.'.join(map(str, item['loc']))}: {item['msg']}" + for item in exc.errors(include_url=False, include_input=False) + ) + raise ProjectError(f"invalid recipe resources: {detail}") from exc + + def requirements( + self, workers: dict[str, Any], *, whole_worker: bool = False + ) -> dict[str, float]: + """Choose Dask resource reservations that fit an individual worker. + + Args: + workers: The ``workers`` mapping from Dask's scheduler information. + whole_worker: Reserve a worker's entire CPU and memory budget for + an arbitrary command without declared resource requirements. + + Returns: + Dask's numeric ``CPU`` and ``MEMORY`` resource requirements. + + Raises: + ProjectError: Capacity is unknown, a request cannot fit, or an + unspecified budget is ambiguous across heterogeneous workers. + """ + capacities: set[tuple[float, float]] = set() + for info in workers.values(): + resources = info.get("resources", {}) if isinstance(info, dict) else {} + values = [] + for name in ("CPU", "MEMORY"): + value = resources.get(name) if isinstance(resources, dict) else None + if ( + isinstance(value, bool) + or not isinstance(value, (int, float)) + or not math.isfinite(value) + or value <= 0 + or not float(value).is_integer() + ): + raise ProjectError( + "cluster workers must advertise positive whole CPU and MEMORY budgets; " + "relaunch the cluster with the current Lightcone installation" + ) + values.append(float(value)) + capacities.add((values[0], values[1])) + if not capacities: + raise ProjectError("cluster has no workers available for execution") + if (whole_worker or self.memory_bytes is None) and len(capacities) != 1: + raise ProjectError( + "unspecified task resources require workers with identical CPU and memory budgets" + ) + available_cpus, available_memory = next(iter(capacities)) + requested = { + "CPU": available_cpus if whole_worker else float(self.cpus), + "MEMORY": ( + available_memory + if whole_worker or self.memory_bytes is None + else float(self.memory_bytes) + ), + } + if not any( + cpus >= requested["CPU"] and memory >= requested["MEMORY"] + for cpus, memory in capacities + ): + raise ProjectError( + f"task needs {requested['CPU']:g} CPUs and " + f"{requested['MEMORY'] / 1024**3:g} GiB on one worker; " + "no worker in this cluster can satisfy that request" + ) + return requested + + +def _memory(value: object) -> int: + if not isinstance(value, str) or not ( + match := re.fullmatch(r"([0-9]+(?:\.[0-9]+)?)\s*([KMGTPE]i?B?|kB?|B)", value) + ): + raise ProjectError("recipe memory must include units, e.g. 512Mi, 16Gi, or 8GB") + unit = match[2].lower() + exponent = 0 if unit == "b" else "kmgtpe".index(unit[0]) + 1 + factor: int = (1024 if "i" in unit else 1000) ** exponent + numerator, denominator = Decimal(match[1]).as_integer_ratio() + amount, remainder = divmod(numerator * factor, denominator) + if amount <= 0 or remainder: + raise ProjectError("recipe memory must be a positive whole number of bytes") + return amount + + +def _duration(value: object) -> int: + if not isinstance(value, str) or not ( + match := re.fullmatch(r"(?:(\d+)d)?(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?", value) + ): + raise ProjectError("recipe time_limit must be a duration, e.g. 30m, 1h30m, or 45s") + seconds = sum(int(part or 0) * unit for part, unit in zip(match.groups(), (86400, 3600, 60, 1))) + if seconds <= 0: + raise ProjectError("recipe time_limit must be positive") + return seconds diff --git a/src/lightcone/engine/materialize.py b/src/lightcone/engine/materialize.py index e9791540..b15c84ec 100644 --- a/src/lightcone/engine/materialize.py +++ b/src/lightcone/engine/materialize.py @@ -34,14 +34,13 @@ import functools import json import re -from collections.abc import Iterator, Sequence +from collections.abc import Iterable, Iterator, Sequence from contextlib import contextmanager from dataclasses import asdict, dataclass, field, replace from pathlib import Path from typing import TYPE_CHECKING, Any, Protocol -from uuid import uuid4 -from lightcone.engine import assets, container, dataset, identity, plan, project, worker +from lightcone.engine import assets, container, dataset, execution, identity, plan, project, worker from lightcone.engine.plan import Graph, Key, Task from lightcone.engine.project import ProjectError @@ -515,94 +514,98 @@ def materialize( # maintainer. Nothing is submitted, so no allocation is needed. _converge_crate(root, report, full, dsid) return report - with cluster_for_run(cluster_id) as scheduler: - _fetch_inputs(root, graph, report) - # Materialize is one of the two verbs allowed to build the image (the - # other is `lc build`); the probe and the rerun entry point only find - # one. Resolved once, then handed to every task — the HEAD discipline. - runtime = container.runtime_for_run(root, build=True) - # Converge the environment: workers pass `--no-sync`, so this is the - # only place on a run's path where it is made to match the lock. (A - # rerun does not come through here; its entry point converges too.) - report.warnings.extend(f"uv: {w}" for w in container.converge(runtime)) - # The run's driver-resolved facts, each read once: HEAD because the - # driver commits as outputs land and a per-task read would stamp - # later manifests with a commit this run created; the uv probe - # because attestation is a fact about the run (and empty is an - # answer, not a failure); one content-hash memo because a declared - # input shared by several outputs is the same bytes every time. - versions = assets.Versions() - for path in { - path - for task in graph.tasks.values() - for name, path in task.inputs.items() - if name not in task.produced_by - }: - try: - versions.of(path) - except Exception: - # Keep unreadable-input failures inside the tasks that need them. - pass - context = worker.RunContext( - env_version=env_version, - head=dataset.head(root), - versions=versions, - runtime=runtime, - uv_version=project.uv_version(root), - ) - # The history question is the driver's to answer — workers have no - # git, by design — so each task is told up front whether its - # directory was last written by something other than its own run - # record. A foreign write contradicts the manifest, and a worker that - # trusted the recorded digest would skip the output forever. Guarded - # on the manifest's presence, as `_classified` is: without one the - # answer is dead — the output is remade regardless — and each ask is - # a git process. - foreign = { - key: _foreign_write(root, task) if task.manifest_path.is_file() else None - for key, task in graph.tasks.items() - } - pending: dict[Key, Any] = {} - # Futures retain dependency ordering; task placement belongs to the - # selected cluster, while commits stay in this one driver thread. - for key in graph.order(): - task = graph.tasks[key] - pending[key] = scheduler.submit( - worker.materialize, - root, - task, - context, - refresh, - foreign[key], - *[pending[dep] for dep in task.depends_on], - key=_name(key), + unreported: set[Key] = set() + scheduler: Scheduler | None = None + try: + with cluster_for_run(cluster_id) as scheduler: + scheduler.validate(graph.tasks.values()) + _fetch_inputs(root, graph, report) + # Materialize is one of the two verbs allowed to build the image (the + # other is `lc build`); the probe and the rerun entry point only find + # one. Resolved once, then handed to every task — the HEAD discipline. + runtime = container.runtime_for_run(root, build=True) + # Converge the environment: workers pass `--no-sync`, so this is the + # only place on a run's path where it is made to match the lock. (A + # rerun does not come through here; its entry point converges too.) + report.warnings.extend(f"uv: {w}" for w in container.converge(runtime)) + # The run's driver-resolved facts, each read once: HEAD because the + # driver commits as outputs land and a per-task read would stamp + # later manifests with a commit this run created; the uv probe + # because attestation is a fact about the run (and empty is an + # answer, not a failure); one content-hash memo because a declared + # input shared by several outputs is the same bytes every time. + versions = assets.Versions() + for path in { + path + for task in graph.tasks.values() + for name, path in task.inputs.items() + if name not in task.produced_by + }: + try: + versions.of(path) + except Exception: + # Keep unreadable-input failures inside the tasks that need them. + pass + context = worker.RunContext( + env_version=env_version, + head=dataset.head(root), + versions=versions, + runtime=runtime, + uv_version=project.uv_version(root), ) - from lightcone.engine.compute import UNSTOPPED - - # An unreported task can still have a running subprocess. Leave its - # partial files in place on interruption rather than restoring over it. - outstanding = len(pending) - for result in scheduler.completed(list(pending.values())): - outstanding -= 1 - try: + # The history question is the driver's to answer — workers have no + # git, by design — so each task is told up front whether its + # directory was last written by something other than its own run + # record. A foreign write contradicts the manifest, and a worker that + # trusted the recorded digest would skip the output forever. Guarded + # on the manifest's presence, as `_classified` is: without one the + # answer is dead — the output is remade regardless — and each ask is + # a git process. + foreign = { + key: _foreign_write(root, task) if task.manifest_path.is_file() else None + for key, task in graph.tasks.items() + } + pending: dict[Key, Any] = {} + # Futures retain dependency ordering; task placement belongs to the + # selected cluster, while commits stay in this one driver thread. + for key in graph.order(): + task = graph.tasks[key] + unreported.add(key) + pending[key] = scheduler.submit( + worker.materialize, + root, + task, + context, + refresh, + foreign[key], + *[pending[dep] for dep in task.depends_on], + key=_name(key), + ) + + for result in scheduler.completed(list(pending.values())): _consume(root, graph.tasks[result.key], result, dsid, runtime, report) - except Exception as exc: - if not outstanding: - raise - raise ProjectError(f"{exc}. {UNSTOPPED}") from exc - # The tree was clean at the start-of-run refusal and save/restore - # keeps `results/` clean, so anything dirty *now* was edited while - # the graph ran — and every manifest records the starting commit, - # which no longer describes that code. A warning, never a manifest - # field: the driver does not rewrite files the worker owns. - if edited := dataset.status(root): - names = ", ".join(sorted(path for _, path in edited)) - report.warnings.append( - f"edited while the run was in flight: {names} — the manifests " - "record the starting commit, which no longer describes this code" - ) - _converge_crate(root, report, full, dsid) - return report + unreported.discard(result.key) + # The tree was clean at the start-of-run refusal and save/restore + # keeps `results/` clean, so anything dirty *now* was edited while + # the graph ran — and every manifest records the starting commit, + # which no longer describes that code. A warning, never a manifest + # field: the driver does not rewrite files the worker owns. + if edited := dataset.status(root): + names = ", ".join(sorted(path for _, path in edited)) + report.warnings.append( + f"edited while the run was in flight: {names} — the manifests " + "record the starting commit, which no longer describes this code" + ) + _converge_crate(root, report, full, dsid) + return report + + except BaseException: + # An outer context's error can mask an uncertain cleanup. Require a + # positive acknowledgment, independent of which exception reached us. + if scheduler is not None and scheduler.stopped: + for key in unreported: + dataset.restore(root, _owned(root, graph.tasks[key])) + raise def _consume( @@ -646,6 +649,15 @@ class Scheduler(Protocol): they land, keeping the commit logic independent of the provider. """ + @property + def stopped(self) -> bool: + """Whether this invocation positively confirmed all claimed tasks stopped.""" + ... + + def validate(self, tasks: Iterable[Task]) -> None: + """Refuse unsatisfiable resource requests before preparation or submission.""" + ... + def submit(self, fn: Any, *args: Any, key: str) -> Any: """Schedule a call. @@ -676,16 +688,32 @@ class _Dask: """A borrowed Dask client, narrowed to what the graph driver needs.""" client: Any - invocation: str + invocation: execution.Invocation output: Forwarder + workers: dict[str, Any] + + @property + def stopped(self) -> bool: + """Expose positive cleanup confirmation after the connection context exits.""" + return self.invocation.stopped + + def validate(self, tasks: Iterable[Task]) -> None: + """Require each selected task to fit a worker before any task starts.""" + for task in tasks: + try: + task.resources.requirements(self.workers) + except ProjectError as exc: + raise ProjectError(f"{_name(task.key)}: {exc}") from exc def submit(self, fn: Any, *args: Any, key: str) -> Any: """Submit an ordinary Dask task with a unique key and forwarded output.""" from lightcone.engine.compute.output import call - return self.client.submit( + task: Task = args[1] + resources = task.resources.requirements(self.workers) + return self.invocation.submit( call, fn, self.output.topic, key, *args, - key=f"lc-{self.invocation}-{key}", pure=False, + key=key, resources=resources, ) def completed(self, handles: list[Any]) -> Iterator[worker.TaskResult]: @@ -725,9 +753,11 @@ def cluster_for_run(cluster_id: str) -> Iterator[Scheduler]: from lightcone.engine.compute.output import forwarding with compute.connect(cluster_id) as client: - invocation = uuid4().hex - with forwarding(client, stdout="stderr") as output: - yield _Dask(client, invocation, output) + with ( + forwarding(client, stdout="stderr") as output, + execution.invocation(client) as invocation, + ): + yield _Dask(client, invocation, output, client.scheduler_info()["workers"]) def _fetch_inputs(root: Path, graph: Graph, report: MaterializeReport) -> None: diff --git a/src/lightcone/engine/plan.py b/src/lightcone/engine/plan.py index 46067016..614e5fbb 100644 --- a/src/lightcone/engine/plan.py +++ b/src/lightcone/engine/plan.py @@ -23,11 +23,12 @@ from __future__ import annotations -from dataclasses import dataclass +from dataclasses import dataclass, field from graphlib import CycleError, TopologicalSorter from pathlib import Path from lightcone.engine import assets, identity +from lightcone.engine.execution_resources import TaskResources from lightcone.engine.project import SPEC_FILENAME, ProjectError #: A task's identity within a run: which universe, which output. @@ -51,6 +52,7 @@ class Task: produced_by: dict[str, Key] decisions: dict[str, str] definition_version: str + resources: TaskResources = field(default_factory=TaskResources) @property def manifest_path(self) -> Path: @@ -331,6 +333,10 @@ def file_of(out: object) -> Path: ) except ValueError as e: raise ProjectError(f"output `{out.id}`: {e}") from e + try: + resources = TaskResources.parse((out.definition.get("recipe") or {}).get("resources")) + except ProjectError as e: + raise ProjectError(f"output `{out.id}`: {e}") from e tasks.append( Task( @@ -344,6 +350,7 @@ def file_of(out: object) -> Path: definition_version=identity.definition_version( recipe=recipe, decisions=out.decisions, fmt=str(out.format) ), + resources=resources, ) ) return tasks diff --git a/src/lightcone/engine/run.py b/src/lightcone/engine/run.py index d54b7ed2..449bdbe2 100644 --- a/src/lightcone/engine/run.py +++ b/src/lightcone/engine/run.py @@ -18,9 +18,9 @@ from dataclasses import replace from pathlib import Path from typing import Any -from uuid import uuid4 -from lightcone.engine import container, sandbox +from lightcone.engine import container, execution, sandbox +from lightcone.engine.execution_resources import TaskResources from lightcone.engine.project import ( SPEC_FILENAME, ProjectError, @@ -51,13 +51,15 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. require_uv() paths = input_paths(project, read_spec(project)) with compute.connect(cluster_id) as client: + resources = TaskResources().requirements( + client.scheduler_info()["workers"], whole_worker=True, + ) runtime = container.runtime_for_run(project, build=False) notes = [f"uv: {warning}" for warning in container.converge(runtime)] - invocation = uuid4().hex - with forwarding(client) as output: - future = client.submit( + with forwarding(client) as output, execution.invocation(client) as invocation: + future = invocation.submit( call, _probe, output.topic, "probe", runtime, paths, tuple(command), - key=f"lc-{invocation}-probe", pure=False, + key="probe", resources=resources, ) try: outcome: sandbox.Outcome = future.result() @@ -84,6 +86,7 @@ def _probe( outcome = sandbox.run( container.backend(runtime), policy, command, cwd=runtime.root, prefix=uv_prefix(runtime.root), env=child_env(), output=output, + cancelled=execution.cancelled, ) return outcome diff --git a/src/lightcone/engine/sandbox/boundary.py b/src/lightcone/engine/sandbox/boundary.py index 7b87ea4f..a475a984 100644 --- a/src/lightcone/engine/sandbox/boundary.py +++ b/src/lightcone/engine/sandbox/boundary.py @@ -10,7 +10,6 @@ from __future__ import annotations import shutil -import subprocess import sys import threading from collections import deque @@ -22,6 +21,7 @@ from lightcone.engine.sandbox import policy as policy_module from lightcone.engine.sandbox.model import Attestation, Backend, Capability, Policy +from lightcone.engine.sandbox.processes import Command #: How much of the child's stderr to keep for the denial classifier. The #: denial is in the last few lines of a traceback, and a recipe that @@ -117,10 +117,17 @@ def scope(policy: Policy) -> Iterator[Policy]: Yields: The same policy, with its ``tmp_home`` removed on exit. """ + from lightcone.engine.execution import ExecutionUncertain + + cleanup = True try: yield policy + except ExecutionUncertain: + cleanup = False + raise finally: - shutil.rmtree(policy.tmp_home, ignore_errors=True) + if cleanup: + shutil.rmtree(policy.tmp_home, ignore_errors=True) def run( @@ -132,6 +139,8 @@ def run( env: dict[str, str], prefix: Sequence[str] = (), output: Callable[[str, bytes], None] | None = None, + cancelled: Callable[[], bool] | None = None, + timeout: float | None = None, ) -> Outcome: """Run a command through a backend, and explain it if it fails. @@ -154,6 +163,8 @@ def run( host-resolved ``env``. output: Optional receiver for unchanged stdout/stderr bytes, used when the caller forwards a remote command's output to its own terminal. + cancelled: A stop predicate polled while the command runs. + timeout: Maximum command runtime in seconds, or no bound. Returns: The exit code, what was actually enforced, and any lines the @@ -176,33 +187,33 @@ def run( if backend.capability.kind == "none": notes.append(_downgrade_note(backend.capability)) - proc = subprocess.Popen( - wrapped, - cwd=cwd, - env=child_env, - stdin=subprocess.DEVNULL if output is not None else None, - stdout=subprocess.PIPE if output is not None else None, - stderr=subprocess.PIPE, - bufsize=0, - ) - assert proc.stderr is not None # Popen was given PIPE - tail = _Tail(proc.stderr, output) - tail.start() - stdout: threading.Thread | None = None - if output is not None: - assert proc.stdout is not None - - def forward() -> None: - assert proc.stdout is not None and output is not None - while chunk := proc.stdout.read(64 * 1024): - output("stdout", chunk) - - stdout = threading.Thread(target=forward, daemon=True) - stdout.start() - returncode = proc.wait() - tail.join(timeout=5) - if stdout is not None: - stdout.join(timeout=5) + with Command( + wrapped, cwd=cwd, env=child_env, capture=output is not None, timeout=timeout, + container=backend.contains_prefix, + ) as command: + proc = command.process + assert proc.stderr is not None # Popen was given PIPE + tail = _Tail(proc.stderr, output) + tail.start() + stdout: threading.Thread | None = None + if output is not None: + assert proc.stdout is not None + + def forward() -> None: + assert proc.stdout is not None and output is not None + while chunk := proc.stdout.read(64 * 1024): + output("stdout", chunk) + + stdout = threading.Thread(target=forward, daemon=True) + stdout.start() + try: + returncode, lifecycle_note = command.wait(cancelled) + finally: + tail.join(timeout=5) + if stdout is not None: + stdout.join(timeout=5) + if lifecycle_note: + notes.append(lifecycle_note) # Imported here, not at module scope: `sandbox/__init__` loads this # module eagerly, and the shim drags ctypes in for one integer. @@ -228,7 +239,7 @@ def forward() -> None: f"above, `{attestation.mechanism}` exit 125) — this is a " "runtime problem, not your command's" ) - elif returncode != 0 and attestation.mechanism != "none": + elif returncode != 0 and attestation.mechanism != "none" and not lifecycle_note: from lightcone.engine.sandbox import denial explanation = denial.explain(tail.text(), policy, cwd=cwd) diff --git a/src/lightcone/engine/sandbox/oci.py b/src/lightcone/engine/sandbox/oci.py index 0acbf4e9..303273f7 100644 --- a/src/lightcone/engine/sandbox/oci.py +++ b/src/lightcone/engine/sandbox/oci.py @@ -83,7 +83,9 @@ def wrap(self, policy: Policy, argv: Sequence[str]) -> list[str]: mounts += [f"--volume={path.resolve()}:{path}:rw" for path in policy.write] overlay = [f"--env={k}={v}" for k, v in sorted(policy.env.items())] return [ - self.runtime, "run", "--rm", + # The custodian retains the native record until it has inspected + # the immutable container ID and confirmed the payload stopped. + self.runtime, "run", "--entrypoint", "", # The rootfs is read-only so a write outside the declared set # is a loud denial rather than bytes vanishing with the diff --git a/src/lightcone/engine/sandbox/processes.py b/src/lightcone/engine/sandbox/processes.py new file mode 100644 index 00000000..52d99c96 --- /dev/null +++ b/src/lightcone/engine/sandbox/processes.py @@ -0,0 +1,339 @@ +"""Keep custody of command processes independently of their Dask worker. + +The small child process owns the command's process group. Its control pipe +closes if the worker dies, so cleanup does not depend on a Python finally block +in that worker. Groups stay in the allocation's session for native shutdown. +""" + +from __future__ import annotations + +import json +import os +import re +import select +import signal +import subprocess +import sys +import tempfile +import time +from collections.abc import Callable, Sequence +from pathlib import Path +from types import FrameType, TracebackType +from typing import Any, Self + +import psutil + +_GRACE = 1.0 +_CLEANUP_TIMEOUT = 15.0 + + +def members(*, group: int | None = None, session: int | None = None) -> list[psutil.Process]: + """Find live owned processes, retaining their birth identities for signalling.""" + found = [] + for process in psutil.process_iter(): + try: + if process.uids().real != os.getuid(): + continue + except (psutil.NoSuchProcess, psutil.AccessDenied): + continue + try: + if process.status() == psutil.STATUS_ZOMBIE: + continue + if group is not None and os.getpgid(process.pid) != group: + continue + if session is not None and os.getsid(process.pid) != session: + continue + process.create_time() + found.append(process) + except (psutil.NoSuchProcess, ProcessLookupError): + continue + return found + + +def has_custodian(processes: Sequence[psutil.Process]) -> bool: + """Check whether allocation shutdown must allow command cleanup to finish.""" + for process in processes: + try: + if str(Path(__file__)) in process.cmdline(): + return True + except (psutil.NoSuchProcess, psutil.AccessDenied): + continue + return False + + +def _drain(process: subprocess.Popen[bytes]) -> bool: + """Stop the whole command group while its unreaped leader pins the group ID.""" + for sig in (signal.SIGTERM, signal.SIGKILL): + try: + os.killpg(process.pid, sig) + except ProcessLookupError: + pass + deadline = time.monotonic() + _GRACE + while members(group=process.pid): + if time.monotonic() >= deadline: + break + time.sleep(0.025) + else: + process.wait() + return True + return False + + +class Command: + """Launch a custodian and collect its verified completion report. + + Args: + argv: Fully wrapped command. + cwd: Command working directory. + env: Command environment. + capture: Whether to pipe stdout and disable stdin. + timeout: Maximum command runtime in seconds, or no bound. + container: Whether argv starts a supported OCI runtime. + """ + + def __init__( + self, argv: Sequence[str], *, cwd: Path, env: dict[str, str], capture: bool, + timeout: float | None = None, container: bool = False, + ) -> None: + self._control, control_write = os.pipe() + status_read, self._status = os.pipe() + self._writer = os.fdopen(control_write, "wb", buffering=0) + self._reader = os.fdopen(status_read, "rb") + self._deadline = ( + time.monotonic() + timeout + _CLEANUP_TIMEOUT if timeout is not None else None + ) + try: + self.process = subprocess.Popen( + [sys.executable, "-P", str(Path(__file__)), str(self._control), str(self._status)], + pass_fds=(self._control, self._status), + stdin=subprocess.DEVNULL if capture else None, + stdout=subprocess.PIPE if capture else None, + stderr=subprocess.PIPE, + bufsize=0, + ) + self._writer.write(json.dumps({ + "argv": list(argv), "cwd": str(cwd), "env": env, + "timeout": timeout, "container": container, + }).encode() + b"\n") + except BaseException: + self._writer.close() + if hasattr(self, "process"): + try: + self.wait() + except Exception as cleanup_error: + from lightcone.engine.execution import ExecutionCancelled + + if not isinstance(cleanup_error, ExecutionCancelled): + raise + else: + self._reader.close() + raise + finally: + os.close(self._control) + os.close(self._status) + + def __enter__(self) -> Self: + return self + + def __exit__( + self, exc_type: type[BaseException] | None, exc: BaseException | None, + traceback: TracebackType | None, + ) -> None: + from lightcone.engine.execution import ExecutionCancelled + + if not self._reader.closed: + self._writer.close() + try: + self.wait() + except ExecutionCancelled: + pass + + def wait(self, cancelled: Callable[[], bool] | None = None) -> tuple[int, str]: + """Wait for command completion; cancellation includes verified cleanup.""" + from lightcone.engine.execution import ExecutionCancelled, ExecutionUncertain + + requested = self._writer.closed + deadline = time.monotonic() + _CLEANUP_TIMEOUT if requested else self._deadline + try: + while self.process.poll() is None: + if not requested and cancelled is not None and cancelled(): + self._writer.close() + requested = True + deadline = time.monotonic() + _CLEANUP_TIMEOUT + if deadline is not None and time.monotonic() >= deadline: + raise ExecutionUncertain("command cleanup did not finish") + time.sleep(0.025) + except BaseException: + self._writer.close() + try: + remaining = ( + _CLEANUP_TIMEOUT if deadline is None + else max(0.0, deadline - time.monotonic()) + ) + self.process.wait(timeout=remaining) + except subprocess.TimeoutExpired as exc: + self._reader.close() + raise ExecutionUncertain("command cleanup did not finish") from exc + report = self._report() + if report.get("error"): + raise ExecutionUncertain(str(report["error"])) + raise + finally: + self._writer.close() + report = self._report() + if report.get("error"): + raise ExecutionUncertain(str(report["error"])) + if report.get("start_error"): + raise OSError(str(report["start_error"])) + if report.get("cancelled"): + raise ExecutionCancelled("command cancelled; its processes have stopped") + return int(report["returncode"]), str(report.get("note", "")) + + def _report(self) -> dict[str, Any]: + from lightcone.engine.execution import ExecutionUncertain + + try: + with self._reader: + result = json.load(self._reader) + if not isinstance(result, dict) or "returncode" not in result: + raise ValueError("missing completion record") + return result + except (OSError, ValueError) as exc: + raise ExecutionUncertain( + "command custodian ended without confirming that its processes stopped" + ) from exc + + +def _container_cleanup(runtime: str, cidfile: Path, env: dict[str, str]) -> None: + """Stop, inspect and remove exactly the container created by this command.""" + try: + identity = cidfile.read_text().strip() + except OSError as exc: + raise RuntimeError("container creation ended before its identity was recorded") from exc + if not re.fullmatch(r"[0-9a-f]{64}", identity): + raise RuntimeError("container runtime did not publish a valid immutable container ID") + + def run(*args: str) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [runtime, *args], env=env, stdin=subprocess.DEVNULL, + capture_output=True, text=True, timeout=2, + ) + + def running() -> bool: + result = run("inspect", "--format", "{{.State.Running}}", identity) + if result.returncode or result.stdout.strip() not in ("true", "false"): + raise RuntimeError(f"cannot establish whether container {identity} stopped") + return result.stdout.strip() == "true" + + try: + if running(): + try: + run("stop", "--time", "1", identity) + except subprocess.TimeoutExpired: + pass + if running(): + run("kill", identity) + deadline = time.monotonic() + 2 + while running(): + if time.monotonic() >= deadline: + raise RuntimeError(f"container {identity} remains running after kill") + time.sleep(0.025) + except BaseException: + # Inspection can fail while native termination still works. Attempt + # cleanup without converting that lack of evidence into success. + try: + run("kill", identity) + except (OSError, subprocess.SubprocessError): + pass + raise + # A stopped container is safe even when the runtime cannot remove its metadata. + try: + run("rm", identity) + except (OSError, subprocess.SubprocessError): + pass + + +def _supervise(control: int, status: int) -> None: + stopped = False + + def stop(_signum: int, _frame: FrameType | None) -> None: + nonlocal stopped + stopped = True + + signal.signal(signal.SIGTERM, stop) + signal.signal(signal.SIGINT, stop) + process: subprocess.Popen[bytes] | None = None + report: dict[str, Any] = {"returncode": 125} + with os.fdopen(control, "rb", buffering=0) as channel, tempfile.TemporaryDirectory( + prefix="lc-command-" + ) as directory: + try: + config = json.loads(channel.readline()) + argv = config["argv"] + cidfile = Path(directory) / "container" + if config["container"]: + argv[2:2] = ["--cidfile", str(cidfile)] + if stopped or select.select([channel], [], [], 0)[0]: + report["cancelled"] = True + return + process = subprocess.Popen( + argv, cwd=config["cwd"], env=config["env"], process_group=0, + ) + started = time.monotonic() + reason = "" + # Keep the leader unreaped until group cleanup is complete, pinning + # its PID/PGID against reuse. Unlike waitid(WNOWAIT), psutil also + # supports macOS with Python 3.11 and 3.12. + leader = psutil.Process(process.pid) + while leader.status() != psutil.STATUS_ZOMBIE: + if stopped or select.select([channel], [], [], 0.025)[0]: + reason = "cancelled" + break + if ( + config["timeout"] is not None + and time.monotonic() - started >= config["timeout"] + ): + reason = "timed out" + break + if not reason and members(group=process.pid): + reason = "left background processes running" + container_stopped = False + if config["container"]: + # A runtime startup failure with no CID has not published a + # container; interruption in that window cannot prove the same. + if cidfile.exists() or reason: + _container_cleanup(argv[0], cidfile, config["env"]) + # Podman removes its cidfile together with the container. + # Keep the verified outcome, not the file's later existence. + container_stopped = True + if not _drain(process): + raise RuntimeError("command processes remain alive after SIGKILL") + if config["container"] and not container_stopped and process.returncode != 125: + raise RuntimeError("container runtime exited without recording its identity") + report["returncode"] = process.returncode + if reason == "cancelled": + report["cancelled"] = True + report["returncode"] = 130 + elif reason: + report["returncode"] = 124 if reason == "timed out" else 1 + report["note"] = f"command {reason}; its processes have stopped" + except BaseException as exc: + if process is None: + report["start_error"] = f"command could not start: {str(exc)[:2048]}" + else: + try: + _drain(process) + except BaseException: + pass + report["error"] = f"cannot confirm command cleanup: {str(exc)[:2048]}" + finally: + with os.fdopen(status, "wb", buffering=0) as result: + try: + result.write(json.dumps(report).encode()) + except BrokenPipeError: + # The worker can die before receiving the cleanup report. + pass + + +if __name__ == "__main__": + _supervise(int(sys.argv[1]), int(sys.argv[2])) diff --git a/src/lightcone/engine/worker.py b/src/lightcone/engine/worker.py index af1be88a..2a16fc88 100644 --- a/src/lightcone/engine/worker.py +++ b/src/lightcone/engine/worker.py @@ -18,11 +18,11 @@ Keep this module cheap to import: no click, no rich. It is on the ``python -m`` path of every rerun, and of every task in every run. -Nothing here writes to git, and nothing here raises. A task that fails -returns a result saying so, because Dask propagates an exception to every -dependent and "who actually failed" would stop being answerable — -reporting every independent failure in one run is most of the point of -owning the loop. +Recipe failures return results, so independent tasks can finish and report +their own failures. Cancellation and uncertain execution propagate instead: +the driver must stop the invocation and establish that its writers have stopped +before it can restore partial outputs. Cluster task resource reservations belong +to the scheduler; the subprocess boundary enforces each recipe's time limit. """ from __future__ import annotations @@ -35,7 +35,7 @@ from pathlib import Path from typing import Literal -from lightcone.engine import assets, container, dataset, identity, plan, project, sandbox +from lightcone.engine import assets, container, dataset, execution, identity, plan, project, sandbox from lightcone.engine.plan import Key, Task from lightcone.engine.project import ( ProjectError, @@ -58,7 +58,7 @@ @dataclass(frozen=True) class TaskResult: - """What one task did. Returned, never raised, and handed to dependents.""" + """A completed task's outcome, passed to its dependents.""" key: Key status: Literal["ok", "current", "behind", "failed", "blocked"] @@ -126,10 +126,9 @@ def materialize( ) -> TaskResult: """Make *task* if it needs making. What Dask submits, once per task. - Where "the worker never raises" is enforced. Dask re-raises a task's - exception in the driver, which would abort every other task in flight, - so the contract is absolute — and one assembled from individually - guarded call sites is only as true as the last person to add one. + Ordinary failures become task results so independent outputs can continue. + Cancellation and uncertain execution abort the invocation: treating either + as an ordinary failure could restore files while a subprocess still writes. Args: root: The project root. @@ -146,11 +145,17 @@ def materialize( output: Optional receiver forwarding recipe stdout and stderr bytes. Returns: - What happened. Never raises. + The output's result, including ordinary recipe failures. + + Raises: + ExecutionCancelled: The invocation revoked this task's authorization. + ExecutionUncertain: A writer may still be running; retain its outputs. """ try: return _materialize(root, task, context, refresh, foreign, upstream, output) - except Exception as e: # the contract is that this function returns + except (execution.ExecutionCancelled, execution.ExecutionUncertain): + raise + except Exception as e: # Ordinary recipe failures remain per-output results. return TaskResult(task.key, "failed", reason=f"{type(e).__name__}: {e}") @@ -213,7 +218,9 @@ def execute( left from a previous run would otherwise enter the content hash and be committed as part of an output that never produced it. The context's ``env_version`` is checked either side of the recipe, so a mid-run - lock edit cannot be recorded as if it had been in force. + lock edit cannot be recorded as if it had been in force. The subprocess + boundary applies ``task.resources.time_seconds`` and drains owned processes + before reporting completion. CPU and memory scheduling happens upstream. Args: root: The project root. @@ -226,7 +233,12 @@ def execute( Returns: ``ok`` with the output's ``data_version``, or ``failed``. Commits nothing and never touches git beyond reading HEAD. + + Raises: + ExecutionCancelled: The invocation revoked this task's authorization. + ExecutionUncertain: Subprocess teardown could not be confirmed. """ + execution.check_cancelled() if moved := _gate(root, context.env_version): return TaskResult(task.key, "failed", reason=moved) @@ -258,9 +270,13 @@ def execute( prefix=uv_prefix(root), env=child_env(), output=output, + timeout=task.resources.time_seconds, + cancelled=execution.cancelled, ) finished_at = _now() + execution.check_cancelled() + if outcome.returncode != 0: return TaskResult( task.key, diff --git a/tests/conftest.py b/tests/conftest.py index 2a4d164a..d71c8532 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -5,7 +5,7 @@ import shutil import subprocess import textwrap -from collections.abc import Callable, Iterator +from collections.abc import Callable, Iterable, Iterator from contextlib import contextmanager from pathlib import Path from unittest.mock import MagicMock @@ -151,6 +151,11 @@ class _Inline: are the upstream results themselves, exactly what the worker expects. """ + stopped = True # Calls are synchronous; no remote work can survive the fixture. + + def validate(self, tasks: Iterable[object]) -> None: + """Run fixture tasks without a finite cluster resource envelope.""" + def submit(self, fn: Callable[..., object], *args: object, key: str) -> object: return fn(*args) @@ -180,7 +185,8 @@ def cluster_id(monkeypatch: pytest.MonkeyPatch) -> Iterator[str]: from lightcone.engine import compute with LocalCluster( - n_workers=1, threads_per_worker=2, processes=False, dashboard_address=None + n_workers=1, threads_per_worker=2, processes=False, dashboard_address=None, + resources={"CPU": 2, "MEMORY": 1024**3}, ) as cluster: @contextmanager def connect(value: str) -> Iterator[Client]: diff --git a/tests/test_cli.py b/tests/test_cli.py index 2fb60cc7..ee1902c1 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -457,21 +457,29 @@ def test_check_explains_that_a_cluster_id_is_not_a_target(runner: CliRunner) -> @pytest.mark.parametrize("command", ["run", "materialize"]) @pytest.mark.parametrize("cluster", [CLUSTER_ID, "analysis"]) +@pytest.mark.parametrize("stopped", [False, True]) def test_execution_interrupt_explains_how_to_stop_remote_work( runner: CliRunner, project: Path, monkeypatch: pytest.MonkeyPatch, command: str, - cluster: str, + cluster: str, stopped: bool, ) -> None: from lightcone.engine import materialize as engine_materialize from lightcone.engine import run as engine_run def interrupt(*args: object, **kwargs: object) -> None: - raise KeyboardInterrupt + exc = KeyboardInterrupt() + exc.execution_stopped = stopped + raise exc monkeypatch.setattr(engine_run, "probe", interrupt) monkeypatch.setattr(engine_materialize, "materialize", interrupt) args = [command, cluster, "--", "true"] if command == "run" else [command, cluster] result = runner.invoke(main, args) assert result.exit_code != 0 + if stopped: + assert "stopped" in result.output + assert "remains available" in result.output + assert "lc compute down" not in result.output + return target = cluster if cluster == CLUSTER_ID else "" assert f"lc compute down {target}" in result.output if cluster != CLUSTER_ID: diff --git a/tests/test_compute_local.py b/tests/test_compute_local.py index 0da9d92a..0f3a9cd4 100644 --- a/tests/test_compute_local.py +++ b/tests/test_compute_local.py @@ -367,7 +367,7 @@ def test_owner_shutdown_drains_recipes_without_a_waiting_cli(provider: LocalProv _ready(provider, identity) child = _ignoring_recipe(provider, identity) os.kill(int(identity.native_id), signal.SIGTERM) - _ended(provider, identity, timeout=10) + _ended(provider, identity, timeout=20) deadline = time.monotonic() + 3 while child.is_running() and child.status() != psutil.STATUS_ZOMBIE: assert time.monotonic() < deadline @@ -387,7 +387,7 @@ def test_termination_escalates_captured_children_when_owner_exits_first( import signal, subprocess, sys, time child = subprocess.Popen([sys.executable, '-c', "import signal,time; signal.signal(signal.SIGTERM, signal.SIG_IGN); " - "print('ready',flush=True); time.sleep(120)"], stdout=subprocess.PIPE) + "print('ready',flush=True); time.sleep(120)"], stdout=subprocess.PIPE, process_group=0) assert child.stdout.readline() == b'ready\\n' print(child.pid, flush=True) signal.signal(signal.SIGTERM, lambda *_: sys.exit(0)) @@ -398,6 +398,7 @@ def test_termination_escalates_captured_children_when_owner_exits_first( ) assert owner.stdout is not None child = psutil.Process(int(owner.stdout.readline())) + assert os.getpgid(child.pid) != owner.pid process = psutil.Process(owner.pid) identity = Identity( namespace=provider.connection.namespace, native_id=str(owner.pid), token=uuid4().hex, diff --git a/tests/test_compute_output.py b/tests/test_compute_output.py index 7a9ebf5e..b1a2bf82 100644 --- a/tests/test_compute_output.py +++ b/tests/test_compute_output.py @@ -10,8 +10,10 @@ import time from collections.abc import Callable, Iterator from pathlib import Path +from types import SimpleNamespace from uuid import uuid4 +import psutil import pytest from lightcone.engine import sandbox @@ -91,38 +93,90 @@ def test_probe_uses_allocation_environment_and_accepts_explicit_command_variable assert result.stdout == expected -def test_interrupt_warns_that_the_remote_command_may_still_run( - analysis: Callable[..., Path], detached_cluster: str, +@pytest.mark.parametrize( + "stop_signal", [signal.SIGINT, signal.SIGKILL], ids=["interrupt", "client-loss"], +) +def test_interrupt_or_client_loss_stops_command_and_keeps_cluster_usable( + analysis: Callable[..., Path], detached_cluster: str, stop_signal: signal.Signals, ) -> None: root = analysis("version: '0.0.13'\nname: analysis\ninputs: []\noutputs: []\n") started = root / "results/started" + cli = [sys.executable, "-c", "from lightcone.cli.commands import main; main()", + "run", detached_cluster, "--"] process = subprocess.Popen( [ - sys.executable, "-c", "from lightcone.cli.commands import main; main()", - "run", detached_cluster, "--", "python", "-c", - "from pathlib import Path; import time; " - "Path('results/started').touch(); time.sleep(30)", + *cli, "python", "-c", + "from pathlib import Path; import os, signal, time; " + "signal.signal(signal.SIGTERM, signal.SIG_IGN); " + "Path('results/started').write_text(str(os.getpid())); time.sleep(60)", ], cwd=root, stdout=subprocess.PIPE, stderr=subprocess.PIPE, ) + remote: psutil.Process | None = None try: deadline = time.monotonic() + 15 - while not started.exists(): + while not started.exists() or not started.read_text(): if process.poll() is not None: pytest.fail(process.communicate()[1].decode(errors="replace")) if time.monotonic() >= deadline: pytest.fail("remote command did not start") time.sleep(0.05) - process.send_signal(signal.SIGINT) - _, stderr = process.communicate(timeout=15) + remote = psutil.Process(int(started.read_text())) + remote.create_time() + process.send_signal(stop_signal) + _, stderr = process.communicate(timeout=30) assert process.returncode != 0 - assert b"may still be running" in stderr - assert f"lc compute down {detached_cluster}".encode() in stderr + if stop_signal == signal.SIGINT: + assert b"the command has stopped" in stderr + assert b"may still be running" not in stderr + deadline = time.monotonic() + (0 if stop_signal == signal.SIGINT else 30) + while True: + try: + alive = remote.is_running() and remote.status() != psutil.STATUS_ZOMBIE + except psutil.NoSuchProcess: + alive = False + if not alive: + break + assert time.monotonic() < deadline, "the departed CLI left its command running" + time.sleep(0.05) assert Compute().status(detached_cluster).phase == "active" + again = subprocess.run( + [*cli, "python", "-c", "print('still usable')"], + cwd=root, capture_output=True, timeout=30, + ) + assert again.returncode == 0, again.stderr.decode(errors="replace") + assert again.stdout == b"still usable\n" finally: if process.poll() is None: process.kill() process.wait(timeout=5) + if remote is not None and remote.is_running(): + try: + remote.kill() + except psutil.NoSuchProcess: + pass + + +def test_failed_output_completion_marker_preserves_execution_uncertainty( + monkeypatch: pytest.MonkeyPatch, +) -> None: + import distributed + + from lightcone.engine.compute.output import call + from lightcone.engine.execution import ExecutionUncertain + + uncertainty = ExecutionUncertain("container termination could not be confirmed") + + def execute(*, output: Callable[[str, bytes], None]) -> None: + raise uncertainty + + def log_event(*args: object) -> None: + raise OSError("scheduler disconnected") + + monkeypatch.setattr(distributed, "get_worker", lambda: SimpleNamespace(log_event=log_event)) + with pytest.raises(ExecutionUncertain) as raised: + call(execute, "topic", "probe") + assert raised.value is uncertainty def test_dask_cleans_output_history_after_the_borrowed_client_disconnects( diff --git a/tests/test_container_smoke.py b/tests/test_container_smoke.py index 8d023438..7eb3c876 100644 --- a/tests/test_container_smoke.py +++ b/tests/test_container_smoke.py @@ -19,14 +19,17 @@ import subprocess import sys from collections.abc import Callable +from dataclasses import replace from pathlib import Path +from uuid import uuid4 import pytest -from lightcone.engine import assets, container, dataset, image +from lightcone.engine import assets, container, dataset, image, sandbox from lightcone.engine import materialize as engine from lightcone.engine import run as engine_run -from lightcone.engine.project import ProjectError, child_env +from lightcone.engine.project import ProjectError, child_env, uv_prefix +from lightcone.engine.sandbox.oci import OCIBackend REQUIRED_ENV = "LC_CONTAINER_TESTS_REQUIRED" @@ -210,6 +213,44 @@ def test_the_probe_and_its_boundary(runtime: str, cproject: Path, cluster_id: st assert loopback.returncode == 0 +def test_timeout_stops_and_removes_a_sigterm_ignoring_container( + runtime: str, cproject: Path, +) -> None: + resolved, _ = container.build(cproject) + container.converge(resolved) + backend = container.backend(resolved) + assert isinstance(backend, OCIBackend) + name = f"lc-timeout-test-{uuid4().hex}" + backend = replace(backend, user_flags=(*backend.user_flags, "--name", name)) + try: + with sandbox.scope(container.policy_for(resolved, [])) as policy: + outcome = sandbox.run( + backend, policy, + ["python", "-c", ( + "import signal,time; from pathlib import Path; " + "signal.signal(signal.SIGTERM, signal.SIG_IGN); " + "Path('results/timeout-started').touch(); time.sleep(60)" + )], cwd=cproject, env=child_env(), prefix=uv_prefix(cproject), timeout=5, + ) + assert (cproject / "results/timeout-started").exists(), "container command never started" + assert outcome.returncode == 124 + assert any("timed out" in note for note in outcome.notes) + inspected = subprocess.run( + [runtime, "inspect", name], capture_output=True, timeout=10, + ) + assert inspected.returncode != 0, "timed-out container was not removed" + # A failed inspection alone could mean the runtime is unavailable. + remaining = subprocess.run( + [runtime, "ps", "--all", "--quiet", "--filter", f"name={name}"], + capture_output=True, check=True, timeout=10, + ) + assert not remaining.stdout.strip() + finally: + subprocess.run( + [runtime, "rm", "--force", name], capture_output=True, timeout=15, + ) + + # ---- lc materialize --------------------------------------------------------- diff --git a/tests/test_execution.py b/tests/test_execution.py new file mode 100644 index 00000000..ec979b13 --- /dev/null +++ b/tests/test_execution.py @@ -0,0 +1,384 @@ +"""Ordinary Dask tasks must not repeat effects or outlive their invocation silently.""" + +from __future__ import annotations + +import time +from collections.abc import Iterator +from pathlib import Path +from types import SimpleNamespace +from typing import Any +from uuid import uuid4 + +import pytest +from distributed import Client, LocalCluster, get_worker + +from lightcone.engine import execution +from lightcone.engine.worker import TaskResult + +_RESOURCES = {"CPU": 1.0, "MEMORY": 1.0} + + +@pytest.fixture +def execution_client(monkeypatch: pytest.MonkeyPatch) -> Iterator[Client]: + monkeypatch.setattr(execution, "_HEARTBEAT", 0.05) + monkeypatch.setattr(execution, "_RPC_TIMEOUT", 2.0) + monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.75) + with LocalCluster( + n_workers=2, threads_per_worker=1, processes=False, + dashboard_address=None, memory_limit=0, resources=_RESOURCES, + ) as cluster, Client(cluster) as client: + yield client + + +def _wait_for(path: Path) -> None: + deadline = time.monotonic() + 5 + while not path.exists(): + if time.monotonic() > deadline: + pytest.fail(f"worker did not create {path.name}") + time.sleep(0.01) + + +def _effect(path: Path) -> TaskResult: + with path.open("a") as stream: + stream.write("executed\n") + return TaskResult(("universe", "output"), "ok", notes=(get_worker().address,)) + + +def _wait_for_release(started: Path, release: Path) -> str: + started.touch() + deadline = time.monotonic() + 5 + while not release.exists(): + if time.monotonic() > deadline: + raise AssertionError("test did not release its running task") + time.sleep(0.01) + return "original" + + +def _cooperate(started: Path, stopped: Path) -> None: + started.touch() + deadline = time.monotonic() + 5 + while not execution.cancelled(): + if time.monotonic() > deadline: + raise AssertionError("task did not observe its invocation's cancellation") + time.sleep(0.01) + stopped.touch() + execution.check_cancelled() + + +def _cooperating_effect(started: Path, stopped: Path) -> None: + with started.open("a") as stream: + stream.write(get_worker().address + "\n") + _cooperate(started, stopped) + + +def _remove_worker(client: Client, address: str) -> None: + worker = next(worker for worker in client.cluster.workers.values() + if worker.address == address) + # close(), unlike close_gracefully(), discards this worker's task data. + # Keeping its executor alive also models a partitioned task that can still + # write even though the scheduler has reassigned its Dask key elsewhere. + client.cluster.sync(worker.close, executor_wait=False, timeout=0.5) + + +def _raise(error: Exception) -> None: + raise error + + +def _records(*, dask_scheduler: Any) -> dict[str, Any]: + return dask_scheduler.extensions.get("lightcone-executions", {}) + + +def _lose_state(*, dask_scheduler: Any) -> None: + dask_scheduler.extensions.pop("lightcone-executions", None) + + +def test_completed_task_replay_on_another_worker_returns_its_original_receipt( + execution_client: Client, tmp_path: Path, +) -> None: + effects = tmp_path / "effects" + with execution.invocation(execution_client) as run: + assert not run.stopped + original = run.submit(_effect, effects, key="recipe", resources=_RESOURCES).result() + other = next(address for address in execution_client.scheduler_info()["workers"] + if address != original.notes[0]) + replay = execution_client.submit( + execution._call, run.id, "recipe", _effect, effects, + key=f"replay-{uuid4().hex}", workers=[other], allow_other_workers=False, + pure=False, resources=_RESOURCES, + ).result() + assert replay == original + assert effects.read_text() == "executed\n" + assert run.stopped + assert run.id not in execution_client.run_on_scheduler(_records) + + +def test_worker_loss_recomputes_the_dask_future_without_repeating_completed_effects( + execution_client: Client, tmp_path: Path, +) -> None: + effects = tmp_path / "effects" + with execution.invocation(execution_client) as run: + future = run.submit(_effect, effects, key="recipe", resources=_RESOURCES) + original = future.result(timeout=3) + lost_worker = original.notes[0] + _remove_worker(execution_client, lost_worker) + deadline = time.monotonic() + 3 + while True: + holders = execution_client.who_has([future])[future.key] + if holders and lost_worker not in holders: + break + assert time.monotonic() < deadline, "Dask did not recompute the lost result" + time.sleep(0.01) + assert future.result(timeout=3) == original + assert effects.read_text() == "executed\n" + + +def test_worker_loss_cannot_replay_effects_while_the_original_execution_still_runs( + execution_client: Client, tmp_path: Path, +) -> None: + started, stopped = tmp_path / "started", tmp_path / "stopped" + with execution.invocation(execution_client) as run: + future = run.submit( + _cooperating_effect, started, stopped, key="recipe", resources=_RESOURCES, + ) + _wait_for(started) + lost_worker = started.read_text().strip() + _remove_worker(execution_client, lost_worker) + with pytest.raises(execution.ExecutionUncertain, match="previous attempt"): + future.result(timeout=3) + _wait_for(stopped) + assert started.read_text().splitlines() == [lost_worker] + assert run.id not in execution_client.run_on_scheduler(_records) + + +def test_duplicate_running_attempt_cannot_execute_or_finish_the_original_claim( + execution_client: Client, tmp_path: Path, +) -> None: + started, release, duplicate_effect = ( + tmp_path / "started", tmp_path / "release", tmp_path / "duplicate" + ) + with execution.invocation(execution_client) as run: + original = run.submit( + _wait_for_release, started, release, key="recipe", resources=_RESOURCES, + ) + try: + _wait_for(started) + duplicate = execution_client.submit( + execution._call, run.id, "recipe", _effect, duplicate_effect, + key=f"duplicate-{uuid4().hex}", pure=False, resources=_RESOURCES, + ) + with pytest.raises(execution.ExecutionUncertain, match="previous attempt"): + duplicate.result(timeout=3) + assert execution._rpc(execution_client, run.id, "pending") == ["recipe"] + assert not duplicate_effect.exists() + finally: + release.touch() + try: + assert original.result(timeout=3) == "original" + except execution.ExecutionCancelled: + # Rejecting the ambiguous duplicate may revoke the invocation before + # the original completes. Only that original may acknowledge its stop. + pass + assert execution._rpc(execution_client, run.id, "pending") == [] + + +def test_late_dispatch_after_invocation_exit_cannot_recreate_authorization( + execution_client: Client, tmp_path: Path, +) -> None: + effects = tmp_path / "effects" + with execution.invocation(execution_client) as run: + pass + late = execution_client.submit( + execution._call, run.id, "late", _effect, effects, pure=False, + ) + with pytest.raises(execution.ExecutionUncertain, match="no longer registered"): + late.result(timeout=3) + assert not effects.exists() + assert run.id not in execution_client.run_on_scheduler(_records) + + +def test_missing_scheduler_state_refuses_both_replay_and_unstarted_tasks( + execution_client: Client, tmp_path: Path, +) -> None: + effects = tmp_path / "effects" + with pytest.raises(execution.ExecutionUncertain, match="confirm execution stopped"): + with execution.invocation(execution_client) as run: + run.submit(_effect, effects, key="recipe", resources=_RESOURCES).result() + execution_client.run_on_scheduler(_lose_state) + for key in ("recipe", "new-recipe"): + future = execution_client.submit( + execution._call, run.id, key, _effect, effects, + key=f"lost-state-{uuid4().hex}", pure=False, + ) + with pytest.raises(execution.ExecutionUncertain, match="no longer registered"): + future.result(timeout=3) + assert effects.read_text() == "executed\n" + assert not run.stopped + + +def test_missing_scheduler_state_stops_running_work_without_claiming_confirmed_cleanup( + execution_client: Client, tmp_path: Path, +) -> None: + started, stopped = tmp_path / "started", tmp_path / "stopped" + with pytest.raises(execution.ExecutionUncertain, match="confirm execution stopped"): + with execution.invocation(execution_client) as run: + future = run.submit( + _cooperate, started, stopped, key="recipe", resources=_RESOURCES, + ) + _wait_for(started) + execution_client.run_on_scheduler(_lose_state) + with pytest.raises(execution.ExecutionCancelled, match="cancelled"): + future.result(timeout=3) + assert stopped.exists() + + +def test_lost_claim_response_never_starts_the_recipe_and_retains_its_unresolved_claim( + execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + effects = tmp_path / "effects" + request = execution._rpc + + def lose_claim_response(*args: Any, **kwargs: Any) -> Any: + result = request(*args, **kwargs) + if args[2] == "claim": + raise TimeoutError("claim accepted but reply lost") + return result + + monkeypatch.setattr(execution, "_rpc", lose_claim_response) + monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.1) + with pytest.raises(execution.ExecutionUncertain, match="unconfirmed tasks: recipe"): + with execution.invocation(execution_client) as run: + future = run.submit(_effect, effects, key="recipe", resources=_RESOURCES) + with pytest.raises(TimeoutError, match="reply lost"): + future.result(timeout=3) + assert execution._rpc(execution_client, run.id, "pending") == ["recipe"] + assert not effects.exists() + + +@pytest.mark.parametrize("loss", ["lease", "client"]) +def test_expired_or_disconnected_invocations_cannot_be_revived_by_heartbeat(loss: str) -> None: + scheduler = SimpleNamespace(extensions={}, clients={"driver": object()}) + execution._state("invocation", "register", value="driver", dask_scheduler=scheduler) + record = scheduler.extensions["lightcone-executions"]["invocation"] + if loss == "lease": + record["deadline"] = 0 + else: + scheduler.clients.clear() + assert not execution._state("invocation", "heartbeat", dask_scheduler=scheduler) + # Neither a fresh connection nor a late heartbeat can resurrect permission. + scheduler.clients["driver"] = object() + record["deadline"] = time.monotonic() + 60 + assert not execution._state("invocation", "heartbeat", dask_scheduler=scheduler) + with pytest.raises(execution.ExecutionCancelled, match="no longer active"): + execution._state("invocation", "claim", "recipe", dask_scheduler=scheduler) + + +def test_invocation_exit_cancels_the_future_and_waits_for_task_cooperation( + execution_client: Client, tmp_path: Path, +) -> None: + started, stopped = tmp_path / "started", tmp_path / "stopped" + with execution.invocation(execution_client) as run: + future = run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + _wait_for(started) + assert future.cancelled() + assert stopped.exists() + assert run.stopped + assert run.id not in execution_client.run_on_scheduler(_records) + + +def test_confirmed_cleanup_survives_an_error_in_the_invoking_driver( + execution_client: Client, tmp_path: Path, +) -> None: + started, stopped = tmp_path / "started", tmp_path / "stopped" + with pytest.raises(ValueError, match="driver failed to commit"): + with execution.invocation(execution_client) as run: + run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + _wait_for(started) + raise ValueError("driver failed to commit") + assert stopped.exists() + assert run.stopped + + +def test_driver_disconnect_revokes_execution_even_while_an_observer_remains_connected( + execution_client: Client, tmp_path: Path, +) -> None: + started, stopped = tmp_path / "started", tmp_path / "stopped" + with Client(execution_client.scheduler.address, set_as_default=False) as owner: + run = execution.Invocation(owner) + execution._rpc(owner, run.id, "register", value=owner.id) + run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + _wait_for(started) + _wait_for(stopped) + assert not execution._rpc(execution_client, run.id, "heartbeat") + deadline = time.monotonic() + 3 + while execution._rpc(execution_client, run.id, "pending"): + assert time.monotonic() < deadline, "task never acknowledged that it stopped" + time.sleep(0.01) + execution._rpc(execution_client, run.id, "forget") + + +def test_a_returned_failed_recipe_result_is_cached_without_rerunning( + execution_client: Client, +) -> None: + failed = TaskResult(("universe", "output"), "failed", reason="recipe exited 2") + with execution.invocation(execution_client) as run: + assert run.submit(lambda: failed, key="recipe", resources=_RESOURCES).result() == failed + replay = execution_client.submit( + execution._call, run.id, "recipe", _raise, AssertionError("must not execute"), + pure=False, + ) + assert replay.result(timeout=3) == failed + + +def test_function_exception_confirms_stop_but_does_not_authorize_reexecution( + execution_client: Client, +) -> None: + with execution.invocation(execution_client) as run: + future = run.submit(_raise, ValueError("recipe failed"), key="recipe", resources=_RESOURCES) + with pytest.raises(ValueError, match="recipe failed"): + future.result(timeout=3) + assert execution._rpc(execution_client, run.id, "pending") == [] + replay = execution_client.submit( + execution._call, run.id, "recipe", _raise, AssertionError("must not execute"), + pure=False, + ) + with pytest.raises(execution.ExecutionUncertain, match="previous attempt"): + replay.result(timeout=3) + + +def test_uncertain_task_remains_unresolved_after_context_exit( + execution_client: Client, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.1) + with pytest.raises(execution.ExecutionUncertain, match="unconfirmed tasks: recipe"): + with execution.invocation(execution_client) as run: + future = run.submit( + _raise, execution.ExecutionUncertain("container may still be alive"), + key="recipe", resources=_RESOURCES, + ) + with pytest.raises(execution.ExecutionUncertain, match="container may still"): + future.result(timeout=3) + assert execution._rpc(execution_client, run.id, "pending") == ["recipe"] + assert execution._rpc(execution_client, run.id, "pending") == ["recipe"] + assert not execution._rpc(execution_client, run.id, "heartbeat") + assert not run.stopped + + +def test_uncertain_cleanup_prevents_independent_tasks_from_using_released_dask_resources( + execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.1) + effects = tmp_path / "effects" + with pytest.raises(execution.ExecutionUncertain, match="unconfirmed tasks: first"): + with execution.invocation(execution_client) as run: + first = run.submit( + _raise, execution.ExecutionUncertain("container may still consume memory"), + key="first", resources=_RESOURCES, + ) + with pytest.raises(execution.ExecutionUncertain, match="container may still"): + first.result(timeout=3) + # Dask has released the first task's reservations, but its external + # work may survive. Authorization must close before another task runs. + later = run.submit(_effect, effects, key="later", resources=_RESOURCES) + error = later.exception(timeout=3) + assert isinstance(error, execution.ExecutionCancelled) + assert not effects.exists() diff --git a/tests/test_execution_processes.py b/tests/test_execution_processes.py new file mode 100644 index 00000000..685838c5 --- /dev/null +++ b/tests/test_execution_processes.py @@ -0,0 +1,229 @@ +"""Real command ownership: timeout, interruption, background children and worker loss.""" + +from __future__ import annotations + +import os +import signal +import subprocess +import sys +import time +from pathlib import Path + +import psutil +import pytest + +from lightcone.engine.execution import ExecutionCancelled, ExecutionUncertain +from lightcone.engine.sandbox import Policy, Unavailable, run +from lightcone.engine.sandbox.processes import Command + + +def _policy(root: Path) -> Policy: + return Policy(read=(root,), write=(root,), execute=(), tmp_home=root) + + +def _gone(pid: int) -> bool: + try: + return psutil.Process(pid).status() == psutil.STATUS_ZOMBIE + except psutil.NoSuchProcess: + return True + + +def test_timeout_escalates_ignoring_command(tmp_path: Path) -> None: + pidfile = tmp_path / "pid" + outcome = run( + Unavailable(), _policy(tmp_path), + [sys.executable, "-c", ( + "import os,signal,time; from pathlib import Path; " + "signal.signal(signal.SIGTERM, signal.SIG_IGN); " + f"Path({str(pidfile)!r}).write_text(str(os.getpid())); time.sleep(60)" + )], cwd=tmp_path, env=dict(os.environ), timeout=0.2, + ) + assert outcome.returncode == 124 + assert any("timed out" in note for note in outcome.notes) + assert _gone(int(pidfile.read_text())) + + +def test_cancel_stops_only_its_command(tmp_path: Path) -> None: + unrelated = subprocess.Popen([sys.executable, "-c", "import time; time.sleep(60)"]) + started = time.monotonic() + try: + with pytest.raises(ExecutionCancelled, match="processes have stopped"): + run( + Unavailable(), _policy(tmp_path), + [sys.executable, "-c", "import time; time.sleep(60)"], + cwd=tmp_path, env=dict(os.environ), + cancelled=lambda: time.monotonic() - started > 0.2, + ) + assert unrelated.poll() is None + finally: + unrelated.kill() + unrelated.wait() + + +def test_successful_leader_cannot_leave_a_background_writer(tmp_path: Path) -> None: + pidfile = tmp_path / "child" + child = ( + "import os,signal,time; from pathlib import Path; " + "signal.signal(signal.SIGTERM, signal.SIG_IGN); " + f"Path({str(pidfile)!r}).write_text(str(os.getpid())); time.sleep(60)" + ) + outcome = run( + Unavailable(), _policy(tmp_path), + [sys.executable, "-c", ( + "import subprocess,sys,time; from pathlib import Path; " + f"subprocess.Popen([sys.executable, '-c', {child!r}]); " + f"\nwhile not Path({str(pidfile)!r}).exists(): time.sleep(.01)" + )], cwd=tmp_path, env=dict(os.environ), + ) + assert outcome.returncode == 1 + assert any("background processes" in note for note in outcome.notes) + assert _gone(int(pidfile.read_text())) + + +def test_worker_sigkill_closes_custody_pipe_and_stops_child(tmp_path: Path) -> None: + pidfile = tmp_path / "child" + command = ( + "import os,signal,time; from pathlib import Path; " + "signal.signal(signal.SIGTERM, signal.SIG_IGN); " + f"Path({str(pidfile)!r}).write_text(str(os.getpid())); time.sleep(60)" + ) + worker = subprocess.Popen( + [sys.executable, "-c", ( + "import os,sys; from pathlib import Path; " + "from lightcone.engine.sandbox.processes import Command; " + f"c=Command([sys.executable, '-c', {command!r}], cwd=Path({str(tmp_path)!r}), " + "env=dict(os.environ), capture=False); c.wait()" + )], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + ) + pid = None + try: + deadline = time.monotonic() + 5 + while not pidfile.exists(): + assert worker.poll() is None + assert time.monotonic() < deadline + time.sleep(0.02) + pid = int(pidfile.read_text()) + worker.kill() + worker.wait() + while not _gone(pid): + assert time.monotonic() < deadline + time.sleep(0.02) + finally: + if worker.poll() is None: + worker.kill() + worker.wait() + if pid is not None and not _gone(pid): + os.kill(pid, signal.SIGKILL) + + +def test_stdout_bytes_are_unchanged_by_custody(tmp_path: Path) -> None: + received: list[bytes] = [] + outcome = run( + Unavailable(), _policy(tmp_path), + [sys.executable, "-c", "import os; os.write(1, b'\\xff\\r\\n')"], + cwd=tmp_path, env=dict(os.environ), + output=lambda stream, value: received.append(value) if stream == "stdout" else None, + ) + assert outcome.returncode == 0 + assert b"".join(received) == b"\xff\r\n" + + +def test_container_timeout_uses_its_immutable_runtime_id(tmp_path: Path) -> None: + runtime = tmp_path / "runtime" + runtime.write_text(f"#!{sys.executable}\n" + ''' +import json, os, signal, subprocess, sys +from pathlib import Path +import psutil +root = Path(os.environ['STATE_ROOT']) +identity = 'a' * 64 +argv = sys.argv[1:] +with (root / 'calls').open('a') as log: + log.write(json.dumps(argv) + '\\n') +if argv[0] == 'run': + process = subprocess.Popen([sys.executable, '-c', + 'import signal,time; signal.signal(signal.SIGTERM, signal.SIG_IGN); time.sleep(60)'], + start_new_session=True) + (root / 'payload').write_text(str(process.pid)) + Path(argv[argv.index('--cidfile') + 1]).write_text(identity) + (root / 'cidfile').write_text(argv[argv.index('--cidfile') + 1]) + process.wait() +elif argv[0] == 'inspect': + try: + alive = psutil.Process(int((root / 'payload').read_text())).status() != psutil.STATUS_ZOMBIE + except psutil.NoSuchProcess: + alive = False + print('true' if alive else 'false') +elif argv[0] == 'kill': + os.kill(int((root / 'payload').read_text()), signal.SIGKILL) +elif argv[0] == 'rm': + Path((root / 'cidfile').read_text()).unlink() +''') + runtime.chmod(0o700) + command = Command( + [str(runtime), "run", "image"], cwd=tmp_path, + env={**os.environ, "STATE_ROOT": str(tmp_path)}, capture=False, + container=True, timeout=0.4, + ) + try: + code, note = command.wait() + assert code == 124 + assert "timed out" in note + assert _gone(int((tmp_path / "payload").read_text())) + import json + + calls = [json.loads(line) for line in (tmp_path / "calls").read_text().splitlines()] + assert ["stop", "--time", "1", "a" * 64] in calls + assert ["kill", "a" * 64] in calls + assert ["rm", "a" * 64] in calls + finally: + path = tmp_path / "payload" + if path.exists() and not _gone(int(path.read_text())): + os.kill(int(path.read_text()), signal.SIGKILL) + + +def test_killed_custodian_reports_uncertainty(tmp_path: Path) -> None: + command = Command( + [sys.executable, "-c", "pass"], cwd=tmp_path, + env=dict(os.environ), capture=False, + ) + command.process.kill() + with pytest.raises(ExecutionUncertain, match="without confirming"): + command.wait() + + +def test_uncertain_cleanup_retains_sandbox_home(tmp_path: Path) -> None: + from lightcone.engine.sandbox import scope + + home = tmp_path / "home" + home.mkdir() + policy = Policy(read=(), write=(), execute=(), tmp_home=home) + with pytest.raises(ExecutionUncertain): + with scope(policy): + raise ExecutionUncertain("worker lost") + assert home.is_dir() + + +def test_stream_setup_failure_stops_command_before_propagating( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + from lightcone.engine.sandbox import boundary + + pidfile = tmp_path / "pid" + + def fail(_self: object) -> None: + deadline = time.monotonic() + 5 + while not pidfile.exists(): + assert time.monotonic() < deadline + time.sleep(0.01) + raise RuntimeError("cannot start stream thread") + + monkeypatch.setattr(boundary._Tail, "start", fail) + with pytest.raises(RuntimeError, match="stream thread"): + run( + Unavailable(), _policy(tmp_path), + [sys.executable, "-c", ( + "import os,time; from pathlib import Path; " + f"Path({str(pidfile)!r}).write_text(str(os.getpid())); time.sleep(60)" + )], cwd=tmp_path, env=dict(os.environ), + ) + assert _gone(int(pidfile.read_text())) diff --git a/tests/test_execution_resources.py b/tests/test_execution_resources.py new file mode 100644 index 00000000..c73f7a37 --- /dev/null +++ b/tests/test_execution_resources.py @@ -0,0 +1,129 @@ +"""Parse recipe requests and refuse impossible execution before submitting work.""" + +from __future__ import annotations + +from typing import Any + +import pytest +from pydantic import ValidationError + +from lightcone.engine.execution_resources import TaskResources +from lightcone.engine.project import ProjectError + +GIB = 1024**3 + + +def _workers(*capacities: tuple[int, int]) -> dict[str, Any]: + return { + f"worker-{index}": {"resources": {"CPU": cpus, "MEMORY": memory}} + for index, (cpus, memory) in enumerate(capacities) + } + + +@pytest.mark.parametrize( + ("memory", "expected"), + [("512Mi", 512 * 1024**2), ("1.5GiB", 3 * GIB // 2), ("8GB", 8_000_000_000), + ("1B", 1), ("2 Ti", 2 * 1024**4), ("1000kB", 1_000_000)], +) +def test_memory_units_have_explicit_decimal_or_binary_meaning(memory: str, expected: int) -> None: + assert TaskResources.parse({"memory": memory}).memory_bytes == expected + + +@pytest.mark.parametrize( + ("duration", "expected"), + [("1h30m", 5400), ("30m", 1800), ("2d3h4m5s", 183845), ("45s", 45)], +) +def test_task_walltime_accepts_compound_durations(duration: str, expected: int) -> None: + assert TaskResources.parse({"time_limit": duration}).time_seconds == expected + + +def test_integral_astra_float_cpu_count_is_accepted_without_rounding() -> None: + assert TaskResources.parse({"cpus": 4.0}).cpus == 4 + with pytest.raises(ProjectError, match="fractional CPUs"): + TaskResources.parse({"cpus": 0.5}) + + +@pytest.mark.parametrize( + "declaration", + [ + {"cpus": 0}, {"cpus": True}, {"cpus": "4"}, {"cpus": None}, + {"memory": "0Gi"}, {"memory": "0.1B"}, {"memory": "16"}, {"memory": 16}, + {"memory": "400m"}, + {"time_limit": "0m"}, {"time_limit": ""}, {"time_limit": "5m2h"}, + {"time_limit": "unlimited"}, {"gpus": 1}, {"disk": "10Gi"}, {"ram": "1Gi"}, + ], +) +def test_invalid_or_unhonored_declarations_are_not_silently_ignored( + declaration: dict[str, Any], +) -> None: + with pytest.raises(ProjectError): + TaskResources.parse(declaration) + + +def test_internal_resource_models_remain_validated() -> None: + with pytest.raises(ValidationError): + TaskResources(memory_bytes=-1) + with pytest.raises(ValidationError): + TaskResources(time_seconds=0) + + +def test_declared_requests_reserve_exact_cpu_and_memory_budgets() -> None: + task = TaskResources.parse({"cpus": 4, "memory": "6Gi"}) + assert task.requirements(_workers((8, 16 * GIB))) == {"CPU": 4, "MEMORY": 6 * GIB} + + +def test_missing_memory_reserves_entire_worker_instead_of_guessing() -> None: + assert TaskResources().requirements(_workers((8, 16 * GIB), (8, 16 * GIB))) == { + "CPU": 1, "MEMORY": 16 * GIB, + } + + +def test_probe_reserves_an_entire_worker() -> None: + assert TaskResources().requirements(_workers((8, 16 * GIB)), whole_worker=True) == { + "CPU": 8, "MEMORY": 16 * GIB, + } + + +def test_cpu_count_is_independent_of_dask_execution_threads() -> None: + workers = _workers((8, 16 * GIB)) + workers["worker-0"]["nthreads"] = 1 + assert TaskResources(cpus=8).requirements(workers)["CPU"] == 8 + + +def test_a_task_must_fit_one_worker_not_the_sum_of_the_cluster() -> None: + with pytest.raises(ProjectError, match="on one worker"): + TaskResources(cpus=8, memory_bytes=20 * GIB).requirements( + _workers((4, 16 * GIB), (4, 16 * GIB)) + ) + + +def test_cpu_and_memory_must_fit_on_the_same_worker() -> None: + with pytest.raises(ProjectError, match="no worker"): + TaskResources(cpus=8, memory_bytes=16 * GIB).requirements( + _workers((8, 4 * GIB), (4, 16 * GIB)) + ) + + +def test_explicit_requests_can_select_a_fitting_worker() -> None: + assert TaskResources(cpus=8, memory_bytes=8 * GIB).requirements( + _workers((4, 4 * GIB), (8, 16 * GIB)) + ) == {"CPU": 8, "MEMORY": 8 * GIB} + + +@pytest.mark.parametrize("whole_worker", [False, True]) +def test_missing_budgets_are_not_guessed_for_heterogeneous_workers(whole_worker: bool) -> None: + with pytest.raises(ProjectError, match="identical"): + TaskResources().requirements( + _workers((4, 4 * GIB), (8, 16 * GIB)), whole_worker=whole_worker + ) + + +@pytest.mark.parametrize( + "workers", + [{}, {"worker": {}}, {"worker": {"resources": {"CPU": 1}}}, + {"worker": {"resources": {"CPU": True, "MEMORY": GIB}}}, + {"worker": {"resources": {"CPU": 1, "MEMORY": float("nan")}}}], +) +def test_absent_or_unknown_worker_capacity_refuses_execution(workers: dict[str, Any]) -> None: + with pytest.raises(ProjectError): + TaskResources().requirements(workers) diff --git a/tests/test_materialize.py b/tests/test_materialize.py index c9b9a29b..a86b5dee 100644 --- a/tests/test_materialize.py +++ b/tests/test_materialize.py @@ -440,7 +440,7 @@ def unexpected(*args: object) -> None: @pytest.mark.parametrize("failing", ["baseline/first", "baseline/second"]) -def test_a_failed_commit_says_whether_remote_tasks_may_still_run( +def test_a_failed_commit_restores_unconsumed_outputs_after_confirmed_stop( root: Path, monkeypatch: pytest.MonkeyPatch, failing: str, ) -> None: _cluster(monkeypatch, _Inline()) @@ -455,8 +455,10 @@ def fail(root: Path, paths: list[Path], message: str) -> None: monkeypatch.setattr(dataset, "save", fail) with pytest.raises(ProjectError, match="git commit failed") as raised: engine.materialize(root, [], cluster_id=CLUSTER_ID) - # Only the first output leaves another one outstanding. - assert ("did not stop the allocation" in str(raised.value)) == (failing == "baseline/first") + assert str(raised.value) == "git commit failed" + assert not dataset.status(root) + assert (root / "results/baseline/first.txt").exists() == (failing == "baseline/second") + assert not (root / "results/baseline/second.txt").exists() def test_shared_inputs_are_hashed_once_before_task_serialization( analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, @@ -729,27 +731,42 @@ def test_a_rebuild_that_fails_puts_the_previous_output_back( assert not dataset.status(root) -def test_an_interrupted_run_retains_what_never_reported( - root: Path, inline: None, monkeypatch: pytest.MonkeyPatch +@pytest.mark.parametrize("failure", ["confirmed", "uncertain", "masked", "second_interrupt"]) +def test_interrupted_outputs_are_restored_only_after_confirmed_stop( + root: Path, inline: None, monkeypatch: pytest.MonkeyPatch, failure: str, ) -> None: - """A sibling that already saved keeps its commit; the output still in - flight may still have a writer, so its partial files are retained.""" + """A completed sibling keeps its commit; unconfirmed writers retain their files.""" + from lightcone.engine.execution import ExecutionUncertain + + uncertain = failure != "confirmed" + engine.materialize(root, [], cluster_id=CLUSTER_ID) (root / "astra.yaml").write_text(_SPEC.replace("echo {decisions.method}", "echo changed")) dataset.save(root, [root], "edit both recipes") class _Interrupted(_Inline): + stopped = not uncertain + def completed(self, handles: list[Any]) -> Iterator[TaskResult]: yield handles[0] + if failure == "masked": + raise RuntimeError("client close masked uncertain cleanup") + if failure == "uncertain": + raise ExecutionUncertain("worker disappeared without acknowledging stop") raise KeyboardInterrupt _cluster(monkeypatch, _Interrupted()) - with pytest.raises(KeyboardInterrupt): + expected = {"masked": RuntimeError, "uncertain": ExecutionUncertain}.get( + failure, KeyboardInterrupt, + ) + with pytest.raises(expected): engine.materialize(root, [], cluster_id=CLUSTER_ID) - assert dataset.status(root) - assert (root / "results/baseline/second.txt").read_text() == "changed\n" + assert bool(dataset.status(root)) is uncertain + assert (root / "results/baseline/second.txt").read_text() == ( + "changed\n" if uncertain else "alpha\n" + ) # ---- the commit message ---------------------------------------------------- @@ -1060,6 +1077,104 @@ def test_the_recorded_command_holds_on_a_fresh_clone( # ---- the scheduler seam ---------------------------------------------------- +def _resource_cluster(monkeypatch: pytest.MonkeyPatch, *, workers: int = 1) -> None: + from distributed import Client, LocalCluster + + from lightcone.engine import compute + + @contextmanager + def connect(cluster_id: str) -> Iterator[Any]: + with LocalCluster( + n_workers=workers, threads_per_worker=4, processes=False, + dashboard_address=None, resources={"CPU": 4, "MEMORY": 2 * 1024**3}, + ) as cluster, Client(cluster, set_as_default=False) as client: + yield client + + monkeypatch.setattr(compute, "connect", connect) + + +@pytest.mark.parametrize("resource_spec", ["cpus: 5", "memory: 3Gi"]) +def test_resource_refusal_precedes_preparation_and_all_recipes( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, resource_spec: str, +) -> None: + spec = _SPEC.replace( + "command: cat", f"resources: {{{resource_spec}}}\n command: cat" + ) + root = analysis(spec, universes={"baseline": _UNIVERSE}) + before = dataset.head(root) + _resource_cluster(monkeypatch) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("preparation began before all resource requests were validated") + + monkeypatch.setattr(engine, "_fetch_inputs", unexpected) + monkeypatch.setattr(engine.container, "runtime_for_run", unexpected) + with pytest.raises(ProjectError, match="baseline/second:.*no worker"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert dataset.head(root) == before + assert not dataset.status(root) + assert not (root / "results/baseline/first.txt").exists() + + +def test_empty_cluster_refuses_before_project_preparation( + root: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + _resource_cluster(monkeypatch, workers=0) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("preparation began without an available worker") + + monkeypatch.setattr(engine, "_fetch_inputs", unexpected) + with pytest.raises(ProjectError, match="no workers"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert not dataset.status(root) + + +@pytest.mark.parametrize( + ("resource_spec", "expected_parallelism"), + [("cpus: 3, memory: 256Mi", 1), ("cpus: 1, memory: 1Gi", 2)], +) +def test_real_dask_respects_recipe_cpu_and_memory_reservations( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, + resource_spec: str, expected_parallelism: int, +) -> None: + # Four Dask threads would run all four subprocesses together without + # resource reservations. Each independent output records its live interval. + spec = 'version: "0.0.13"\nname: analysis\ninputs: []\noutputs:\n' + "".join( + f" - id: task{index}\n" + " type: metric\n" + " format: json\n" + " recipe:\n" + f" resources: {{{resource_spec}}}\n" + " command: python src/work.py {output}\n" + for index in range(4) + ) + root = analysis(spec, files={"src/work.py": """ + import json + import sys + import time + from pathlib import Path + start = time.monotonic() + time.sleep(0.5) + Path(sys.argv[1]).write_text(json.dumps([start, time.monotonic()])) + """}) + _resource_cluster(monkeypatch) + + report = engine.materialize(root, [], cluster_id=CLUSTER_ID) + + assert report.ok and len(report.made) == 4 + events = [] + for path in (root / "results/baseline").glob("task*.json"): + start, finish = json.loads(path.read_text()) + events.extend([(start, 1), (finish, -1)]) + live = peak = 0 + for _, change in sorted(events): + live += change + peak = max(peak, live) + assert peak == expected_parallelism + assert not dataset.status(root) + + def test_a_real_cluster_still_fits_through_the_seam(root: Path, cluster_id: str) -> None: """The one test that starts Dask. The seam is only worth having if the thing it abstracts still goes through it.""" @@ -1084,7 +1199,8 @@ def test_a_processes_cluster_fits_through_the_seam( @contextmanager def processes(cluster_id: str) -> Iterator[Any]: with LocalCluster( - n_workers=2, threads_per_worker=1, processes=True, dashboard_address=None + n_workers=2, threads_per_worker=1, processes=True, dashboard_address=None, + resources={"CPU": 1, "MEMORY": 1024**3}, ) as cluster: with Client(cluster, set_as_default=False) as client: yield client @@ -1320,3 +1436,56 @@ def test_an_output_the_spec_dropped_is_excluded_and_named(root: Path, inline: No assert any(".second.manifest.json" in w for w in report.warnings) document = (root / "ro-crate-metadata.json").read_text() assert "results/baseline/second.txt" not in document + + +def test_recipe_time_limit_stops_writes_restores_output_and_keeps_cluster_usable( + root: Path, cluster_id: str, capsys: pytest.CaptureFixture[str], +) -> None: + import psutil + + engine.materialize(root, ["first"], cluster_id=cluster_id) + output = root / "results/baseline/first.txt" + manifest = root / "results/baseline/.first.manifest.json" + original_output, original_manifest = output.read_bytes(), manifest.read_bytes() + + script = root / "src/slow.py" + script.parent.mkdir(exist_ok=True) + script.write_text( + "import os, sys, time\n" + "from pathlib import Path\n" + "Path(sys.argv[1]).write_text('partial output')\n" + "print('slow-recipe-pid:', os.getpid(), flush=True)\n" + "time.sleep(30)\n" + "Path(sys.argv[1]).write_text('should never finish')\n" + ) + spec = root / "astra.yaml" + spec.write_text(spec.read_text().replace( + "command: echo {decisions.method} > {output}", + "resources: {time_limit: 1s}\n command: python src/slow.py {output}", + )) + dataset.save(root, [spec, script], "Run the recipe with a walltime limit") + capsys.readouterr() + + failed = engine.materialize(root, ["first"], cluster_id=cluster_id) + + assert failed.failed == ["baseline/first"] + assert any("timed out" in note for note in failed.notes) + pid_line = next( + line for line in capsys.readouterr().err.splitlines() + if line.startswith("slow-recipe-pid:") + ) + assert not psutil.pid_exists(int(pid_line.split(":", 1)[1])) + assert output.read_bytes() == original_output + assert manifest.read_bytes() == original_manifest + assert not dataset.status(root) + + # The fixture holds the same LocalCluster across both invocations; a + # successful new recipe demonstrates that timeout did not terminate it. + spec.write_text(spec.read_text().replace( + "python src/slow.py {output}", "echo recovered > {output}", + )) + dataset.save(root, [spec], "Use a recipe that completes within its limit") + recovered = engine.materialize(root, ["first"], cluster_id=cluster_id) + assert recovered.made == ["baseline/first"] + assert output.read_text() == "recovered\n" + assert not dataset.status(root) diff --git a/tests/test_plan.py b/tests/test_plan.py index 48a994b0..a182b9e1 100644 --- a/tests/test_plan.py +++ b/tests/test_plan.py @@ -122,6 +122,31 @@ def test_an_output_addresses_its_own_file(tmp_path: Path) -> None: assert "results/baseline/fit.json" in task.recipe +def test_recipe_resources_survive_graph_resolution(tmp_path: Path) -> None: + spec = _SPEC.replace( + "command: python src/fit.py", + "resources: {cpus: 4, memory: 6Gi, time_limit: 1h30m}\n" + " command: python src/fit.py", + ) + graph = _build(_project(tmp_path, spec)) + task = graph.tasks[("baseline", "fit")] + assert task.resources.cpus == 4 + assert task.resources.memory_bytes == 6 * 1024**3 + assert task.resources.time_seconds == 5400 + unspecified = graph.tasks[("baseline", "report")].resources + assert unspecified.cpus == 1 + assert unspecified.memory_bytes is None + + +def test_unhonored_recipe_resources_refuse_graph_construction(tmp_path: Path) -> None: + spec = _SPEC.replace( + "command: python src/fit.py", + "resources: {gpus: 1}\n command: python src/fit.py", + ) + with pytest.raises(ProjectError, match="unsupported recipe resource.*gpus"): + _build(_project(tmp_path, spec)) + + def test_a_declared_input_resolves_to_its_source(tmp_path: Path) -> None: task = _build(_project(tmp_path)).tasks[("baseline", "fit")] assert task.inputs == {"catalog": tmp_path / "data" / "catalog.fits"} @@ -281,4 +306,3 @@ def test_an_output_without_a_format_is_refused_by_name(tmp_path: Path) -> None: - diff --git a/tests/test_sandbox_oci.py b/tests/test_sandbox_oci.py index cf8cdccd..26e89b18 100644 --- a/tests/test_sandbox_oci.py +++ b/tests/test_sandbox_oci.py @@ -7,7 +7,6 @@ from __future__ import annotations -import subprocess from pathlib import Path from typing import Any @@ -221,7 +220,7 @@ def test_the_attestation_is_derived_from_the_flags(root: Path, policy: Policy) - class _Recorder: - """A Popen stand-in that records the argv and exits as told.""" + """A custodian stand-in recording the sandbox's fully wrapped argv.""" def __init__(self, returncode: int = 0) -> None: self.argv: list[str] | None = None @@ -234,13 +233,22 @@ def __call__(self, argv: list[str], **kwargs: Any) -> Any: class _Proc: import io - stderr = io.StringIO("") + stderr = io.BytesIO(b"") returncode = code - def wait(self) -> int: - return code + class _Command: + process = _Proc() - return _Proc() + def __enter__(self) -> _Command: + return self + + def __exit__(self, *args: Any) -> None: + pass + + def wait(self, cancelled: Any) -> tuple[int, str]: + return code, "" + + return _Command() def test_a_world_backend_takes_the_prefix_inside( @@ -250,7 +258,7 @@ def test_a_world_backend_takes_the_prefix_inside( is part of the world being entered, so it lands after the image in the argv rather than in front of the runtime.""" recorder = _Recorder() - monkeypatch.setattr(subprocess, "Popen", recorder) + monkeypatch.setattr(boundary, "Command", recorder) boundary.run( _backend(root), @@ -275,7 +283,7 @@ def test_a_host_backend_keeps_the_prefix_outside( """The existing composition, pinned: uv's config and caches are trusted plumbing outside a host mechanism's rewrite.""" recorder = _Recorder() - monkeypatch.setattr(subprocess, "Popen", recorder) + monkeypatch.setattr(boundary, "Command", recorder) boundary.run( Unavailable(), @@ -296,7 +304,7 @@ def test_exit_97_is_the_shims_only_under_landlock( """97 is the shim's reserved code, and there is no shim in a container — a recipe legitimately exiting 97 must not be told lc could not set up the sandbox.""" - monkeypatch.setattr(subprocess, "Popen", _Recorder(returncode=97)) + monkeypatch.setattr(boundary, "Command", _Recorder(returncode=97)) outcome = boundary.run(_backend(root), policy, ["true"], cwd=root, env={}) assert not any("could not set up" in note for note in outcome.notes) @@ -307,7 +315,7 @@ def test_exit_125_names_the_runtime_not_the_command( """The runtimes reserve 125 for their own failures — the command never ran, so neither the denial heuristics nor the trailer should point at it.""" - monkeypatch.setattr(subprocess, "Popen", _Recorder(returncode=125)) + monkeypatch.setattr(boundary, "Command", _Recorder(returncode=125)) outcome = boundary.run(_backend(root), policy, ["true"], cwd=root, env={}) assert any("runtime failed before the command ran" in note for note in outcome.notes) assert not any("ran under the lc sandbox" in note for note in outcome.notes) From 184dc0fee45983e36806e87bf9d842b84345aa40 Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Mon, 28 Sep 2026 10:39:53 -0700 Subject: [PATCH 2/8] Wait for worker address publication in loss test --- tests/test_execution.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/test_execution.py b/tests/test_execution.py index ec979b13..46d32f6c 100644 --- a/tests/test_execution.py +++ b/tests/test_execution.py @@ -68,6 +68,8 @@ def _cooperate(started: Path, stopped: Path) -> None: def _cooperating_effect(started: Path, stopped: Path) -> None: with started.open("a") as stream: stream.write(get_worker().address + "\n") + # Publish readiness after closing the append, never during file creation. + started.with_suffix(".ready").touch() _cooperate(started, stopped) @@ -140,7 +142,7 @@ def test_worker_loss_cannot_replay_effects_while_the_original_execution_still_ru future = run.submit( _cooperating_effect, started, stopped, key="recipe", resources=_RESOURCES, ) - _wait_for(started) + _wait_for(started.with_suffix(".ready")) lost_worker = started.read_text().strip() _remove_worker(execution_client, lost_worker) with pytest.raises(execution.ExecutionUncertain, match="previous attempt"): From c1a1b67f82c6efea8c93a3526aab226f30bf196f Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Mon, 28 Sep 2026 11:41:12 -0700 Subject: [PATCH 3/8] Recognize completed process groups on macOS --- src/lightcone/engine/sandbox/processes.py | 10 ++++++ tests/test_execution_processes.py | 42 +++++++++++++++++++++++ 2 files changed, 52 insertions(+) diff --git a/src/lightcone/engine/sandbox/processes.py b/src/lightcone/engine/sandbox/processes.py index 52d99c96..f8711135 100644 --- a/src/lightcone/engine/sandbox/processes.py +++ b/src/lightcone/engine/sandbox/processes.py @@ -64,10 +64,20 @@ def has_custodian(processes: Sequence[psutil.Process]) -> bool: def _drain(process: subprocess.Popen[bytes]) -> bool: """Stop the whole command group while its unreaped leader pins the group ID.""" for sig in (signal.SIGTERM, signal.SIGKILL): + if not members(group=process.pid): + process.wait() + return True try: os.killpg(process.pid, sig) except ProcessLookupError: pass + except PermissionError: + # Darwin reports EPERM when only zombies remain. Accept that race + # only after confirming no live group member still needs stopping. + if members(group=process.pid): + raise + process.wait() + return True deadline = time.monotonic() + _GRACE while members(group=process.pid): if time.monotonic() >= deadline: diff --git a/tests/test_execution_processes.py b/tests/test_execution_processes.py index 685838c5..4b28fed8 100644 --- a/tests/test_execution_processes.py +++ b/tests/test_execution_processes.py @@ -8,6 +8,7 @@ import sys import time from pathlib import Path +from unittest.mock import Mock import psutil import pytest @@ -17,6 +18,47 @@ from lightcone.engine.sandbox.processes import Command +def test_finished_group_is_reaped_without_signalling(monkeypatch: pytest.MonkeyPatch) -> None: + from lightcone.engine.sandbox import processes + + process = Mock(spec=subprocess.Popen, pid=1234) + signal_group = Mock(side_effect=PermissionError("no live signalable processes")) + monkeypatch.setattr(processes, "members", lambda **kwargs: []) + monkeypatch.setattr(processes.os, "killpg", signal_group) + assert processes._drain(process) + signal_group.assert_not_called() + process.wait.assert_called_once_with() + + +def test_permission_error_after_last_group_member_exits_is_safe( + monkeypatch: pytest.MonkeyPatch, +) -> None: + from lightcone.engine.sandbox import processes + + process = Mock(spec=subprocess.Popen, pid=1234) + monkeypatch.setattr(processes, "members", Mock(side_effect=[[object()], []])) + monkeypatch.setattr( + processes.os, "killpg", Mock(side_effect=PermissionError("no live signalable processes")), + ) + assert processes._drain(process) + process.wait.assert_called_once_with() + + +def test_permission_error_with_a_live_group_member_remains_an_error( + monkeypatch: pytest.MonkeyPatch, +) -> None: + from lightcone.engine.sandbox import processes + + process = Mock(spec=subprocess.Popen, pid=1234) + monkeypatch.setattr(processes, "members", lambda **kwargs: [object()]) + monkeypatch.setattr( + processes.os, "killpg", Mock(side_effect=PermissionError("not permitted")), + ) + with pytest.raises(PermissionError, match="not permitted"): + processes._drain(process) + process.wait.assert_not_called() + + def _policy(root: Path) -> Policy: return Policy(read=(root,), write=(root,), execute=(), tmp_home=root) From fa9129244c675e062b996d167ee24b01b54bdd9f Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Mon, 28 Sep 2026 16:01:16 -0700 Subject: [PATCH 4/8] Harden execution leases, cleanup, and resource admission --- CLAUDE.md | 6 +- docs/api/compute.md | 31 ++- docs/api/plan.md | 22 +- docs/api/sandbox.md | 14 +- docs/api/worker.md | 2 +- docs/cli/compute.md | 3 +- docs/user/cluster.md | 11 +- src/lightcone/cli/commands.py | 7 +- src/lightcone/cli/compute.py | 11 +- src/lightcone/engine/compute/__init__.py | 8 - src/lightcone/engine/compute/local.py | 29 +-- src/lightcone/engine/compute/local_runtime.py | 51 +++-- src/lightcone/engine/compute/model.py | 23 +- src/lightcone/engine/execution.py | 148 +++++++++---- src/lightcone/engine/execution_resources.py | 23 +- src/lightcone/engine/materialize.py | 30 +-- src/lightcone/engine/plan.py | 10 +- src/lightcone/engine/run.py | 4 +- src/lightcone/engine/sandbox/boundary.py | 19 +- src/lightcone/engine/sandbox/model.py | 10 + src/lightcone/engine/sandbox/oci.py | 4 + src/lightcone/engine/sandbox/processes.py | 99 ++++++--- src/lightcone/engine/units.py | 45 ++++ src/lightcone/engine/worker.py | 6 +- tests/conftest.py | 8 +- tests/test_cli.py | 6 +- tests/test_compute.py | 30 +++ tests/test_compute_local.py | 47 ++++ tests/test_execution.py | 209 ++++++++++++++++-- tests/test_execution_processes.py | 77 ++++++- tests/test_execution_resources.py | 9 +- tests/test_materialize.py | 28 ++- tests/test_plan.py | 33 +-- tests/test_run.py | 3 +- tests/test_sandbox_oci.py | 18 ++ 35 files changed, 819 insertions(+), 265 deletions(-) create mode 100644 src/lightcone/engine/units.py diff --git a/CLAUDE.md b/CLAUDE.md index e54bc81b..a7f8c95a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1786,8 +1786,10 @@ drains the invocation. Unconfirmed cleanup raises `ExecutionUncertain` and retai partial outputs; completing cleanup does not terminate the reusable allocation. **Recipe resources use standard Dask admission.** Preserve ASTRA `recipe.resources` -in `plan.Task` as validated `TaskResources`: whole CPUs, memory bytes, and optional -command walltime. Validate the whole selected graph before preparation or submission. +in `plan.Task` as raw mappings so `status` and `--check` remain independent of +executor support. Parse `TaskResources` at execution admission: whole CPUs, memory +bytes, and optional command walltime. Validate the whole selected graph before +preparation or submission, then pass reservations explicitly to submission. Workers advertise CPU/MEMORY; tasks reserve their declarations, with omitted RAM reserving a whole worker's memory and probes reserving both whole-worker budgets. Thread slots remain a separate concurrency cap. Reservations are cooperative, not diff --git a/docs/api/compute.md b/docs/api/compute.md index ea0dd865..0c5b667b 100644 --- a/docs/api/compute.md +++ b/docs/api/compute.md @@ -116,20 +116,25 @@ preparation and the existing task runtime/sandbox checks remain in their owners. Workers advertise standard Dask `CPU` and `MEMORY` resources; memory is measured in bytes. `engine.execution_resources.TaskResources` validates ASTRA's -`recipe.resources` into whole CPUs, bytes, and optional walltime seconds, and -`plan.Task` carries that request. `requirements(workers)` checks that one worker -can satisfy it and returns the resource dictionary used by `Client.submit`. +`recipe.resources` into whole CPUs, bytes, and optional walltime seconds at +execution admission. `plan.Task` preserves the ASTRA mapping so read-only +classification does not impose executor restrictions. `requirements(workers)` +checks that one worker can satisfy it and returns the resource dictionary used +by `Client.submit`. An omitted memory request reserves the full homogeneous worker budget; `whole_worker=True` reserves CPU and memory for a probe. Unsupported GPU/disk requests and fractional CPUs fail before execution. The materialize scheduler validates every selected task before preparation or submission, preventing earlier tasks from starting before a later impossible -request is discovered. Standard Dask scheduling accounts for concurrent CPU and -memory reservations; Dask execution-thread counts remain a separate concurrency -cap. Reservations do not impose hard limits on recipe subprocesses. Task walltime -uses the subprocess boundary's timeout and teardown, independent of the native -allocation's lifetime. +request is discovered, then passes each task's reservation explicitly to +submission. Allocation and task requests share byte and duration conversion +utilities; their models remain distinct because allocation selection supports +minimum quantities and node counts. Standard Dask scheduling accounts for +concurrent CPU and memory reservations; Dask execution-thread counts remain a +separate concurrency cap. Reservations do not impose hard limits on recipe +subprocesses. Task walltime uses the subprocess boundary's timeout and teardown, +independent of the native allocation's lifetime. `output.py` transports byte chunks through standard Dask events so detached workers' output reaches the invoking CLI. It uses the borrowed client's event @@ -145,14 +150,20 @@ receipts preserve the original result if Dask recomputes a lost result; a runnin or uncertain claim refuses replay and revokes the invocation. Missing state also refuses execution. This uses ordinary tasks and `run_on_scheduler`, without a custom worker, service, project lock, or persistent execution registry. +Driver heartbeats and worker authorization polls retry transient RPC failures +within the last confirmed 15-second lease. A failed RPC does not extend that +lease; explicit revocation, missing state, or expiry stops execution. On exit the invocation revokes admission, cancels pending futures, and waits for claimed tasks to acknowledge cleanup. Dask cancellation alone is insufficient: running tasks poll authorization and the subprocess boundary stops their commands. Only a positive `Invocation.stopped` flag permits restoring unconsumed outputs; an exception from closing another context cannot manufacture that confirmation. -Scheduler loss or an unacknowledged attempt raises `ExecutionUncertain` and retains -outputs. Receipts are removed after confirmed cleanup; uncertain records remain +Without positive completion evidence, scheduler loss or an unacknowledged attempt +raises `ExecutionUncertain` and retains outputs. Known terminal uncertainty is +reported immediately. A finished task already confirms command cleanup and receipt +publication, so metadata cleanup failures cannot discard its result. Receipts are +removed best-effort after confirmed cleanup; uncertain records remain until the allocation ends. They are not a recovery log for a later invocation. Local teardown drains the allocation's validated process session rather than diff --git a/docs/api/plan.md b/docs/api/plan.md index 9a67117a..abc1641e 100644 --- a/docs/api/plan.md +++ b/docs/api/plan.md @@ -4,8 +4,7 @@ The spec, read as a graph of tasks. `astra.yaml` × `universes/*.yaml` gives one task per `(universe, output)` pair that has a recipe; a task carries everything executing it needs — the rendered command, where its bytes go, what it reads, its decisions, its `definition_version`, and its -resource requirements — and -nothing about *how* it will be executed. +resource requirements — and nothing about *how* it will be executed. Source: `src/lightcone/engine/plan.py`. @@ -15,8 +14,7 @@ Source: `src/lightcone/engine/plan.py`. |---|---| | `build(root)` | Validate the spec with ASTRA's own validators, resolve every universe, return the `Graph`. | | `Graph` | Tasks keyed on `(universe_id, output_id)`; `order()` for the read-only topological walk, `resolve(targets)` for what a user typed, `closure(keys)` to narrow a run. | -| `Task` | One output in one universe, frozen, including its parsed `resources`. | -| `TaskResources` | Frozen Pydantic request from `engine.execution_resources`: positive whole CPUs, optional memory bytes, and optional walltime seconds. | +| `Task` | One output in one universe, frozen, retaining ASTRA's resource declaration in `resources`. | | `declared_path(root, path)` | The one rule that names a path: project-relative inside the tree, absolute outside, never resolved. | ## What must stay true @@ -34,11 +32,13 @@ Source: `src/lightcone/engine/plan.py`. resolution answers what a *valid* spec means and does not re-check that it is one. - **Resource declarations survive resolution.** `build` reads - `recipe.resources` from ASTRA's resolved output definition and validates - units and supported requirements through `TaskResources.parse`. It rejects - unsupported GPU/disk requests and fractional CPUs with the output's name. - Cluster capacity is checked later, before materialize prepares the project - or submits any task. No worker placement belongs in this module. + `recipe.resources` from ASTRA's resolved output definition and preserves the + mapping. A valid declaration remains readable by `status` and + `materialize --check` even when this executor cannot honor it. Execution + validates supported requirements through `TaskResources.parse` and checks + cluster capacity before materialize prepares the project or submits any + task. No worker placement or executor-specific resource validation belongs + in this module. - **The layout is flat and path-addressed.** `results//.`, and the path in a rendered recipe *is* the path on disk — no staging, no relocation. @@ -58,8 +58,8 @@ Source: `src/lightcone/engine/plan.py`. ## Tests `tests/test_plan.py` — pure; tests what lc *adds* (directories, edges, -versions, resource requirements, the validation gate), never what a spec means — that -coverage lives in astra-tools' own suite, and re-asserting it here +versions, resource preservation, the validation gate), never what a spec means — +that coverage lives in astra-tools' own suite, and re-asserting it here would recreate the second implementation this module deleted. Every fixture must be a spec `astra validate` accepts; the gate enforces it for free. diff --git a/docs/api/sandbox.md b/docs/api/sandbox.md index 40d5d2e9..7a706373 100644 --- a/docs/api/sandbox.md +++ b/docs/api/sandbox.md @@ -29,15 +29,21 @@ classification. Without a receiver, stdout remains inherited. `processes.Command` starts a small supervisor outside the sandbox. The supervisor owns the wrapped command's process group and watches a control pipe: worker death closes the pipe and triggers cleanup even when the worker cannot run `finally`. -Timeouts and cancellation use the same TERM/KILL cleanup. A command that exits -while leaving background processes is failed after those processes are stopped. +Timeouts include command startup. Timeouts and cancellation use the same TERM/KILL +cleanup, with one 15-second deadline shared by process and container operations. +After the leader exits, short-lived helpers get up to one second to exit naturally. +A command that still leaves background processes is failed after they are stopped. The unreaped leader pins the process-group ID until cleanup completes. -OCI commands use a private `--cidfile`. Cleanup inspects that immutable container -ID, stops or kills a running container, verifies it stopped, then removes it. +The OCI backend adds `--cidfile` to its command. Its native client runs in the +supervisor's private directory, while the payload's `--workdir` remains the project. +The boundary passes the OCI runtime explicitly; containing a prefix alone does +not imply container lifecycle behavior. Cleanup inspects the immutable container +ID, stops or kills a running container, verifies it stopped, then attempts removal. An absent or unverifiable identity during interruption is uncertain, not success. Only confirmed cleanup allows an ordinary result or `ExecutionCancelled`; unconfirmed cleanup raises `ExecutionUncertain` and retains the temporary home. +Both lifecycle exceptions are defined in `sandbox.model`, independently of Dask. Recipes must not detach into other process sessions. A hard-killed supervisor cannot guarantee cleanup of containers managed by an external runtime. diff --git a/docs/api/worker.md b/docs/api/worker.md index 5e943f04..3451b332 100644 --- a/docs/api/worker.md +++ b/docs/api/worker.md @@ -41,7 +41,7 @@ perform Dask resource admission. not enter the ordinary failed-output restore path: cleanup first establishes that writers have stopped, and uncertainty retains partial outputs. - **Task completion includes subprocess teardown.** The boundary owns process - and container cleanup, applies `task.resources.time_seconds`, and reports + and container cleanup, applies the parsed resource request's time limit, and reports uncertain teardown as an exception. A time limit that stops the recipe becomes an ordinary failed result. CPU and memory reservations are standard Dask scheduling constraints, not OS limits imposed by this module. diff --git a/docs/cli/compute.md b/docs/cli/compute.md index e4d697eb..d4b806c8 100644 --- a/docs/cli/compute.md +++ b/docs/cli/compute.md @@ -55,7 +55,8 @@ connections are unavailable. No name registry is maintained. CPU quantities are logical CPUs **per node**, memory is **GiB per node**, and `--num-nodes` defaults to one. Bare quantities are exact; `4+` means at least four. -Time accepts positive whole minutes or hours, such as `30m` or `2h`. Without +Time accepts positive durations with day/hour/minute/second units, such as `30m`, +`1h30m`, or `45s`. Without `--time`, the chosen offer's default applies. `fast` is a service class, not a queue-time promise. Limits apply to each allocation; aggregate quotas remain with the native backend. diff --git a/docs/user/cluster.md b/docs/user/cluster.md index aac69d9e..ccd2bcc7 100644 --- a/docs/user/cluster.md +++ b/docs/user/cluster.md @@ -143,7 +143,8 @@ list: Catalog errors identify the invalid field, for example `offers.0.resources.cpus`. Unknown common fields and duplicate YAML keys are rejected. CPU and node counts must be positive integers; memory is in GiB and may be fractional if it is an -exact number of bytes, and durations use minutes or hours such as `30m` or `2h`. +exact number of bytes. Durations use ordered day/hour/minute/second units, such as +`30m`, `1h30m`, or `45s`. Selection takes the first offer in catalog order that matches the request. An offer this host cannot provide is skipped: a local offer with more nodes, CPUs or @@ -341,7 +342,8 @@ are rejected rather than ignored. before fetching inputs, preparing the environment, or starting a recipe. This also validates currently complete outputs, which workers may need to rebuild after an upstream change. Use `lc materialize --check` to inspect currency -without allocation. +without allocation. Read-only `status` and `--check` accept valid ASTRA resource +declarations even when this executor cannot satisfy them. These are scheduling reservations, not per-recipe CPU or RAM enforcement. Recipes must respect their declarations; a subprocess can otherwise exceed @@ -382,8 +384,9 @@ before retrying. The allocation remains available after ordinary cancellation. The existing Dask scheduler holds invocation claims and completion receipts. After a worker disappears, a replacement task cannot rerun a recipe whose result is uncertain. A completed task returns its original receipt. Loss of the client -or its heartbeat revokes further work; a small command supervisor also stops the -command if its worker dies. There is no automatic recovery or replay after an +or expiry of its 15-second heartbeat lease revokes further work; brief RPC +failures are retried within the last confirmed lease. A small command supervisor +stops the command if its worker dies. There is no automatic recovery or replay after an uncertain execution, and no additional server or checkout state directory. Use one execution invocation per project at a time: there is no checkout lock diff --git a/src/lightcone/cli/commands.py b/src/lightcone/cli/commands.py index 2d959d3d..8d04041c 100644 --- a/src/lightcone/cli/commands.py +++ b/src/lightcone/cli/commands.py @@ -179,6 +179,7 @@ def run(cluster_id: str, command: tuple[str, ...]) -> None: remains available after the command finishes. """ from lightcone.engine import run as engine_run + from lightcone.engine.execution import ExecutionInterrupted from lightcone.engine.project import current_project _require_cluster_id(cluster_id) @@ -189,7 +190,7 @@ def run(cluster_id: str, command: tuple[str, ...]) -> None: try: outcome = engine_run.probe(current_project(), command, cluster_id=cluster_id) except KeyboardInterrupt as exc: - if getattr(exc, "execution_stopped", False): + if isinstance(exc, ExecutionInterrupted): click.echo( "Interrupted; the command has stopped. The cluster remains available.", err=True, ) @@ -383,7 +384,9 @@ def materialize( try: report = engine.materialize(root, targets, cluster_id=cluster_id, refresh=refresh) except KeyboardInterrupt as exc: - if getattr(exc, "execution_stopped", False): + from lightcone.engine.execution import ExecutionInterrupted + + if isinstance(exc, ExecutionInterrupted): click.echo( "Interrupted; recipes have stopped and uncommitted outputs were restored. " "The cluster remains available.", err=True, diff --git a/src/lightcone/cli/compute.py b/src/lightcone/cli/compute.py index 59fb2517..03427693 100644 --- a/src/lightcone/cli/compute.py +++ b/src/lightcone/cli/compute.py @@ -49,6 +49,11 @@ def _table(headers: list[str], rows: list[list[str]]) -> None: Console(markup=False).print(table) +def _duration(seconds: int) -> str: + minutes, remainder = divmod(seconds, 60) + return (f"{minutes}m" if minutes else "") + (f"{remainder}s" if remainder else "") + + @click.group() def compute() -> None: """Allocate resources, inspect clusters, and end allocations. @@ -77,8 +82,8 @@ def resources(as_json: bool) -> None: str(offer["resources"]["cpus"]), f"{offer['resources']['memory']:g} GiB", str(offer["max_nodes"]), - f"{offer['time']['default_seconds'] // 60}m", - f"{offer['time']['max_seconds'] // 60}m", + _duration(offer["time"]["default_seconds"]), + _duration(offer["time"]["max_seconds"]), offer["startup"], ] for offer in data["offers"] @@ -92,7 +97,7 @@ def resources(as_json: bool) -> None: @click.option("--memory", required=True, help="GiB per node; suffix + requests a minimum.") @click.option("--num-nodes", default=1, type=click.IntRange(min=1), show_default=True) @click.option( - "--time", "walltime", help="Requested walltime, e.g. 30m or 2h; defaults to the offer." + "--time", "walltime", help="Requested walltime, e.g. 30m or 1h30m; defaults to the offer." ) @click.option( "--startup", type=click.Choice(["fast"]), help="Require a fast startup service class." diff --git a/src/lightcone/engine/compute/__init__.py b/src/lightcone/engine/compute/__init__.py index 0f3c40c6..d1635726 100644 --- a/src/lightcone/engine/compute/__init__.py +++ b/src/lightcone/engine/compute/__init__.py @@ -39,14 +39,6 @@ def _slurm(connection: Connection) -> Provider: # The lifecycle seam is intentionally small: execution never dispatches on a provider. PROVIDERS: dict[str, ProviderFactory] = {"local": _local, "slurm": _slurm} -#: What a driver leaving early must say: closing a client cannot prove that a -#: remote subprocess has stopped. -UNSTOPPED = ( - "lc did not stop the allocation; tasks that did not report may still be " - "running, and any files they wrote remain" -) - - def validate_id(value: str) -> None: """Reject invalid execution targets before preparing a project.""" if value.startswith("clu_"): diff --git a/src/lightcone/engine/compute/local.py b/src/lightcone/engine/compute/local.py index 7ae1d3b6..0f84e97a 100644 --- a/src/lightcone/engine/compute/local.py +++ b/src/lightcone/engine/compute/local.py @@ -388,31 +388,26 @@ def terminate(self, identity: Identity) -> None: self._retire(directory) def _stop(self, identity: Identity, directory: Path, record: dict[str, Any]) -> None: + from lightcone.engine.sandbox.processes import ( + _CLEANUP_TIMEOUT, + has_custodian, + ) + from lightcone.engine.sandbox.processes import members as session_members + process = self._process(identity, directory, record) if process is None: return - members = [] - for member in psutil.process_iter(): - try: - if ( - member.uids().real == os.getuid() - and os.getsid(member.pid) == process.pid - ): - # Capture each birth identity while the owner still proves - # this session is ours. psutil's signal methods check reuse. - member.create_time() - members.append(member) - except (psutil.NoSuchProcess, psutil.AccessDenied, ProcessLookupError): - continue + # Capture birth identities while the owner establishes session custody. + members = session_members(session=process.pid) for member in members: try: member.terminate() except psutil.NoSuchProcess: pass for escalation in (False, True): - from lightcone.engine.sandbox.processes import has_custodian - - grace = max(_STOP_GRACE, 16) if has_custodian(members) else _STOP_GRACE + grace = ( + max(_STOP_GRACE, _CLEANUP_TIMEOUT + 1) if has_custodian(members) else _STOP_GRACE + ) deadline = time.monotonic() + grace while members and time.monotonic() < deadline: living = [] @@ -436,8 +431,6 @@ def _stop(self, identity: Identity, directory: Path, record: dict[str, Any]) -> if process is not None: # Commands have separate groups inside this session. The # live owner establishes custody of newly created members. - from lightcone.engine.sandbox.processes import members as session_members - for member in session_members(session=process.pid): try: member.kill() diff --git a/src/lightcone/engine/compute/local_runtime.py b/src/lightcone/engine/compute/local_runtime.py index 202b2e8b..2eed287e 100644 --- a/src/lightcone/engine/compute/local_runtime.py +++ b/src/lightcone/engine/compute/local_runtime.py @@ -11,6 +11,8 @@ from pathlib import Path from types import FrameType +import psutil + from lightcone.engine.compute.runtime import ( SCHEDULER_CONFIG, create_security, @@ -18,34 +20,43 @@ read_private_json, write_private_json, ) +from lightcone.engine.sandbox.processes import _CLEANUP_TIMEOUT, has_custodian, members def _stop_session() -> None: """Give command custodians time to drain, then kill the allocation session.""" - import psutil - - from lightcone.engine.sandbox.processes import has_custodian, members - owner = os.getpid() - remaining = [member for member in members(session=owner) if member.pid != owner] - for member in remaining: - try: - member.terminate() - except psutil.NoSuchProcess: - pass + remaining: list[psutil.Process] = [] started = time.monotonic() - while remaining: - grace = 15 if has_custodian(remaining) else 2.5 - if time.monotonic() - started >= grace: - break - time.sleep(0.05) + try: remaining = [member for member in members(session=owner) if member.pid != owner] - for member in remaining: + for member in remaining: + try: + member.terminate() + except psutil.NoSuchProcess: + pass + while remaining: + grace = _CLEANUP_TIMEOUT if has_custodian(remaining) else 2.5 + if time.monotonic() - started >= grace: + break + time.sleep(0.05) + remaining = [member for member in members(session=owner) if member.pid != owner] + except Exception: + # Enumeration can fail during shutdown. The original group is still + # ours; let its custodians drain their command groups before hard kill. + os.killpg(owner, signal.SIGTERM) + time.sleep(max(0.0, started + _CLEANUP_TIMEOUT - time.monotonic())) + finally: try: - member.kill() - except psutil.NoSuchProcess: - pass - os.kill(owner, signal.SIGKILL) + for member in remaining: + try: + member.kill() + except (psutil.NoSuchProcess, psutil.AccessDenied): + pass + finally: + # This also covers workers the last enumeration could not observe. + # The owner was verified as session/group leader before startup. + os.killpg(owner, signal.SIGKILL) def main() -> None: diff --git a/src/lightcone/engine/compute/model.py b/src/lightcone/engine/compute/model.py index ba7a25d1..bc078130 100644 --- a/src/lightcone/engine/compute/model.py +++ b/src/lightcone/engine/compute/model.py @@ -7,7 +7,7 @@ import re from collections.abc import Callable, Sequence from contextlib import AbstractContextManager -from decimal import Decimal, InvalidOperation, localcontext +from decimal import Decimal, localcontext from typing import Annotated, Any, Literal, Protocol, Self from uuid import UUID @@ -23,6 +23,7 @@ ) from lightcone.engine.project import ProjectError +from lightcone.engine.units import duration_seconds, whole_bytes GIB = 1024**3 @@ -43,11 +44,11 @@ class UnavailableOfferError(ComputeError): def duration(value: object) -> int: - """Parse an explicit positive whole-minute/hour duration into seconds.""" - match = re.fullmatch(r"([1-9][0-9]*)([mh])", str(value)) - if match is None: - raise ComputeError("duration must be a positive number of minutes or hours, e.g. 30m or 1h") - return int(match[1]) * (60 if match[2] == "m" else 3600) + """Parse an explicit positive duration into seconds.""" + try: + return duration_seconds(value) + except ValueError as exc: + raise ComputeError(str(exc)) from exc def memory_bytes(value: object) -> int: @@ -55,13 +56,9 @@ def memory_bytes(value: object) -> int: if isinstance(value, bool) or not re.fullmatch(r"[0-9]+(?:\.[0-9]+)?", str(value)): raise ComputeError("memory must be a positive number of GiB") try: - numerator, denominator = Decimal(str(value)).as_integer_ratio() - except InvalidOperation as exc: - raise ComputeError("memory must be a positive number of GiB") from exc - amount, remainder = divmod(numerator * GIB, denominator) - if amount <= 0 or remainder: - raise ComputeError("memory must be positive GiB exactly representable in bytes") - return amount + return whole_bytes(str(value), GIB) + except ValueError as exc: + raise ComputeError("memory must be positive GiB exactly representable in bytes") from exc def gib_from_bytes(value: int) -> Decimal: diff --git a/src/lightcone/engine/execution.py b/src/lightcone/engine/execution.py index 728e49e2..de8d27c8 100644 --- a/src/lightcone/engine/execution.py +++ b/src/lightcone/engine/execution.py @@ -13,10 +13,11 @@ from contextlib import contextmanager from contextvars import ContextVar from dataclasses import dataclass, field -from typing import Any +from typing import Any, Literal from uuid import uuid4 -from lightcone.engine.project import ProjectError +from lightcone.engine.sandbox.model import ExecutionCancelled as ExecutionCancelled +from lightcone.engine.sandbox.model import ExecutionUncertain as ExecutionUncertain _HEARTBEAT = 2.0 _LEASE = 15.0 @@ -27,12 +28,27 @@ ) -class ExecutionUncertain(ProjectError): # noqa: N818 - """Execution may still own writers; its partial outputs must be retained.""" +Operation = Literal[ + "register", "heartbeat", "active", "claim", "finished", "stopped", "uncertain", + "revoke", "pending", "forget", +] -class ExecutionCancelled(ProjectError): # noqa: N818 - """Execution was revoked and its command has stopped.""" +class ExecutionInterrupted(KeyboardInterrupt): + """An interrupt whose invocation has positively confirmed command cleanup.""" + + +@dataclass(frozen=True) +class _Claim: + fresh: bool + result: Any + remaining: float + + +@dataclass(frozen=True) +class _Progress: + running: tuple[str, ...] + uncertain: tuple[str, ...] def cancelled() -> bool: @@ -47,7 +63,7 @@ def check_cancelled() -> None: def _state( - invocation: str, operation: str, task: str = "", value: Any = None, + invocation: str, operation: Operation, task: str = "", value: Any = None, *, dask_scheduler: Any, ) -> Any: # Scheduler callbacks run serially on its event loop. Registration is a @@ -70,18 +86,18 @@ def _state( record["deadline"] = now + _LEASE return record["active"] if operation == "active": - return record["active"] + return max(0.0, record["deadline"] - now) if record["active"] else 0.0 if operation == "claim": if not record["active"]: raise ExecutionCancelled("execution is no longer active") previous = record["tasks"].get(task) if previous is not None: if previous["state"] == "finished": - return False, previous["result"] + return _Claim(False, previous["result"], 0.0) record["active"] = False raise ExecutionUncertain(f"{task}: a previous attempt has no confirmed result") record["tasks"][task] = {"state": "running", "attempt": value} - return True, None + return _Claim(True, None, record["deadline"] - now) if operation in {"finished", "stopped", "uncertain"}: attempt, result = value if record["tasks"].get(task, {}).get("attempt") != attempt: @@ -93,21 +109,22 @@ def _state( if operation == "revoke": record["active"] = False if operation in {"revoke", "pending"}: - return [name for name, item in record["tasks"].items() - if item["state"] in {"running", "uncertain"}] + return _Progress( + tuple(name for name, item in record["tasks"].items() if item["state"] == "running"), + tuple(name for name, item in record["tasks"].items() if item["state"] == "uncertain"), + ) if operation == "forget": del records[invocation] return None raise ValueError(f"unknown execution operation: {operation}") -async def _request(client: Any, invocation: str, operation: str, task: str, value: Any) -> Any: - return await client.run_on_scheduler(_state, invocation, operation, task, value) - - -def _rpc(client: Any, invocation: str, operation: str, task: str = "", value: Any = None) -> Any: +def _rpc( + client: Any, invocation: str, operation: Operation, task: str = "", value: Any = None, +) -> Any: return client.sync( - _request, client, invocation, operation, task, value, callback_timeout=_RPC_TIMEOUT, + client.run_on_scheduler, _state, invocation, operation, task, value, + callback_timeout=_RPC_TIMEOUT, ) @@ -117,30 +134,49 @@ def _call(invocation: str, task: str, function: Callable[..., Any], *args: Any) client = get_client() # A lost claim reply is ambiguous. Do not execute unless it was received. attempt = uuid4().hex - claimed, result = _rpc(client, invocation, "claim", task, attempt) - if not claimed: - return result + requested_at = time.monotonic() + claim: _Claim = _rpc(client, invocation, "claim", task, attempt) + if not claim.fresh: + return claim.result + deadline = requested_at + claim.remaining stopped = threading.Event() revoked = threading.Event() def monitor() -> None: + nonlocal deadline while not stopped.wait(_HEARTBEAT): + requested_at = time.monotonic() + if requested_at >= deadline: + revoked.set() + return try: - active = _rpc(client, invocation, "active") + remaining = _rpc(client, invocation, "active") + except ExecutionUncertain: + revoked.set() + return except Exception: - active = False - if not active: + # An unavailable RPC is not a revocation. Keep the last grant, + # without extending it, while retrying within its deadline. + continue + if not remaining or time.monotonic() >= deadline: revoked.set() return + deadline = requested_at + remaining + + def is_cancelled() -> bool: + if time.monotonic() >= deadline: + revoked.set() + return revoked.is_set() watcher = threading.Thread(target=monitor, daemon=True) watcher.start() - token = _CANCELLED.set(revoked.is_set) + token = _CANCELLED.set(is_cancelled) try: + check_cancelled() result = function(*args) check_cancelled() except BaseException as exc: - state = "uncertain" if isinstance(exc, ExecutionUncertain) else "stopped" + state: Operation = "uncertain" if isinstance(exc, ExecutionUncertain) else "stopped" try: _rpc(client, invocation, state, task, (attempt, None)) except Exception: @@ -151,7 +187,7 @@ def monitor() -> None: _rpc(client, invocation, "finished", task, (attempt, result)) except Exception as exc: raise ExecutionUncertain( - f"{task}: could not record completion; outputs retained, refusing replay" + f"{task}: could not record completion; refusing replay" ) from exc return result finally: @@ -167,11 +203,15 @@ class Invocation: id: str = field(default_factory=lambda: uuid4().hex) futures: list[Any] = field(default_factory=list) stopped: bool = False + _submissions: int = 0 def submit( self, function: Callable[..., Any], *args: Any, key: str, resources: dict[str, float], ) -> Any: """Claim each logical task inside its worker before it can mutate files.""" + # A submit failure may follow native acceptance. Count it before the + # call so even a caught exception cannot manufacture full completion. + self._submissions += 1 future = self.client.submit( _call, self.id, key, function, *args, key=f"lc-{self.id}-{key}", pure=False, retries=0, resources=resources, @@ -184,44 +224,68 @@ def submit( def invocation(client: Any) -> Iterator[Invocation]: """Own authorization and wait for running commands to stop before detaching.""" run = Invocation(client) + registered_at = time.monotonic() _rpc(client, run.id, "register", value=client.id) stopped = threading.Event() def heartbeat() -> None: + deadline = registered_at + _LEASE while not stopped.wait(_HEARTBEAT): + requested_at = time.monotonic() + if requested_at >= deadline: + return try: if not _rpc(client, run.id, "heartbeat"): return - except Exception: + except ExecutionUncertain: return + except Exception: + continue + deadline = requested_at + _LEASE threading.Thread(target=heartbeat, daemon=True).start() - interruption: KeyboardInterrupt | None = None + failure: BaseException | None = None try: yield run - except KeyboardInterrupt as exc: - interruption = exc + except BaseException as exc: + failure = exc raise finally: stopped.set() + # A finished wrapper has already acknowledged command cleanup and + # published its result. Losing metadata-cleanup RPCs cannot undo that. + run.stopped = ( + run._submissions == len(run.futures) + and all(future.status == "finished" for future in run.futures) + ) try: pending = _rpc(client, run.id, "revoke") unfinished = [future for future in run.futures if not future.done()] if unfinished: client.sync(client.cancel, unfinished, callback_timeout=_RPC_TIMEOUT) deadline = time.monotonic() + _STOP_TIMEOUT - while pending and time.monotonic() < deadline: + while pending.running and not pending.uncertain and time.monotonic() < deadline: time.sleep(0.1) pending = _rpc(client, run.id, "pending") - if pending: - raise ExecutionUncertain("unconfirmed tasks: " + ", ".join(pending)) - # A late dispatch cannot recreate this record: missing is a refusal. - _rpc(client, run.id, "forget") + if pending.running or pending.uncertain: + run.stopped = False + raise ExecutionUncertain( + "unconfirmed tasks: " + ", ".join((*pending.running, *pending.uncertain)) + ) run.stopped = True - if interruption is not None: - interruption.execution_stopped = True # type: ignore[attr-defined] except Exception as exc: - raise ExecutionUncertain( - f"could not confirm execution stopped: {exc}; partial outputs were retained. " - "Stop the allocation and verify its commands/containers have ended before retrying" - ) from exc + if not run.stopped: + raise ExecutionUncertain( + f"could not confirm execution stopped: {exc}; partial outputs were retained. " + "Stop the allocation and verify its commands/containers have ended " + "before retrying" + ) from exc + else: + # Revoked admission plus no unfinished claims already proves stop. + # Forgetting receipts is metadata cleanup, not another safety gate. + try: + _rpc(client, run.id, "forget") + except Exception: + pass + if isinstance(failure, KeyboardInterrupt) and run.stopped: + raise ExecutionInterrupted() from failure diff --git a/src/lightcone/engine/execution_resources.py b/src/lightcone/engine/execution_resources.py index 256a8797..8c21ad81 100644 --- a/src/lightcone/engine/execution_resources.py +++ b/src/lightcone/engine/execution_resources.py @@ -4,12 +4,12 @@ import math import re -from decimal import Decimal from typing import Any, Self from pydantic import BaseModel, ConfigDict, Field, ValidationError, field_validator from lightcone.engine.project import ProjectError +from lightcone.engine.units import duration_seconds, whole_bytes class TaskResources(BaseModel): @@ -137,19 +137,14 @@ def _memory(value: object) -> int: unit = match[2].lower() exponent = 0 if unit == "b" else "kmgtpe".index(unit[0]) + 1 factor: int = (1024 if "i" in unit else 1000) ** exponent - numerator, denominator = Decimal(match[1]).as_integer_ratio() - amount, remainder = divmod(numerator * factor, denominator) - if amount <= 0 or remainder: - raise ProjectError("recipe memory must be a positive whole number of bytes") - return amount + try: + return whole_bytes(match[1], factor) + except ValueError as exc: + raise ProjectError(f"recipe {exc}") from exc def _duration(value: object) -> int: - if not isinstance(value, str) or not ( - match := re.fullmatch(r"(?:(\d+)d)?(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?", value) - ): - raise ProjectError("recipe time_limit must be a duration, e.g. 30m, 1h30m, or 45s") - seconds = sum(int(part or 0) * unit for part, unit in zip(match.groups(), (86400, 3600, 60, 1))) - if seconds <= 0: - raise ProjectError("recipe time_limit must be positive") - return seconds + try: + return duration_seconds(value) + except ValueError as exc: + raise ProjectError(f"recipe time_limit: {exc}") from exc diff --git a/src/lightcone/engine/materialize.py b/src/lightcone/engine/materialize.py index b15c84ec..620e070a 100644 --- a/src/lightcone/engine/materialize.py +++ b/src/lightcone/engine/materialize.py @@ -14,8 +14,9 @@ **It owns git, alone.** Workers execute and return; the driver commits, in one thread, as results arrive. That is not a preference: concurrent git operations on one repository race on the index lock. The same loop -restores what a completed failed task left behind. Unreported tasks may -still be writing, so interruptions retain their partial files. +restores what a completed failed task left behind. On interruption it restores +unreported outputs only after confirming their writers stopped; otherwise it +retains those partial files. One consequence, checked rather than assumed: a dependent starts as soon as its upstream's *worker* returns, which is milliseconds before the @@ -41,6 +42,7 @@ from typing import TYPE_CHECKING, Any, Protocol from lightcone.engine import assets, container, dataset, execution, identity, plan, project, worker +from lightcone.engine.execution_resources import TaskResources from lightcone.engine.plan import Graph, Key, Task from lightcone.engine.project import ProjectError @@ -518,7 +520,7 @@ def materialize( scheduler: Scheduler | None = None try: with cluster_for_run(cluster_id) as scheduler: - scheduler.validate(graph.tasks.values()) + requirements = scheduler.validate(graph.tasks.values()) _fetch_inputs(root, graph, report) # Materialize is one of the two verbs allowed to build the image (the # other is `lc build`); the probe and the rerun entry point only find @@ -580,6 +582,7 @@ def materialize( foreign[key], *[pending[dep] for dep in task.depends_on], key=_name(key), + resources=requirements[key], ) for result in scheduler.completed(list(pending.values())): @@ -654,17 +657,18 @@ def stopped(self) -> bool: """Whether this invocation positively confirmed all claimed tasks stopped.""" ... - def validate(self, tasks: Iterable[Task]) -> None: - """Refuse unsatisfiable resource requests before preparation or submission.""" + def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: + """Validate all requests and return their Dask resource reservations.""" ... - def submit(self, fn: Any, *args: Any, key: str) -> Any: + def submit(self, fn: Any, *args: Any, key: str, resources: dict[str, float]) -> Any: """Schedule a call. Args: fn: The function to run. *args: Its arguments, upstream handles included. key: A display name for the task. + resources: Validated reservations for this task. Returns: A handle to pass to dependents. @@ -697,20 +701,20 @@ def stopped(self) -> bool: """Expose positive cleanup confirmation after the connection context exits.""" return self.invocation.stopped - def validate(self, tasks: Iterable[Task]) -> None: + def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: """Require each selected task to fit a worker before any task starts.""" + requests = {} for task in tasks: try: - task.resources.requirements(self.workers) + requests[task.key] = TaskResources.parse(task.resources).requirements(self.workers) except ProjectError as exc: raise ProjectError(f"{_name(task.key)}: {exc}") from exc + return requests - def submit(self, fn: Any, *args: Any, key: str) -> Any: + def submit(self, fn: Any, *args: Any, key: str, resources: dict[str, float]) -> Any: """Submit an ordinary Dask task with a unique key and forwarded output.""" from lightcone.engine.compute.output import call - task: Task = args[1] - resources = task.resources.requirements(self.workers) return self.invocation.submit( call, fn, self.output.topic, key, *args, key=key, resources=resources, @@ -734,9 +738,7 @@ def completed(self, handles: list[Any]) -> Iterator[worker.TaskResult]: except ProjectError: raise except Exception as exc: - from lightcone.engine.compute import UNSTOPPED - - raise ProjectError(f"cluster execution failed: {exc}. {UNSTOPPED}") from exc + raise ProjectError(f"cluster execution failed: {exc}") from exc @contextmanager diff --git a/src/lightcone/engine/plan.py b/src/lightcone/engine/plan.py index 614e5fbb..513f410c 100644 --- a/src/lightcone/engine/plan.py +++ b/src/lightcone/engine/plan.py @@ -26,9 +26,9 @@ from dataclasses import dataclass, field from graphlib import CycleError, TopologicalSorter from pathlib import Path +from typing import Any from lightcone.engine import assets, identity -from lightcone.engine.execution_resources import TaskResources from lightcone.engine.project import SPEC_FILENAME, ProjectError #: A task's identity within a run: which universe, which output. @@ -52,7 +52,8 @@ class Task: produced_by: dict[str, Key] decisions: dict[str, str] definition_version: str - resources: TaskResources = field(default_factory=TaskResources) + #: ASTRA's declaration; executor support is checked only when executing. + resources: dict[str, Any] = field(default_factory=dict) @property def manifest_path(self) -> Path: @@ -333,10 +334,7 @@ def file_of(out: object) -> Path: ) except ValueError as e: raise ProjectError(f"output `{out.id}`: {e}") from e - try: - resources = TaskResources.parse((out.definition.get("recipe") or {}).get("resources")) - except ProjectError as e: - raise ProjectError(f"output `{out.id}`: {e}") from e + resources = dict((out.definition.get("recipe") or {}).get("resources") or {}) tasks.append( Task( diff --git a/src/lightcone/engine/run.py b/src/lightcone/engine/run.py index 449bdbe2..faeb10df 100644 --- a/src/lightcone/engine/run.py +++ b/src/lightcone/engine/run.py @@ -66,9 +66,7 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. except ProjectError: raise except Exception as exc: - raise ProjectError( - f"cluster execution failed: {exc}. {compute.UNSTOPPED}" - ) from exc + raise ProjectError(f"cluster execution failed: {exc}") from exc if not output.wait("probe"): notes.append("remote output forwarding did not finish before its deadline") if warning := uv_scrub_warning(): diff --git a/src/lightcone/engine/sandbox/boundary.py b/src/lightcone/engine/sandbox/boundary.py index a475a984..4a66c7d3 100644 --- a/src/lightcone/engine/sandbox/boundary.py +++ b/src/lightcone/engine/sandbox/boundary.py @@ -20,7 +20,13 @@ from typing import IO from lightcone.engine.sandbox import policy as policy_module -from lightcone.engine.sandbox.model import Attestation, Backend, Capability, Policy +from lightcone.engine.sandbox.model import ( + Attestation, + Backend, + Capability, + ExecutionUncertain, + Policy, +) from lightcone.engine.sandbox.processes import Command #: How much of the child's stderr to keep for the denial classifier. The @@ -117,8 +123,6 @@ def scope(policy: Policy) -> Iterator[Policy]: Yields: The same policy, with its ``tmp_home`` removed on exit. """ - from lightcone.engine.execution import ExecutionUncertain - cleanup = True try: yield policy @@ -175,6 +179,11 @@ def run( else: wrapped = [*prefix, *backend.wrap(policy, [*env_argv(policy), *argv])] attestation = backend.attest(policy) + oci_runtime = ( + attestation.mechanism + if attestation.mechanism in {"podman", "docker", "podman-hpc"} + else None + ) # `policy.env` is deliberately **not** merged here: it went inside # the wrap, above, via :func:`env_argv`. Everything *outside* the # rewrite has to keep the real environment — `uv` resolves its cache @@ -189,7 +198,7 @@ def run( with Command( wrapped, cwd=cwd, env=child_env, capture=output is not None, timeout=timeout, - container=backend.contains_prefix, + oci_runtime=oci_runtime, ) as command: proc = command.process assert proc.stderr is not None # Popen was given PIPE @@ -230,7 +239,7 @@ def forward() -> None: "lc could not set up the sandbox (see above) — this is an lc " "problem, not your command's" ) - elif returncode == 125 and backend.contains_prefix: + elif returncode == 125 and oci_runtime is not None: # The runtimes reserve 125 for their own failures (a bad flag, a # vanished mount source): the command never ran, so the denial # heuristics have nothing to say about it. diff --git a/src/lightcone/engine/sandbox/model.py b/src/lightcone/engine/sandbox/model.py index b8ad809b..9389e34d 100644 --- a/src/lightcone/engine/sandbox/model.py +++ b/src/lightcone/engine/sandbox/model.py @@ -26,6 +26,16 @@ from pathlib import Path from typing import Literal, Protocol +from lightcone.engine.project import ProjectError + + +class ExecutionUncertain(ProjectError): # noqa: N818 + """Command cleanup is unconfirmed; retain any files it could still write.""" + + +class ExecutionCancelled(ProjectError): # noqa: N818 + """Execution was revoked and its command has stopped.""" + #: Bumped when the meaning of the exec allowlist changes. It is recorded #: in the attestation, so a run stays interpretable after the list #: grows — the allowlist is a maintained policy surface. diff --git a/src/lightcone/engine/sandbox/oci.py b/src/lightcone/engine/sandbox/oci.py index 303273f7..dcc44451 100644 --- a/src/lightcone/engine/sandbox/oci.py +++ b/src/lightcone/engine/sandbox/oci.py @@ -24,6 +24,7 @@ from lightcone.engine.sandbox.boundary import SANDBOX_ENV from lightcone.engine.sandbox.model import Attestation, Capability, Policy +from lightcone.engine.sandbox.processes import CIDFILE #: The runtimes this backend can speak for — the one statement of the #: set, so the type does not get hand-copied out of step at its uses. @@ -86,6 +87,9 @@ def wrap(self, policy: Policy, argv: Sequence[str]) -> list[str]: # The custodian retains the native record until it has inspected # the immutable container ID and confirmed the payload stopped. self.runtime, "run", + # The supervisor runs this native client in a private directory; + # --workdir below independently sets the payload's project cwd. + "--cidfile", CIDFILE, "--entrypoint", "", # The rootfs is read-only so a write outside the declared set # is a loud denial rather than bytes vanishing with the diff --git a/src/lightcone/engine/sandbox/processes.py b/src/lightcone/engine/sandbox/processes.py index f8711135..8cf6f57b 100644 --- a/src/lightcone/engine/sandbox/processes.py +++ b/src/lightcone/engine/sandbox/processes.py @@ -24,7 +24,10 @@ import psutil _GRACE = 1.0 +_EXIT_GRACE = 1.0 _CLEANUP_TIMEOUT = 15.0 +_REPORT_GRACE = 1.0 +CIDFILE = "container.cid" def members(*, group: int | None = None, session: int | None = None) -> list[psutil.Process]: @@ -61,8 +64,10 @@ def has_custodian(processes: Sequence[psutil.Process]) -> bool: return False -def _drain(process: subprocess.Popen[bytes]) -> bool: +def _drain(process: subprocess.Popen[bytes], *, deadline: float | None = None) -> bool: """Stop the whole command group while its unreaped leader pins the group ID.""" + if deadline is None: + deadline = time.monotonic() + _CLEANUP_TIMEOUT for sig in (signal.SIGTERM, signal.SIGKILL): if not members(group=process.pid): process.wait() @@ -78,9 +83,9 @@ def _drain(process: subprocess.Popen[bytes]) -> bool: raise process.wait() return True - deadline = time.monotonic() + _GRACE + until = min(deadline, time.monotonic() + _GRACE) if sig == signal.SIGTERM else deadline while members(group=process.pid): - if time.monotonic() >= deadline: + if time.monotonic() >= until: break time.sleep(0.025) else: @@ -97,20 +102,22 @@ class Command: cwd: Command working directory. env: Command environment. capture: Whether to pipe stdout and disable stdin. - timeout: Maximum command runtime in seconds, or no bound. - container: Whether argv starts a supported OCI runtime. + timeout: Maximum command lifetime including startup, or no bound, in seconds. + oci_runtime: The native OCI runtime, or None for a host command. """ def __init__( self, argv: Sequence[str], *, cwd: Path, env: dict[str, str], capture: bool, - timeout: float | None = None, container: bool = False, + timeout: float | None = None, oci_runtime: str | None = None, ) -> None: self._control, control_write = os.pipe() status_read, self._status = os.pipe() self._writer = os.fdopen(control_write, "wb", buffering=0) self._reader = os.fdopen(status_read, "rb") + execution_deadline = time.monotonic() + timeout if timeout is not None else None self._deadline = ( - time.monotonic() + timeout + _CLEANUP_TIMEOUT if timeout is not None else None + execution_deadline + _CLEANUP_TIMEOUT + _REPORT_GRACE + if execution_deadline is not None else None ) try: self.process = subprocess.Popen( @@ -123,7 +130,7 @@ def __init__( ) self._writer.write(json.dumps({ "argv": list(argv), "cwd": str(cwd), "env": env, - "timeout": timeout, "container": container, + "deadline": execution_deadline, "oci_runtime": oci_runtime, }).encode() + b"\n") except BaseException: self._writer.close() @@ -131,7 +138,7 @@ def __init__( try: self.wait() except Exception as cleanup_error: - from lightcone.engine.execution import ExecutionCancelled + from lightcone.engine.sandbox.model import ExecutionCancelled if not isinstance(cleanup_error, ExecutionCancelled): raise @@ -149,7 +156,7 @@ def __exit__( self, exc_type: type[BaseException] | None, exc: BaseException | None, traceback: TracebackType | None, ) -> None: - from lightcone.engine.execution import ExecutionCancelled + from lightcone.engine.sandbox.model import ExecutionCancelled if not self._reader.closed: self._writer.close() @@ -160,16 +167,18 @@ def __exit__( def wait(self, cancelled: Callable[[], bool] | None = None) -> tuple[int, str]: """Wait for command completion; cancellation includes verified cleanup.""" - from lightcone.engine.execution import ExecutionCancelled, ExecutionUncertain + from lightcone.engine.sandbox.model import ExecutionCancelled, ExecutionUncertain requested = self._writer.closed - deadline = time.monotonic() + _CLEANUP_TIMEOUT if requested else self._deadline + deadline = ( + time.monotonic() + _CLEANUP_TIMEOUT + _REPORT_GRACE if requested else self._deadline + ) try: while self.process.poll() is None: if not requested and cancelled is not None and cancelled(): self._writer.close() requested = True - deadline = time.monotonic() + _CLEANUP_TIMEOUT + deadline = time.monotonic() + _CLEANUP_TIMEOUT + _REPORT_GRACE if deadline is not None and time.monotonic() >= deadline: raise ExecutionUncertain("command cleanup did not finish") time.sleep(0.025) @@ -177,8 +186,11 @@ def wait(self, cancelled: Callable[[], bool] | None = None) -> tuple[int, str]: self._writer.close() try: remaining = ( - _CLEANUP_TIMEOUT if deadline is None - else max(0.0, deadline - time.monotonic()) + _CLEANUP_TIMEOUT + _REPORT_GRACE if deadline is None + else min( + _CLEANUP_TIMEOUT + _REPORT_GRACE, + max(0.0, deadline - time.monotonic()), + ) ) self.process.wait(timeout=remaining) except subprocess.TimeoutExpired as exc: @@ -200,7 +212,7 @@ def wait(self, cancelled: Callable[[], bool] | None = None) -> tuple[int, str]: return int(report["returncode"]), str(report.get("note", "")) def _report(self) -> dict[str, Any]: - from lightcone.engine.execution import ExecutionUncertain + from lightcone.engine.sandbox.model import ExecutionUncertain try: with self._reader: @@ -214,7 +226,9 @@ def _report(self) -> dict[str, Any]: ) from exc -def _container_cleanup(runtime: str, cidfile: Path, env: dict[str, str]) -> None: +def _container_cleanup( + runtime: str, cidfile: Path, env: dict[str, str], *, deadline: float, +) -> None: """Stop, inspect and remove exactly the container created by this command.""" try: identity = cidfile.read_text().strip() @@ -224,9 +238,12 @@ def _container_cleanup(runtime: str, cidfile: Path, env: dict[str, str]) -> None raise RuntimeError("container runtime did not publish a valid immutable container ID") def run(*args: str) -> subprocess.CompletedProcess[str]: + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("command cleanup deadline expired") return subprocess.run( [runtime, *args], env=env, stdin=subprocess.DEVNULL, - capture_output=True, text=True, timeout=2, + capture_output=True, text=True, timeout=remaining, ) def running() -> bool: @@ -243,7 +260,6 @@ def running() -> bool: pass if running(): run("kill", identity) - deadline = time.monotonic() + 2 while running(): if time.monotonic() >= deadline: raise RuntimeError(f"container {identity} remains running after kill") @@ -273,6 +289,7 @@ def stop(_signum: int, _frame: FrameType | None) -> None: signal.signal(signal.SIGTERM, stop) signal.signal(signal.SIGINT, stop) process: subprocess.Popen[bytes] | None = None + cleanup_deadline: float | None = None report: dict[str, Any] = {"returncode": 125} with os.fdopen(control, "rb", buffering=0) as channel, tempfile.TemporaryDirectory( prefix="lc-command-" @@ -280,16 +297,18 @@ def stop(_signum: int, _frame: FrameType | None) -> None: try: config = json.loads(channel.readline()) argv = config["argv"] - cidfile = Path(directory) / "container" - if config["container"]: - argv[2:2] = ["--cidfile", str(cidfile)] + cidfile = Path(directory) / CIDFILE + oci_runtime = config["oci_runtime"] if stopped or select.select([channel], [], [], 0)[0]: report["cancelled"] = True return + if config["deadline"] is not None and time.monotonic() >= config["deadline"]: + report.update(returncode=124, note="command timed out before startup") + return process = subprocess.Popen( - argv, cwd=config["cwd"], env=config["env"], process_group=0, + argv, cwd=directory if oci_runtime else config["cwd"], + env=config["env"], process_group=0, ) - started = time.monotonic() reason = "" # Keep the leader unreaped until group cleanup is complete, pinning # its PID/PGID against reuse. Unlike waitid(WNOWAIT), psutil also @@ -300,25 +319,39 @@ def stop(_signum: int, _frame: FrameType | None) -> None: reason = "cancelled" break if ( - config["timeout"] is not None - and time.monotonic() - started >= config["timeout"] + config["deadline"] is not None + and time.monotonic() >= config["deadline"] ): reason = "timed out" break - if not reason and members(group=process.pid): - reason = "left background processes running" + # Helpers such as multiprocessing's resource_tracker finish after + # their parent closes its pipe. Give normal teardown a short grace. + exit_deadline = time.monotonic() + _EXIT_GRACE + while not reason and members(group=process.pid): + if stopped or select.select([channel], [], [], 0.025)[0]: + reason = "cancelled" + elif ( + config["deadline"] is not None + and time.monotonic() >= config["deadline"] + ): + reason = "timed out" + elif time.monotonic() >= exit_deadline: + reason = "left background processes running" + cleanup_deadline = time.monotonic() + _CLEANUP_TIMEOUT container_stopped = False - if config["container"]: + if oci_runtime: # A runtime startup failure with no CID has not published a # container; interruption in that window cannot prove the same. if cidfile.exists() or reason: - _container_cleanup(argv[0], cidfile, config["env"]) + _container_cleanup( + oci_runtime, cidfile, config["env"], deadline=cleanup_deadline, + ) # Podman removes its cidfile together with the container. # Keep the verified outcome, not the file's later existence. container_stopped = True - if not _drain(process): + if not _drain(process, deadline=cleanup_deadline): raise RuntimeError("command processes remain alive after SIGKILL") - if config["container"] and not container_stopped and process.returncode != 125: + if oci_runtime and not container_stopped and process.returncode != 125: raise RuntimeError("container runtime exited without recording its identity") report["returncode"] = process.returncode if reason == "cancelled": @@ -332,7 +365,7 @@ def stop(_signum: int, _frame: FrameType | None) -> None: report["start_error"] = f"command could not start: {str(exc)[:2048]}" else: try: - _drain(process) + _drain(process, deadline=cleanup_deadline) except BaseException: pass report["error"] = f"cannot confirm command cleanup: {str(exc)[:2048]}" diff --git a/src/lightcone/engine/units.py b/src/lightcone/engine/units.py new file mode 100644 index 00000000..2d47130a --- /dev/null +++ b/src/lightcone/engine/units.py @@ -0,0 +1,45 @@ +"""Exact byte quantities and explicit walltime units used by execution and compute.""" + +from __future__ import annotations + +import re +from decimal import Decimal, InvalidOperation + + +def whole_bytes(amount: str, unit_bytes: int) -> int: + """Scale a decimal memory amount into positive bytes without rounding. + + Args: + amount: A decimal quantity in the caller's units. + unit_bytes: The number of bytes represented by one unit. + + Raises: + ValueError: The quantity is not positive or exactly representable in bytes. + """ + try: + numerator, denominator = Decimal(amount).as_integer_ratio() + except (InvalidOperation, ValueError, OverflowError) as exc: + raise ValueError("memory must be positive and exactly representable in bytes") from exc + result, remainder = divmod(numerator * unit_bytes, denominator) + if result <= 0 or remainder: + raise ValueError("memory must be positive and exactly representable in bytes") + return result + + +def duration_seconds(value: object) -> int: + """Parse a positive duration such as ``30m``, ``1h30m``, or ``45s``. + + Args: + value: Ordered day, hour, minute, and second components. + + Raises: + ValueError: The duration is malformed, empty, or zero. + """ + if not isinstance(value, str) or not ( + match := re.fullmatch(r"(?:(\d+)d)?(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?", value) + ): + raise ValueError("duration must use explicit units, e.g. 30m, 1h30m, or 45s") + seconds = sum(int(part or 0) * unit for part, unit in zip(match.groups(), (86400, 3600, 60, 1))) + if seconds <= 0: + raise ValueError("duration must be positive") + return seconds diff --git a/src/lightcone/engine/worker.py b/src/lightcone/engine/worker.py index 2a16fc88..d0d0c8df 100644 --- a/src/lightcone/engine/worker.py +++ b/src/lightcone/engine/worker.py @@ -36,6 +36,7 @@ from typing import Literal from lightcone.engine import assets, container, dataset, execution, identity, plan, project, sandbox +from lightcone.engine.execution_resources import TaskResources from lightcone.engine.plan import Key, Task from lightcone.engine.project import ( ProjectError, @@ -219,7 +220,7 @@ def execute( committed as part of an output that never produced it. The context's ``env_version`` is checked either side of the recipe, so a mid-run lock edit cannot be recorded as if it had been in force. The subprocess - boundary applies ``task.resources.time_seconds`` and drains owned processes + boundary applies the recipe's time limit and drains owned processes before reporting completion. CPU and memory scheduling happens upstream. Args: @@ -239,6 +240,7 @@ def execute( ExecutionUncertain: Subprocess teardown could not be confirmed. """ execution.check_cancelled() + resources = TaskResources.parse(task.resources) if moved := _gate(root, context.env_version): return TaskResult(task.key, "failed", reason=moved) @@ -270,7 +272,7 @@ def execute( prefix=uv_prefix(root), env=child_env(), output=output, - timeout=task.resources.time_seconds, + timeout=resources.time_seconds, cancelled=execution.cancelled, ) finished_at = _now() diff --git a/tests/conftest.py b/tests/conftest.py index d71c8532..8980a535 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -15,6 +15,7 @@ from lightcone.engine import dataset, project, templates from lightcone.engine.compute.model import Identity +from lightcone.engine.plan import Key, Task from lightcone.engine.project import _run as _real_run CLUSTER_ID = Identity( @@ -153,10 +154,13 @@ class _Inline: stopped = True # Calls are synchronous; no remote work can survive the fixture. - def validate(self, tasks: Iterable[object]) -> None: + def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: """Run fixture tasks without a finite cluster resource envelope.""" + return {task.key: {} for task in tasks} - def submit(self, fn: Callable[..., object], *args: object, key: str) -> object: + def submit( + self, fn: Callable[..., object], *args: object, key: str, resources: dict[str, float], + ) -> object: return fn(*args) def completed(self, handles: list[object]) -> Iterator[object]: diff --git a/tests/test_cli.py b/tests/test_cli.py index ee1902c1..1f6be9a4 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -466,9 +466,9 @@ def test_execution_interrupt_explains_how_to_stop_remote_work( from lightcone.engine import run as engine_run def interrupt(*args: object, **kwargs: object) -> None: - exc = KeyboardInterrupt() - exc.execution_stopped = stopped - raise exc + from lightcone.engine.execution import ExecutionInterrupted + + raise ExecutionInterrupted() if stopped else KeyboardInterrupt() monkeypatch.setattr(engine_run, "probe", interrupt) monkeypatch.setattr(engine_materialize, "materialize", interrupt) diff --git a/tests/test_compute.py b/tests/test_compute.py index 7e56ab99..dac26101 100644 --- a/tests/test_compute.py +++ b/tests/test_compute.py @@ -29,6 +29,7 @@ Startup, TimeLimits, UnavailableOfferError, + duration, memory_bytes, validate_name, ) @@ -364,6 +365,25 @@ def test_memory_conversion_does_not_round_fractional_bytes() -> None: memory_bytes("0.000000000931322574615478515625000000000000000001") +@pytest.mark.parametrize( + ("value", "seconds"), + [("1h30m", 5400), ("45s", 45), ("2d3h4m5s", 183845)], +) +def test_allocation_and_recipe_durations_use_the_same_units(value: str, seconds: int) -> None: + from lightcone.engine.execution_resources import TaskResources + + assert duration(value) == seconds + assert TimeLimits(default=value, max="3d").default_seconds == seconds + assert Request.parse("1", "1", time=value).seconds == seconds + assert TaskResources.parse({"time_limit": value}).time_seconds == seconds + + +@pytest.mark.parametrize("value", ["", "0s", "1.5h", "30m1h", "1h30", 60, True]) +def test_allocation_duration_refuses_ambiguous_or_zero_values(value: object) -> None: + with pytest.raises(ComputeError): + duration(value) + + @pytest.mark.parametrize("size", [1, GIB // 2, 8 * GIB, 2**80 + 1]) def test_resource_units_survive_construction_serialization_and_updates(size: int) -> None: whole, fraction = divmod(size, GIB) @@ -737,6 +757,16 @@ def test_cli_resources_dry_run_launch_down(catalog: Path, provider: MagicMock) - provider.terminate.assert_called_once_with(IDENTITY) +def test_cli_resources_preserves_seconds(catalog: Path) -> None: + data = yaml.safe_load(catalog.read_text()) + data["offers"][0]["time"] = {"default": "45s", "max": "1m30s"} + catalog.write_text(yaml.safe_dump(data)) + result = CliRunner().invoke(main, ["compute", "resources"]) + assert result.exit_code == 0, result.output + assert "45s" in result.output + assert "1m30s" in result.output + + def test_cli_launch_name_output_can_be_captured_without_json( catalog: Path, provider: MagicMock, ) -> None: diff --git a/tests/test_compute_local.py b/tests/test_compute_local.py index 0f3a9cd4..31b9358b 100644 --- a/tests/test_compute_local.py +++ b/tests/test_compute_local.py @@ -378,6 +378,53 @@ def test_owner_shutdown_drains_recipes_without_a_waiting_cli(provider: LocalProv child.kill() +@pytest.mark.parametrize("after_capture", [False, True]) +def test_owner_walltime_still_hard_kills_when_process_enumeration_fails( + after_capture: bool, +) -> None: + script = f""" +import os, signal, subprocess, sys, time +from lightcone.engine.compute import local_runtime + +after_capture = {after_capture!r} +called = False +def fail(**kwargs): + global called + if after_capture and not called: + called = True + return [local_runtime.psutil.Process(child.pid)] + raise RuntimeError('process enumeration unavailable') + +child = subprocess.Popen([sys.executable, '-c', + "import signal,time; signal.signal(signal.SIGTERM, signal.SIG_IGN); " + "print('ready',flush=True); time.sleep(60)"], stdout=subprocess.PIPE, + process_group=0 if after_capture else os.getpgrp()) +assert child.stdout.readline() == b'ready\\n' +print(child.pid, flush=True) +local_runtime.members = fail +local_runtime._CLEANUP_TIMEOUT = .1 +signal.signal(signal.SIGTERM, signal.SIG_IGN) +signal.signal(signal.SIGALRM, lambda *_: local_runtime._stop_session()) +signal.setitimer(signal.ITIMER_REAL, .1) +time.sleep(60) +""" + owner = subprocess.Popen( + [sys.executable, "-c", script], start_new_session=True, stdout=subprocess.PIPE, + ) + assert owner.stdout is not None + child = psutil.Process(int(owner.stdout.readline())) + try: + assert owner.wait(timeout=5) == -signal.SIGKILL + assert not child.is_running() or child.status() == psutil.STATUS_ZOMBIE + finally: + if owner.poll() is None: + owner.kill() + owner.wait(timeout=5) + if child.is_running(): + child.kill() + owner.stdout.close() + + def test_termination_escalates_captured_children_when_owner_exits_first( provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, ) -> None: diff --git a/tests/test_execution.py b/tests/test_execution.py index 46d32f6c..b8649cad 100644 --- a/tests/test_execution.py +++ b/tests/test_execution.py @@ -86,6 +86,14 @@ def _raise(error: Exception) -> None: raise error +def _stay_authorized(seconds: float) -> str: + deadline = time.monotonic() + seconds + while time.monotonic() < deadline: + execution.check_cancelled() + time.sleep(0.01) + return "completed" + + def _records(*, dask_scheduler: Any) -> dict[str, Any]: return dask_scheduler.extensions.get("lightcone-executions", {}) @@ -170,7 +178,7 @@ def test_duplicate_running_attempt_cannot_execute_or_finish_the_original_claim( ) with pytest.raises(execution.ExecutionUncertain, match="previous attempt"): duplicate.result(timeout=3) - assert execution._rpc(execution_client, run.id, "pending") == ["recipe"] + assert execution._rpc(execution_client, run.id, "pending").running == ("recipe",) assert not duplicate_effect.exists() finally: release.touch() @@ -180,7 +188,7 @@ def test_duplicate_running_attempt_cannot_execute_or_finish_the_original_claim( # Rejecting the ambiguous duplicate may revoke the invocation before # the original completes. Only that original may acknowledge its stop. pass - assert execution._rpc(execution_client, run.id, "pending") == [] + assert execution._rpc(execution_client, run.id, "pending").running == () def test_late_dispatch_after_invocation_exit_cannot_recreate_authorization( @@ -202,19 +210,18 @@ def test_missing_scheduler_state_refuses_both_replay_and_unstarted_tasks( execution_client: Client, tmp_path: Path, ) -> None: effects = tmp_path / "effects" - with pytest.raises(execution.ExecutionUncertain, match="confirm execution stopped"): - with execution.invocation(execution_client) as run: - run.submit(_effect, effects, key="recipe", resources=_RESOURCES).result() - execution_client.run_on_scheduler(_lose_state) - for key in ("recipe", "new-recipe"): - future = execution_client.submit( - execution._call, run.id, key, _effect, effects, - key=f"lost-state-{uuid4().hex}", pure=False, - ) - with pytest.raises(execution.ExecutionUncertain, match="no longer registered"): - future.result(timeout=3) + with execution.invocation(execution_client) as run: + run.submit(_effect, effects, key="recipe", resources=_RESOURCES).result() + execution_client.run_on_scheduler(_lose_state) + for key in ("recipe", "new-recipe"): + future = execution_client.submit( + execution._call, run.id, key, _effect, effects, + key=f"lost-state-{uuid4().hex}", pure=False, + ) + with pytest.raises(execution.ExecutionUncertain, match="no longer registered"): + future.result(timeout=3) assert effects.read_text() == "executed\n" - assert not run.stopped + assert run.stopped # The only admitted task returned its confirmed completion. def test_missing_scheduler_state_stops_running_work_without_claiming_confirmed_cleanup( @@ -252,7 +259,7 @@ def lose_claim_response(*args: Any, **kwargs: Any) -> Any: future = run.submit(_effect, effects, key="recipe", resources=_RESOURCES) with pytest.raises(TimeoutError, match="reply lost"): future.result(timeout=3) - assert execution._rpc(execution_client, run.id, "pending") == ["recipe"] + assert execution._rpc(execution_client, run.id, "pending").running == ("recipe",) assert not effects.exists() @@ -312,7 +319,7 @@ def test_driver_disconnect_revokes_execution_even_while_an_observer_remains_conn _wait_for(stopped) assert not execution._rpc(execution_client, run.id, "heartbeat") deadline = time.monotonic() + 3 - while execution._rpc(execution_client, run.id, "pending"): + while execution._rpc(execution_client, run.id, "pending").running: assert time.monotonic() < deadline, "task never acknowledged that it stopped" time.sleep(0.01) execution._rpc(execution_client, run.id, "forget") @@ -338,7 +345,7 @@ def test_function_exception_confirms_stop_but_does_not_authorize_reexecution( future = run.submit(_raise, ValueError("recipe failed"), key="recipe", resources=_RESOURCES) with pytest.raises(ValueError, match="recipe failed"): future.result(timeout=3) - assert execution._rpc(execution_client, run.id, "pending") == [] + assert execution._rpc(execution_client, run.id, "pending").running == () replay = execution_client.submit( execution._call, run.id, "recipe", _raise, AssertionError("must not execute"), pure=False, @@ -359,8 +366,8 @@ def test_uncertain_task_remains_unresolved_after_context_exit( ) with pytest.raises(execution.ExecutionUncertain, match="container may still"): future.result(timeout=3) - assert execution._rpc(execution_client, run.id, "pending") == ["recipe"] - assert execution._rpc(execution_client, run.id, "pending") == ["recipe"] + assert execution._rpc(execution_client, run.id, "pending").uncertain == ("recipe",) + assert execution._rpc(execution_client, run.id, "pending").uncertain == ("recipe",) assert not execution._rpc(execution_client, run.id, "heartbeat") assert not run.stopped @@ -384,3 +391,167 @@ def test_uncertain_cleanup_prevents_independent_tasks_from_using_released_dask_r error = later.exception(timeout=3) assert isinstance(error, execution.ExecutionCancelled) assert not effects.exists() + + +def test_transient_heartbeat_and_monitor_failures_preserve_the_confirmed_lease( + execution_client: Client, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(execution, "_LEASE", 0.4) + request = execution._rpc + calls = {"heartbeat": 0, "active": 0} + + def fail_once(*args: Any, **kwargs: Any) -> Any: + operation = args[2] + if operation in calls: + calls[operation] += 1 + if calls[operation] == 1: + raise TimeoutError("temporary scheduler RPC failure") + return request(*args, **kwargs) + + monkeypatch.setattr(execution, "_rpc", fail_once) + with execution.invocation(execution_client) as run: + future = run.submit(_stay_authorized, 0.8, key="recipe", resources=_RESOURCES) + assert future.result(timeout=3) == "completed" + assert calls["heartbeat"] > 2 + assert calls["active"] > 2 + assert run.stopped + + +def test_unreachable_monitor_expires_its_last_grant_even_when_driver_heartbeats_continue( + execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(execution, "_LEASE", 0.3) + request = execution._rpc + failed_polls = 0 + + def lose_monitor(*args: Any, **kwargs: Any) -> Any: + nonlocal failed_polls + if args[2] == "active": + failed_polls += 1 + raise TimeoutError("worker cannot reach scheduler") + return request(*args, **kwargs) + + monkeypatch.setattr(execution, "_rpc", lose_monitor) + started, stopped = tmp_path / "started", tmp_path / "stopped" + with execution.invocation(execution_client) as run: + future = run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + with pytest.raises(execution.ExecutionCancelled, match="cancelled"): + future.result(timeout=3) + assert failed_polls > 1 # A single failed RPC did not revoke valid authorization. + assert stopped.exists() + + +@pytest.mark.parametrize("operation", ["revoke", "forget"]) +def test_completed_results_survive_metadata_cleanup_rpc_failure( + execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, operation: str, +) -> None: + request = execution._rpc + effects = tmp_path / "effects" + + def lose_cleanup(*args: Any, **kwargs: Any) -> Any: + if args[2] == operation: + raise TimeoutError("scheduler unavailable during metadata cleanup") + return request(*args, **kwargs) + + with execution.invocation(execution_client) as run: + result = run.submit(_effect, effects, key="recipe", resources=_RESOURCES).result(timeout=3) + monkeypatch.setattr(execution, "_rpc", lose_cleanup) + assert result.status == "ok" + assert effects.read_text() == "executed\n" + assert run.stopped + + +def test_caught_submit_failure_cannot_manufacture_completion_from_an_empty_future_list( + execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + request, submit = execution._rpc, execution_client.submit + effects = tmp_path / "effects" + + def accept_then_fail(*args: Any, **kwargs: Any) -> Any: + future = submit(*args, **kwargs) + future.result(timeout=3) + raise TimeoutError("submission accepted but its handle was lost") + + def lose_revoke(*args: Any, **kwargs: Any) -> Any: + if args[2] == "revoke": + raise TimeoutError("cannot query accepted submissions") + return request(*args, **kwargs) + + monkeypatch.setattr(execution_client, "submit", accept_then_fail) + monkeypatch.setattr(execution, "_rpc", lose_revoke) + with pytest.raises(execution.ExecutionUncertain, match="confirm execution stopped"): + with execution.invocation(execution_client) as run: + with pytest.raises(TimeoutError, match="handle was lost"): + run.submit(_effect, effects, key="recipe", resources=_RESOURCES) + assert not run.futures + assert effects.read_text() == "executed\n" + assert not run.stopped + + +@pytest.mark.parametrize("submit", [False, True]) +def test_positive_task_completion_preserves_driver_error_during_cleanup_outage( + execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, submit: bool, +) -> None: + request = execution._rpc + failure = ValueError("driver could not commit") + + def lose_revoke(*args: Any, **kwargs: Any) -> Any: + if args[2] == "revoke": + raise TimeoutError("scheduler unavailable during cleanup") + return request(*args, **kwargs) + + with pytest.raises(ValueError, match="could not commit") as caught: + with execution.invocation(execution_client) as run: + if submit: + run.submit( + _effect, tmp_path / "effects", key="recipe", resources=_RESOURCES, + ).result() + monkeypatch.setattr(execution, "_rpc", lose_revoke) + raise failure + assert caught.value is failure + assert run.stopped + + +def test_confirmed_revocation_and_drain_remain_valid_when_forgetting_receipts_fails( + execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + request = execution._rpc + started, stopped = tmp_path / "started", tmp_path / "stopped" + interruption = KeyboardInterrupt() + + def lose_forget(*args: Any, **kwargs: Any) -> Any: + if args[2] == "forget": + raise TimeoutError("receipt cleanup reply lost") + return request(*args, **kwargs) + + monkeypatch.setattr(execution, "_rpc", lose_forget) + with pytest.raises(execution.ExecutionInterrupted) as caught: + with execution.invocation(execution_client) as run: + run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + _wait_for(started) + raise interruption + assert caught.value.__cause__ is interruption + assert stopped.exists() + assert run.stopped + + +def test_known_terminal_uncertainty_is_reported_without_polling_for_a_different_result( + execution_client: Client, monkeypatch: pytest.MonkeyPatch, +) -> None: + request = execution._rpc + + def refuse_polling(*args: Any, **kwargs: Any) -> Any: + if args[2] == "pending": + pytest.fail("a terminal uncertain receipt cannot become confirmed by waiting") + return request(*args, **kwargs) + + monkeypatch.setattr(execution, "_rpc", refuse_polling) + with pytest.raises(execution.ExecutionUncertain, match="unconfirmed tasks: recipe"): + with execution.invocation(execution_client) as run: + future = run.submit( + _raise, execution.ExecutionUncertain("container still unaccounted for"), + key="recipe", resources=_RESOURCES, + ) + with pytest.raises(execution.ExecutionUncertain, match="unaccounted"): + future.result(timeout=3) + assert not run.stopped diff --git a/tests/test_execution_processes.py b/tests/test_execution_processes.py index 4b28fed8..d476586f 100644 --- a/tests/test_execution_processes.py +++ b/tests/test_execution_processes.py @@ -13,9 +13,9 @@ import psutil import pytest -from lightcone.engine.execution import ExecutionCancelled, ExecutionUncertain from lightcone.engine.sandbox import Policy, Unavailable, run -from lightcone.engine.sandbox.processes import Command +from lightcone.engine.sandbox.model import ExecutionCancelled, ExecutionUncertain +from lightcone.engine.sandbox.processes import CIDFILE, Command def test_finished_group_is_reaped_without_signalling(monkeypatch: pytest.MonkeyPatch) -> None: @@ -59,6 +59,28 @@ def test_permission_error_with_a_live_group_member_remains_an_error( process.wait.assert_not_called() +def test_kill_confirmation_uses_the_remaining_cleanup_budget( + monkeypatch: pytest.MonkeyPatch, +) -> None: + from lightcone.engine.sandbox import processes + + clock = [0.0] + process = Mock(spec=subprocess.Popen, pid=1234) + signal_group = Mock() + monkeypatch.setattr(processes.time, "monotonic", lambda: clock[0]) + monkeypatch.setattr( + processes.time, "sleep", lambda seconds: clock.__setitem__(0, clock[0] + seconds), + ) + monkeypatch.setattr(processes, "members", lambda **kwargs: [object()] if clock[0] < 2.6 else []) + monkeypatch.setattr(processes.os, "killpg", signal_group) + assert processes._drain(process, deadline=10) + assert [call.args[1] for call in signal_group.call_args_list] == [ + signal.SIGTERM, signal.SIGKILL, + ] + assert 2.6 <= clock[0] < 3 + process.wait.assert_called_once_with() + + def _policy(root: Path) -> Policy: return Policy(read=(root,), write=(root,), execute=(), tmp_home=root) @@ -78,7 +100,7 @@ def test_timeout_escalates_ignoring_command(tmp_path: Path) -> None: "import os,signal,time; from pathlib import Path; " "signal.signal(signal.SIGTERM, signal.SIG_IGN); " f"Path({str(pidfile)!r}).write_text(str(os.getpid())); time.sleep(60)" - )], cwd=tmp_path, env=dict(os.environ), timeout=0.2, + )], cwd=tmp_path, env=dict(os.environ), timeout=1, ) assert outcome.returncode == 124 assert any("timed out" in note for note in outcome.notes) @@ -102,6 +124,21 @@ def test_cancel_stops_only_its_command(tmp_path: Path) -> None: unrelated.wait() +def test_interrupt_cleanup_does_not_wait_for_the_original_task_timeout( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + command = Command( + [sys.executable, "-c", "import time; time.sleep(60)"], cwd=tmp_path, + env=dict(os.environ), capture=False, timeout=60, + ) + wait = Mock(wraps=command.process.wait) + monkeypatch.setattr(command.process, "poll", Mock(side_effect=KeyboardInterrupt)) + monkeypatch.setattr(command.process, "wait", wait) + with pytest.raises(KeyboardInterrupt): + command.wait() + assert wait.call_args.kwargs["timeout"] <= 16 + + def test_successful_leader_cannot_leave_a_background_writer(tmp_path: Path) -> None: pidfile = tmp_path / "child" child = ( @@ -122,6 +159,24 @@ def test_successful_leader_cannot_leave_a_background_writer(tmp_path: Path) -> N assert _gone(int(pidfile.read_text())) +def test_short_lived_helpers_can_finish_after_the_command(tmp_path: Path) -> None: + helper = ( + "import time; from pathlib import Path; " + "Path('ready').touch(); time.sleep(.2); Path('finished').touch()" + ) + outcome = run( + Unavailable(), _policy(tmp_path), + [sys.executable, "-c", ( + "import subprocess,sys,time; from pathlib import Path; " + f"subprocess.Popen([sys.executable, '-c', {helper!r}]); " + "\nwhile not Path('ready').exists(): time.sleep(.01)" + )], cwd=tmp_path, env=dict(os.environ), + ) + assert outcome.returncode == 0 + assert (tmp_path / "finished").exists() + assert not any("background processes" in note for note in outcome.notes) + + def test_worker_sigkill_closes_custody_pipe_and_stops_child(tmp_path: Path) -> None: pidfile = tmp_path / "child" command = ( @@ -170,10 +225,13 @@ def test_stdout_bytes_are_unchanged_by_custody(tmp_path: Path) -> None: assert b"".join(received) == b"\xff\r\n" -def test_container_timeout_uses_its_immutable_runtime_id(tmp_path: Path) -> None: +@pytest.mark.parametrize("inspection_delay", [0, 2.2]) +def test_container_timeout_uses_its_immutable_runtime_id( + tmp_path: Path, inspection_delay: float, +) -> None: runtime = tmp_path / "runtime" runtime.write_text(f"#!{sys.executable}\n" + ''' -import json, os, signal, subprocess, sys +import json, os, signal, subprocess, sys, time from pathlib import Path import psutil root = Path(os.environ['STATE_ROOT']) @@ -187,9 +245,10 @@ def test_container_timeout_uses_its_immutable_runtime_id(tmp_path: Path) -> None start_new_session=True) (root / 'payload').write_text(str(process.pid)) Path(argv[argv.index('--cidfile') + 1]).write_text(identity) - (root / 'cidfile').write_text(argv[argv.index('--cidfile') + 1]) + (root / 'cidfile').write_text(str(Path(argv[argv.index('--cidfile') + 1]).resolve())) process.wait() elif argv[0] == 'inspect': + time.sleep(float(os.environ['INSPECTION_DELAY'])) try: alive = psutil.Process(int((root / 'payload').read_text())).status() != psutil.STATUS_ZOMBIE except psutil.NoSuchProcess: @@ -202,9 +261,9 @@ def test_container_timeout_uses_its_immutable_runtime_id(tmp_path: Path) -> None ''') runtime.chmod(0o700) command = Command( - [str(runtime), "run", "image"], cwd=tmp_path, - env={**os.environ, "STATE_ROOT": str(tmp_path)}, capture=False, - container=True, timeout=0.4, + [str(runtime), "run", "--cidfile", CIDFILE, "image"], cwd=tmp_path, + env={**os.environ, "STATE_ROOT": str(tmp_path), "INSPECTION_DELAY": str(inspection_delay)}, + capture=False, oci_runtime=str(runtime), timeout=1, ) try: code, note = command.wait() diff --git a/tests/test_execution_resources.py b/tests/test_execution_resources.py index c73f7a37..3ce96793 100644 --- a/tests/test_execution_resources.py +++ b/tests/test_execution_resources.py @@ -48,7 +48,7 @@ def test_integral_astra_float_cpu_count_is_accepted_without_rounding() -> None: [ {"cpus": 0}, {"cpus": True}, {"cpus": "4"}, {"cpus": None}, {"memory": "0Gi"}, {"memory": "0.1B"}, {"memory": "16"}, {"memory": 16}, - {"memory": "400m"}, + {"memory": "400m"}, {"memory": None}, {"time_limit": "0m"}, {"time_limit": ""}, {"time_limit": "5m2h"}, {"time_limit": "unlimited"}, {"gpus": 1}, {"disk": "10Gi"}, {"ram": "1Gi"}, ], @@ -67,6 +67,13 @@ def test_internal_resource_models_remain_validated() -> None: TaskResources(time_seconds=0) +def test_recipe_memory_uses_exact_bytes_without_decimal_context_rounding() -> None: + one_byte = "0.000000000931322574615478515625" + assert TaskResources.parse({"memory": f"{one_byte}Gi"}).memory_bytes == 1 + with pytest.raises(ProjectError, match="exactly representable"): + TaskResources.parse({"memory": f"{one_byte}00000000000000001Gi"}) + + def test_declared_requests_reserve_exact_cpu_and_memory_budgets() -> None: task = TaskResources.parse({"cpus": 4, "memory": "6Gi"}) assert task.requirements(_workers((8, 16 * GIB))) == {"CPU": 4, "MEMORY": 6 * GIB} diff --git a/tests/test_materialize.py b/tests/test_materialize.py index a86b5dee..ebd2d8fe 100644 --- a/tests/test_materialize.py +++ b/tests/test_materialize.py @@ -473,7 +473,9 @@ def digest(path: Path) -> str: return real(path) class _Copied(_Inline): - def submit(self, fn: Callable[..., object], *args: object, key: str) -> object: + def submit( + self, fn: Callable[..., object], *args: object, key: str, resources: dict[str, float], + ) -> object: return fn(*pickle.loads(pickle.dumps(args))) monkeypatch.setattr(assets, "data_version", digest) @@ -1116,6 +1118,30 @@ def unexpected(*args: object, **kwargs: object) -> None: assert not (root / "results/baseline/first.txt").exists() +@pytest.mark.parametrize("resource_spec", [ + "gpus: 1", "disk: 1Gi", "cpus: 0.5", "cpus: 0", "memory: null", +]) +def test_execution_only_resources_do_not_block_read_only_commands( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, resource_spec: str, +) -> None: + spec = _SPEC.replace( + "command: cat", f"resources: {{{resource_spec}}}\n command: cat", + ) + root = analysis(spec, universes={"baseline": _UNIVERSE}) + assert len(engine.status(root).outputs) == 2 + assert set(engine.check(root, []).planned) == {"baseline/first", "baseline/second"} + + _resource_cluster(monkeypatch) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("preparation began before execution requirements were validated") + + monkeypatch.setattr(engine, "_fetch_inputs", unexpected) + with pytest.raises(ProjectError, match="baseline/second"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert not dataset.status(root) + + def test_empty_cluster_refuses_before_project_preparation( root: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: diff --git a/tests/test_plan.py b/tests/test_plan.py index a182b9e1..9f069d38 100644 --- a/tests/test_plan.py +++ b/tests/test_plan.py @@ -130,21 +130,29 @@ def test_recipe_resources_survive_graph_resolution(tmp_path: Path) -> None: ) graph = _build(_project(tmp_path, spec)) task = graph.tasks[("baseline", "fit")] - assert task.resources.cpus == 4 - assert task.resources.memory_bytes == 6 * 1024**3 - assert task.resources.time_seconds == 5400 - unspecified = graph.tasks[("baseline", "report")].resources - assert unspecified.cpus == 1 - assert unspecified.memory_bytes is None - - -def test_unhonored_recipe_resources_refuse_graph_construction(tmp_path: Path) -> None: + assert task.resources == {"cpus": 4, "memory": "6Gi", "time_limit": "1h30m"} + assert graph.tasks[("baseline", "report")].resources == {} + + +@pytest.mark.parametrize( + ("declaration", "expected"), + [ + ("gpus: 1", {"gpus": 1}), + ("disk: 10Gi", {"disk": "10Gi"}), + ("cpus: 0.5", {"cpus": 0.5}), + ("cpus: 0", {"cpus": 0}), + ("memory: null", {"memory": None}), + ], +) +def test_execution_support_does_not_limit_graph_construction( + tmp_path: Path, declaration: str, expected: dict[str, object], +) -> None: spec = _SPEC.replace( "command: python src/fit.py", - "resources: {gpus: 1}\n command: python src/fit.py", + f"resources: {{{declaration}}}\n command: python src/fit.py", ) - with pytest.raises(ProjectError, match="unsupported recipe resource.*gpus"): - _build(_project(tmp_path, spec)) + task = _build(_project(tmp_path, spec)).tasks[("baseline", "fit")] + assert task.resources == expected def test_a_declared_input_resolves_to_its_source(tmp_path: Path) -> None: @@ -305,4 +313,3 @@ def test_an_output_without_a_format_is_refused_by_name(tmp_path: Path) -> None: - diff --git a/tests/test_run.py b/tests/test_run.py index 8412faa5..03a0cabe 100644 --- a/tests/test_run.py +++ b/tests/test_run.py @@ -309,7 +309,8 @@ def test_a_remote_task_exception_is_an_engine_error_and_leaves_compute_available monkeypatch.setattr(container, "backend", lambda _: Unavailable()) with pytest.raises(ProjectError, match="cluster execution failed") as raised: engine_run.probe(project, ["true"], cluster_id=cluster_id) - assert "did not stop the allocation" in str(raised.value) + assert "may still be running" not in str(raised.value) + assert "Stop the allocation" not in str(raised.value) with compute.connect(cluster_id) as client: assert client.scheduler_info()["workers"] diff --git a/tests/test_sandbox_oci.py b/tests/test_sandbox_oci.py index 26e89b18..b4074ff5 100644 --- a/tests/test_sandbox_oci.py +++ b/tests/test_sandbox_oci.py @@ -224,10 +224,12 @@ class _Recorder: def __init__(self, returncode: int = 0) -> None: self.argv: list[str] | None = None + self.options: dict[str, Any] = {} self.returncode = returncode def __call__(self, argv: list[str], **kwargs: Any) -> Any: self.argv = list(argv) + self.options = kwargs code = self.returncode class _Proc: @@ -271,6 +273,8 @@ def test_a_world_backend_takes_the_prefix_inside( assert recorder.argv is not None assert recorder.argv[0] == "podman" + assert recorder.options["oci_runtime"] == "podman" + assert "--cidfile" in recorder.argv assert recorder.argv[recorder.argv.index(_IMAGE_ID) + 1 :] == [ "uv", "run", "--locked", "--no-sync", "--project", str(root), "--", "bash", "-c", "true", @@ -296,6 +300,20 @@ def test_a_host_backend_keeps_the_prefix_outside( assert recorder.argv is not None assert recorder.argv[:3] == ["uv", "run", "--"] + assert recorder.options["oci_runtime"] is None + + +def test_a_world_backend_does_not_implicitly_get_oci_lifecycle( + root: Path, policy: Policy, monkeypatch: pytest.MonkeyPatch, +) -> None: + recorder = _Recorder(returncode=125) + monkeypatch.setattr(boundary, "Command", recorder) + outcome = boundary.run( + Unavailable(contains_prefix=True), policy, ["true"], cwd=root, env={}, + ) + assert recorder.options["oci_runtime"] is None + assert "--cidfile" not in recorder.argv + assert not any("runtime failed before the command ran" in note for note in outcome.notes) def test_exit_97_is_the_shims_only_under_landlock( From cab43f853facbcce484172794cafd47c4a782737 Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Mon, 28 Sep 2026 19:45:35 -0700 Subject: [PATCH 5/8] Close child pipe descriptors before startup error cleanup --- src/lightcone/engine/sandbox/processes.py | 26 ++++++------ tests/test_execution_processes.py | 48 +++++++++++++++++++++++ 2 files changed, 63 insertions(+), 11 deletions(-) diff --git a/src/lightcone/engine/sandbox/processes.py b/src/lightcone/engine/sandbox/processes.py index 8cf6f57b..921a6e2d 100644 --- a/src/lightcone/engine/sandbox/processes.py +++ b/src/lightcone/engine/sandbox/processes.py @@ -120,14 +120,21 @@ def __init__( if execution_deadline is not None else None ) try: - self.process = subprocess.Popen( - [sys.executable, "-P", str(Path(__file__)), str(self._control), str(self._status)], - pass_fds=(self._control, self._status), - stdin=subprocess.DEVNULL if capture else None, - stdout=subprocess.PIPE if capture else None, - stderr=subprocess.PIPE, - bufsize=0, - ) + try: + self.process = subprocess.Popen( + [sys.executable, "-P", str(Path(__file__)), + str(self._control), str(self._status)], + pass_fds=(self._control, self._status), + stdin=subprocess.DEVNULL if capture else None, + stdout=subprocess.PIPE if capture else None, + stderr=subprocess.PIPE, + bufsize=0, + ) + finally: + # Only the custodian owns these ends. In particular, keeping + # its report writer here would prevent EOF during failed setup. + os.close(self._control) + os.close(self._status) self._writer.write(json.dumps({ "argv": list(argv), "cwd": str(cwd), "env": env, "deadline": execution_deadline, "oci_runtime": oci_runtime, @@ -145,9 +152,6 @@ def __init__( else: self._reader.close() raise - finally: - os.close(self._control) - os.close(self._status) def __enter__(self) -> Self: return self diff --git a/tests/test_execution_processes.py b/tests/test_execution_processes.py index d476586f..67b99089 100644 --- a/tests/test_execution_processes.py +++ b/tests/test_execution_processes.py @@ -92,6 +92,54 @@ def _gone(pid: int) -> bool: return True +def test_failed_configuration_handoff_reaps_custodian_without_waiting_for_pipe_eof( + tmp_path: Path, +) -> None: + # Bound the reproduction in another process: the old constructor retained + # the report pipe's writer and blocked forever while reading its child reply. + result = subprocess.run( + [sys.executable, "-c", """ +import sys +from pathlib import Path +import psutil +from lightcone.engine.sandbox.processes import Command + +try: + Command([sys.executable, '-c', "open('started', 'w').close()"], + cwd=Path(sys.argv[1]), env={'INVALID': object()}, capture=False) +except OSError: + pass +else: + raise AssertionError('unserializable configuration was accepted') +assert not psutil.Process().children(), 'custodian was not reaped' +""", str(tmp_path)], capture_output=True, text=True, timeout=5, + ) + assert result.returncode == 0, result.stderr + assert not (tmp_path / "started").exists() + + +def test_custodian_spawn_failure_closes_all_pipe_descriptors( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + from lightcone.engine.sandbox import processes + + pipe = os.pipe + descriptors: list[int] = [] + + def record_pipe() -> tuple[int, int]: + pair = pipe() + descriptors.extend(pair) + return pair + + monkeypatch.setattr(processes.os, "pipe", record_pipe) + monkeypatch.setattr(processes.subprocess, "Popen", Mock(side_effect=OSError("spawn failed"))) + with pytest.raises(OSError, match="spawn failed"): + Command([sys.executable], cwd=tmp_path, env=dict(os.environ), capture=False) + for descriptor in descriptors: + with pytest.raises(OSError): + os.fstat(descriptor) + + def test_timeout_escalates_ignoring_command(tmp_path: Path) -> None: pidfile = tmp_path / "pid" outcome = run( From 7dcbd44b369b3370cdc8f65f3eee2362a8965c9f Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Mon, 28 Sep 2026 19:45:35 -0700 Subject: [PATCH 6/8] Separate task resource management from execution safety --- CLAUDE.md | 10 - docs/api/compute.md | 22 --- docs/api/index.md | 2 +- docs/api/materialize.md | 5 +- docs/api/plan.md | 18 +- docs/api/worker.md | 12 +- docs/architecture.md | 7 +- docs/cli/compute.md | 3 +- docs/cli/materialize.md | 4 - docs/cli/run.md | 1 - docs/user/cluster.md | 44 +---- src/lightcone/cli/compute.py | 11 +- src/lightcone/engine/compute/local_runtime.py | 2 - src/lightcone/engine/compute/model.py | 23 ++- .../engine/compute/slurm_bootstrap.py | 1 - src/lightcone/engine/execution.py | 4 +- src/lightcone/engine/execution_resources.py | 150 --------------- src/lightcone/engine/materialize.py | 29 +-- src/lightcone/engine/plan.py | 7 +- src/lightcone/engine/run.py | 6 +- src/lightcone/engine/units.py | 45 ----- src/lightcone/engine/worker.py | 6 +- tests/conftest.py | 10 +- tests/test_compute.py | 30 --- tests/test_execution.py | 58 +++--- tests/test_execution_resources.py | 136 ------------- tests/test_materialize.py | 178 +----------------- tests/test_plan.py | 35 +--- 28 files changed, 73 insertions(+), 786 deletions(-) delete mode 100644 src/lightcone/engine/execution_resources.py delete mode 100644 src/lightcone/engine/units.py delete mode 100644 tests/test_execution_resources.py diff --git a/CLAUDE.md b/CLAUDE.md index a7f8c95a..b12f28f9 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1785,16 +1785,6 @@ Any driver failure while tasks are outstanding (a failed commit included) first drains the invocation. Unconfirmed cleanup raises `ExecutionUncertain` and retains partial outputs; completing cleanup does not terminate the reusable allocation. -**Recipe resources use standard Dask admission.** Preserve ASTRA `recipe.resources` -in `plan.Task` as raw mappings so `status` and `--check` remain independent of -executor support. Parse `TaskResources` at execution admission: whole CPUs, memory -bytes, and optional command walltime. Validate the whole selected graph before -preparation or submission, then pass reservations explicitly to submission. -Workers advertise CPU/MEMORY; tasks reserve their declarations, with omitted RAM -reserving a whole worker's memory and probes reserving both whole-worker budgets. -Thread slots remain a separate concurrency cap. Reservations are cooperative, not -per-command OS CPU/RAM limits; unsupported GPU/disk requests fail explicitly. - **One catalog selector, `LC_COMPUTE_CONFIG` (2026-09).** `lc compute --config` was removed: `run` and `materialize` resolve clusters through the catalog too, and a per-invocation override on one command group launched allocations those diff --git a/docs/api/compute.md b/docs/api/compute.md index 0c5b667b..8150bcb9 100644 --- a/docs/api/compute.md +++ b/docs/api/compute.md @@ -114,28 +114,6 @@ unintended reuse across commands. There is no worker-selection layer, per-worker preflight orchestration, source fingerprinting, or login-node guard. Driver-side preparation and the existing task runtime/sandbox checks remain in their owners. -Workers advertise standard Dask `CPU` and `MEMORY` resources; memory is measured -in bytes. `engine.execution_resources.TaskResources` validates ASTRA's -`recipe.resources` into whole CPUs, bytes, and optional walltime seconds at -execution admission. `plan.Task` preserves the ASTRA mapping so read-only -classification does not impose executor restrictions. `requirements(workers)` -checks that one worker can satisfy it and returns the resource dictionary used -by `Client.submit`. -An omitted memory request reserves the full homogeneous worker budget; -`whole_worker=True` reserves CPU and memory for a probe. Unsupported GPU/disk -requests and fractional CPUs fail before execution. - -The materialize scheduler validates every selected task before preparation or -submission, preventing earlier tasks from starting before a later impossible -request is discovered, then passes each task's reservation explicitly to -submission. Allocation and task requests share byte and duration conversion -utilities; their models remain distinct because allocation selection supports -minimum quantities and node counts. Standard Dask scheduling accounts for -concurrent CPU and memory reservations; Dask execution-thread counts remain a -separate concurrency cap. Reservations do not impose hard limits on recipe -subprocesses. Task walltime uses the subprocess boundary's timeout and teardown, -independent of the native allocation's lifetime. - `output.py` transports byte chunks through standard Dask events so detached workers' output reaches the invoking CLI. It uses the borrowed client's event topic, which the schedulers lc launches drop as soon as the client disconnects diff --git a/docs/api/index.md b/docs/api/index.md index 2702c160..89bc8b47 100644 --- a/docs/api/index.md +++ b/docs/api/index.md @@ -18,7 +18,7 @@ is responsibility and contract, not every signature. | [`worker`](worker.md) | Making one output; the rerun entry point | impure | | [`materialize`](materialize.md) | The driver: gates, scheduling, the save/restore loop, status | impure | | [`compute`](compute.md) | Resource requests, native allocation lifecycle, borrowed Dask clients | impure | -| [`execution` & `execution_resources`](compute.md) | Invocation claims, completion receipts, cleanup confirmation, task resource admission | mixed | +| [`execution`](compute.md) | Invocation claims, completion receipts, cleanup confirmation | mixed | | [`sandbox`](sandbox.md) | The exec boundary: policy, backends, attestation, denials | mixed | | [`image` & `container`](container.md) | The container hatch: declaration → image → archive → runtime | pure / impure | | [`crate`](crate.md) | The publication view: the repo as an RO-Crate | pure | diff --git a/docs/api/materialize.md b/docs/api/materialize.md index 47fe87a9..f010dc9e 100644 --- a/docs/api/materialize.md +++ b/docs/api/materialize.md @@ -18,7 +18,7 @@ driver's stderr, independently of success or failure, leaving stdout for the rep | `check(root, targets, *, refresh)` | The same classification without executing, committing, or fetching. Exempt from the dirty refusal. | | `status(root)` | The report: every output's state and provenance commit, plus the mode/image/sandbox header facts. | | `MaterializeReport` / `StatusReport` | The JSON surfaces; `ok` and `up_to_date` first. | -| `cluster_for_run(cluster_id)` | Borrow the cluster; expose resource validation, submission, completion, and positive cleanup confirmation. | +| `cluster_for_run(cluster_id)` | Borrow the cluster; expose submission, completion, and positive cleanup confirmation. | | `run_record(...)` / `datalad_run_subject(...)` | The commit message `datalad rerun` replays, and the one spelling of its subject line — shared with the foreign-write comparator, because two strings here would drift. | | `_engine_requirement()` | How a record pins its engine: by version for a release, by source commit (hatch-vcs) for a dev build. | @@ -27,8 +27,7 @@ driver's stderr, independently of success or failure, leaving stdout for the rep 1. **Read-only project checks before connecting** — tool, committer, dirty-tree, spec and lock errors do not require a reachable cluster to report. 2. **Explicit cluster before preparing the environment** — validate native - allocation identity, connect, and validate every selected task's CPU/memory - request before fetching inputs or building an image. + allocation identity and connect before fetching inputs or building an image. The dirty refusal has already run: in containerized mode the converge can commit an image archive, and `dataset.save` commits the whole index; on a dirty tree the user's diff --git a/docs/api/plan.md b/docs/api/plan.md index abc1641e..93cf8e35 100644 --- a/docs/api/plan.md +++ b/docs/api/plan.md @@ -3,8 +3,8 @@ The spec, read as a graph of tasks. `astra.yaml` × `universes/*.yaml` gives one task per `(universe, output)` pair that has a recipe; a task carries everything executing it needs — the rendered command, where its -bytes go, what it reads, its decisions, its `definition_version`, and its -resource requirements — and nothing about *how* it will be executed. +bytes go, what it reads, its decisions, its `definition_version` — and +nothing about *how* it will be executed. Source: `src/lightcone/engine/plan.py`. @@ -14,7 +14,7 @@ Source: `src/lightcone/engine/plan.py`. |---|---| | `build(root)` | Validate the spec with ASTRA's own validators, resolve every universe, return the `Graph`. | | `Graph` | Tasks keyed on `(universe_id, output_id)`; `order()` for the read-only topological walk, `resolve(targets)` for what a user typed, `closure(keys)` to narrow a run. | -| `Task` | One output in one universe, frozen, retaining ASTRA's resource declaration in `resources`. | +| `Task` | One output in one universe, frozen. | | `declared_path(root, path)` | The one rule that names a path: project-relative inside the tree, absolute outside, never resolved. | ## What must stay true @@ -31,14 +31,6 @@ Source: `src/lightcone/engine/plan.py`. schema, file, and universe validators before resolving anything — resolution answers what a *valid* spec means and does not re-check that it is one. -- **Resource declarations survive resolution.** `build` reads - `recipe.resources` from ASTRA's resolved output definition and preserves the - mapping. A valid declaration remains readable by `status` and - `materialize --check` even when this executor cannot honor it. Execution - validates supported requirements through `TaskResources.parse` and checks - cluster capacity before materialize prepares the project or submits any - task. No worker placement or executor-specific resource validation belongs - in this module. - **The layout is flat and path-addressed.** `results//.`, and the path in a rendered recipe *is* the path on disk — no staging, no relocation. @@ -58,8 +50,8 @@ Source: `src/lightcone/engine/plan.py`. ## Tests `tests/test_plan.py` — pure; tests what lc *adds* (directories, edges, -versions, resource preservation, the validation gate), never what a spec means — -that coverage lives in astra-tools' own suite, and re-asserting it here +versions, the validation gate), never what a spec means — that +coverage lives in astra-tools' own suite, and re-asserting it here would recreate the second implementation this module deleted. Every fixture must be a spec `astra validate` accepts; the gate enforces it for free. diff --git a/docs/api/worker.md b/docs/api/worker.md index 3451b332..8397e01d 100644 --- a/docs/api/worker.md +++ b/docs/api/worker.md @@ -18,17 +18,14 @@ Source: `src/lightcone/engine/worker.py`. Cluster execution supplies an output receiver to `materialize`/`execute`, which passes byte chunks from the sandbox back to the invocation. Standalone reruns -retain direct terminal output. The driver submits each cluster task with its -CPU and memory reservations; `execute` applies the task's walltime limit through -the sandbox boundary. Standalone reruns also apply that time limit, but do not -perform Dask resource admission. +retain direct terminal output. ## Key symbols | Symbol | Role | |---|---| | `materialize(root, task, context, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns ordinary failures as `TaskResult`; propagates execution safety exceptions. | -| `execute(root, task, input_versions, context)` | Run a recipe unconditionally with its time limit and cancellation checks, then record its payload and manifest. | +| `execute(root, task, input_versions, context)` | Run a recipe unconditionally with cancellation checks, then record its payload and manifest. | | `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, reason, and diagnostic notes. `.usable` is what dependents check. | | `main(argv)` | The rerun entry point: guards, converges the project environment from the commit's own lock, resolves its own HEAD and runtime, executes. | | `lc_version()` | The engine version every manifest records. | @@ -41,10 +38,7 @@ perform Dask resource admission. not enter the ordinary failed-output restore path: cleanup first establishes that writers have stopped, and uncertainty retains partial outputs. - **Task completion includes subprocess teardown.** The boundary owns process - and container cleanup, applies the parsed resource request's time limit, and reports - uncertain teardown as an exception. A time limit that stops the recipe - becomes an ordinary failed result. CPU and memory reservations are standard - Dask scheduling constraints, not OS limits imposed by this module. + and container cleanup and reports uncertain teardown as an exception. - **`data_version` is computed here, before anything is staged** — the dependent's argument *is* this return value, so the digest must exist while the files are still unannexed. Deriving it from diff --git a/docs/architecture.md b/docs/architecture.md index 66b2f411..79fb1b7e 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -41,7 +41,6 @@ lc materialize "$CLUSTER" │ plan: astra validate + resolve → Graph of Tasks │ (no tasks → converge the crate and stop; nothing connects) │ connect: native identity + Dask readiness - │ admit: each task's CPU and memory request fits a worker │ fetch: git annex get (declared inputs not in this clone) │ converge: uv.lock ⇄ .venv (and the image, containerized) ├─► workers: reset output file → sandbox → recipe → hash → manifest @@ -66,10 +65,6 @@ The division of labor is strict and load-bearing: Cancellation and uncertain execution propagate instead, aborting the invocation. Partial outputs are restored only after writers are confirmed stopped; uncertainty retains those files for inspection. -- **Dask accounts for task resources.** Workers advertise CPU and memory - budgets; submissions reserve the recipe's requirements. The subprocess - boundary applies time limits and owns process teardown. CPU and memory - reservations coordinate scheduling rather than imposing per-recipe OS limits. - **Values are resolved once and handed down.** HEAD, the container runtime, and the foreign-write facts are read by the driver and passed to workers as values — a worker that asked git itself could @@ -172,7 +167,7 @@ material, not a registry. `compute.connect(CLUSTER_ID)` borrows a standard Dask client and closes only that client on exit. Both execution commands require a cluster ID. The materialization -scheduler validates resource requests, then keeps its `submit`/`completed` seam; +scheduler keeps its `submit`/`completed` seam; ordinary Dask scheduling places the tasks. Driver preparation and task runtime checks remain in their existing owners. `engine.execution` holds invocation claims and receipts in the scheduler, revokes admission on cancellation, and waits for diff --git a/docs/cli/compute.md b/docs/cli/compute.md index d4b806c8..e4d697eb 100644 --- a/docs/cli/compute.md +++ b/docs/cli/compute.md @@ -55,8 +55,7 @@ connections are unavailable. No name registry is maintained. CPU quantities are logical CPUs **per node**, memory is **GiB per node**, and `--num-nodes` defaults to one. Bare quantities are exact; `4+` means at least four. -Time accepts positive durations with day/hour/minute/second units, such as `30m`, -`1h30m`, or `45s`. Without +Time accepts positive whole minutes or hours, such as `30m` or `2h`. Without `--time`, the chosen offer's default applies. `fast` is a service class, not a queue-time promise. Limits apply to each allocation; aggregate quotas remain with the native backend. diff --git a/docs/cli/materialize.md b/docs/cli/materialize.md index 9e78a049..1dab5729 100644 --- a/docs/cli/materialize.md +++ b/docs/cli/materialize.md @@ -54,10 +54,6 @@ never touched, under any flag. uncertain, outputs stay in place: stop the allocation by its full ID and verify its commands and containers have stopped before repairing results. See [execution limits](../user/cluster.md#execution-requirements-and-limits). -- **Honors recipe resources.** CPU and memory requests must fit one worker and - are reserved through standard Dask scheduling; time limits stop overrunning - commands. The whole selected graph is checked before preparation or submission. - See [recipe resource requirements](../user/cluster.md#recipe-resource-requirements). - **Fetches what it needs.** Declared inputs whose annexed content is not in this clone are fetched before anything hashes. - **Commits as it goes.** Each output lands in its own commit, written diff --git a/docs/cli/run.md b/docs/cli/run.md index 63d9bd3a..b6baa96f 100644 --- a/docs/cli/run.md +++ b/docs/cli/run.md @@ -40,7 +40,6 @@ that variable to an existing cluster. Set command-specific values inside the command, for example `lc run "$CLUSTER" -- env NAME=value python script.py`. Containerized commands use the image's environment and the sandbox overlays. -The command reserves one worker's full CPU and memory budgets for its duration. Interrupting the CLI requests cancellation and waits for command cleanup; the cluster remains available. If cleanup cannot be confirmed, the error says so. Stop the allocation using `lc compute down` with its full ID (names can be reused) diff --git a/docs/user/cluster.md b/docs/user/cluster.md index ccd2bcc7..15d1b671 100644 --- a/docs/user/cluster.md +++ b/docs/user/cluster.md @@ -143,8 +143,7 @@ list: Catalog errors identify the invalid field, for example `offers.0.resources.cpus`. Unknown common fields and duplicate YAML keys are rejected. CPU and node counts must be positive integers; memory is in GiB and may be fractional if it is an -exact number of bytes. Durations use ordered day/hour/minute/second units, such as -`30m`, `1h30m`, or `45s`. +exact number of bytes, and durations use minutes or hours such as `30m` or `2h`. Selection takes the first offer in catalog order that matches the request. An offer this host cannot provide is skipped: a local offer with more nodes, CPUs or @@ -310,47 +309,6 @@ has as long to start. A failed check or timeout logs `Slurm Dask startup failed: …` to the submission log and exits nonzero. Look there when a job is active but never becomes ready. -## Recipe resource requirements - -Declare each recipe's needs in `astra.yaml`: - -```yaml -recipe: - command: python src/fit.py {output} - resources: - cpus: 4 - memory: 8Gi - time_limit: 1h30m -``` - -Each recipe runs on one worker. Its CPU and memory request must fit that -worker, even when the cluster has several nodes. Dask reserves both budgets -while the task runs, so recipes can run together only when their combined -requests fit. `task_slots_per_node` also caps concurrent tasks; it does not -limit how many CPUs a single recipe may request. - -CPUs must be positive whole numbers and default to one. Memory needs units: -`512Mi` and `8Gi` are binary sizes; `8GB` is decimal. Without a memory -declaration, a recipe reserves the worker's entire memory budget, so only -one such recipe runs per worker. `lc run` reserves an entire worker's CPU and -memory budgets because its arbitrary command has no recipe declaration. -Time limits accept combinations such as `1h30m` or `45s`; exceeding the limit -stops the recipe and reports failure. Fractional CPUs, GPUs, and disk requests -are rejected rather than ignored. - -`lc materialize` validates the complete selected graph against the cluster -before fetching inputs, preparing the environment, or starting a recipe. This -also validates currently complete outputs, which workers may need to rebuild -after an upstream change. Use `lc materialize --check` to inspect currency -without allocation. Read-only `status` and `--check` accept valid ASTRA resource -declarations even when this executor cannot satisfy them. - -These are scheduling reservations, not per-recipe CPU or RAM enforcement. -Recipes must respect their declarations; a subprocess can otherwise exceed -its request. Slurm enforces the overall allocation, while local execution -uses cooperative budgets. Leave capacity for the scheduler, workers, and other -overhead when declaring recipe requirements. - ## Execution requirements and limits Driver and workers must see the same project, prepared environment, and inputs diff --git a/src/lightcone/cli/compute.py b/src/lightcone/cli/compute.py index 03427693..59fb2517 100644 --- a/src/lightcone/cli/compute.py +++ b/src/lightcone/cli/compute.py @@ -49,11 +49,6 @@ def _table(headers: list[str], rows: list[list[str]]) -> None: Console(markup=False).print(table) -def _duration(seconds: int) -> str: - minutes, remainder = divmod(seconds, 60) - return (f"{minutes}m" if minutes else "") + (f"{remainder}s" if remainder else "") - - @click.group() def compute() -> None: """Allocate resources, inspect clusters, and end allocations. @@ -82,8 +77,8 @@ def resources(as_json: bool) -> None: str(offer["resources"]["cpus"]), f"{offer['resources']['memory']:g} GiB", str(offer["max_nodes"]), - _duration(offer["time"]["default_seconds"]), - _duration(offer["time"]["max_seconds"]), + f"{offer['time']['default_seconds'] // 60}m", + f"{offer['time']['max_seconds'] // 60}m", offer["startup"], ] for offer in data["offers"] @@ -97,7 +92,7 @@ def resources(as_json: bool) -> None: @click.option("--memory", required=True, help="GiB per node; suffix + requests a minimum.") @click.option("--num-nodes", default=1, type=click.IntRange(min=1), show_default=True) @click.option( - "--time", "walltime", help="Requested walltime, e.g. 30m or 1h30m; defaults to the offer." + "--time", "walltime", help="Requested walltime, e.g. 30m or 2h; defaults to the offer." ) @click.option( "--startup", type=click.Choice(["fast"]), help="Require a fast startup service class." diff --git a/src/lightcone/engine/compute/local_runtime.py b/src/lightcone/engine/compute/local_runtime.py index 2eed287e..515555df 100644 --- a/src/lightcone/engine/compute/local_runtime.py +++ b/src/lightcone/engine/compute/local_runtime.py @@ -93,7 +93,6 @@ def expire(_signum: int, _frame: FrameType | None) -> None: from distributed import LocalCluster security = create_security(directory) - allocation = read_private_json(directory / "identity.json") with dask.config.set(SCHEDULER_CONFIG), LocalCluster( # type: ignore[no-untyped-call] n_workers=1, threads_per_worker=int(launch["task_slots"]), @@ -113,7 +112,6 @@ def expire(_signum: int, _frame: FrameType | None) -> None: # Recipes use subprocesses: Dask's Python-process RSS cannot enforce # their RAM envelope. Local resource limits are explicitly cooperative. memory_limit=0, - resources={"CPU": int(allocation["cpus"]), "MEMORY": int(allocation["memory"])}, silence_logs=50, ) as cluster: write_private_json( diff --git a/src/lightcone/engine/compute/model.py b/src/lightcone/engine/compute/model.py index bc078130..ba7a25d1 100644 --- a/src/lightcone/engine/compute/model.py +++ b/src/lightcone/engine/compute/model.py @@ -7,7 +7,7 @@ import re from collections.abc import Callable, Sequence from contextlib import AbstractContextManager -from decimal import Decimal, localcontext +from decimal import Decimal, InvalidOperation, localcontext from typing import Annotated, Any, Literal, Protocol, Self from uuid import UUID @@ -23,7 +23,6 @@ ) from lightcone.engine.project import ProjectError -from lightcone.engine.units import duration_seconds, whole_bytes GIB = 1024**3 @@ -44,11 +43,11 @@ class UnavailableOfferError(ComputeError): def duration(value: object) -> int: - """Parse an explicit positive duration into seconds.""" - try: - return duration_seconds(value) - except ValueError as exc: - raise ComputeError(str(exc)) from exc + """Parse an explicit positive whole-minute/hour duration into seconds.""" + match = re.fullmatch(r"([1-9][0-9]*)([mh])", str(value)) + if match is None: + raise ComputeError("duration must be a positive number of minutes or hours, e.g. 30m or 1h") + return int(match[1]) * (60 if match[2] == "m" else 3600) def memory_bytes(value: object) -> int: @@ -56,9 +55,13 @@ def memory_bytes(value: object) -> int: if isinstance(value, bool) or not re.fullmatch(r"[0-9]+(?:\.[0-9]+)?", str(value)): raise ComputeError("memory must be a positive number of GiB") try: - return whole_bytes(str(value), GIB) - except ValueError as exc: - raise ComputeError("memory must be positive GiB exactly representable in bytes") from exc + numerator, denominator = Decimal(str(value)).as_integer_ratio() + except InvalidOperation as exc: + raise ComputeError("memory must be a positive number of GiB") from exc + amount, remainder = divmod(numerator * GIB, denominator) + if amount <= 0 or remainder: + raise ComputeError("memory must be positive GiB exactly representable in bytes") + return amount def gib_from_bytes(value: int) -> Decimal: diff --git a/src/lightcone/engine/compute/slurm_bootstrap.py b/src/lightcone/engine/compute/slurm_bootstrap.py index e184d728..6e35cdf0 100644 --- a/src/lightcone/engine/compute/slurm_bootstrap.py +++ b/src/lightcone/engine/compute/slurm_bootstrap.py @@ -101,7 +101,6 @@ async def run(args: argparse.Namespace) -> None: **address, "nthreads": args.task_slots, "memory_limit": 0, - "resources": {"CPU": args.cpus, "MEMORY": args.memory_bytes}, "local_directory": str(scratch), "dashboard_address": "127.0.0.1:0", "dashboard": False, diff --git a/src/lightcone/engine/execution.py b/src/lightcone/engine/execution.py index de8d27c8..19ad195f 100644 --- a/src/lightcone/engine/execution.py +++ b/src/lightcone/engine/execution.py @@ -206,7 +206,7 @@ class Invocation: _submissions: int = 0 def submit( - self, function: Callable[..., Any], *args: Any, key: str, resources: dict[str, float], + self, function: Callable[..., Any], *args: Any, key: str, ) -> Any: """Claim each logical task inside its worker before it can mutate files.""" # A submit failure may follow native acceptance. Count it before the @@ -214,7 +214,7 @@ def submit( self._submissions += 1 future = self.client.submit( _call, self.id, key, function, *args, - key=f"lc-{self.id}-{key}", pure=False, retries=0, resources=resources, + key=f"lc-{self.id}-{key}", pure=False, retries=0, ) self.futures.append(future) return future diff --git a/src/lightcone/engine/execution_resources.py b/src/lightcone/engine/execution_resources.py deleted file mode 100644 index 8c21ad81..00000000 --- a/src/lightcone/engine/execution_resources.py +++ /dev/null @@ -1,150 +0,0 @@ -"""Recipe resource requests and admission to stock Dask workers.""" - -from __future__ import annotations - -import math -import re -from typing import Any, Self - -from pydantic import BaseModel, ConfigDict, Field, ValidationError, field_validator - -from lightcone.engine.project import ProjectError -from lightcone.engine.units import duration_seconds, whole_bytes - - -class TaskResources(BaseModel): - """Reserve CPUs and memory for a recipe, with an optional walltime limit.""" - - model_config = ConfigDict(frozen=True, strict=True, extra="forbid") - - cpus: int = Field(default=1, gt=0) - memory_bytes: int | None = Field(default=None, gt=0) - time_seconds: int | None = Field(default=None, gt=0) - - @field_validator("cpus", mode="before") - @classmethod - def _whole_cpus(cls, value: object) -> object: - # ASTRA permits fractional CPUs. This executor reserves whole CPUs; - # accepting 4.0 is exact, whereas rounding 0.5 would hide a policy change. - if isinstance(value, float): - if not math.isfinite(value) or not value.is_integer(): - raise ValueError("fractional CPUs are not supported; request whole CPUs") - return int(value) - return value - - @classmethod - def parse(cls, value: object) -> Self: - """Parse ASTRA's ``recipe.resources`` into explicit execution units. - - Args: - value: The recipe resource mapping, or ``None`` when omitted. - - Returns: - A validated CPU, byte, and second request. - - Raises: - ProjectError: A requirement is invalid or cannot be honored. - """ - if value is None: - return cls() - if not isinstance(value, dict): - raise ProjectError("recipe.resources must be a mapping") - if extra := value.keys() - {"cpus", "memory", "time_limit"}: - names = ", ".join(sorted(map(str, extra))) - raise ProjectError(f"unsupported recipe resource requirements: {names}") - parsed = {"cpus": value.get("cpus", 1)} - if "memory" in value: - parsed["memory_bytes"] = _memory(value["memory"]) - if "time_limit" in value: - parsed["time_seconds"] = _duration(value["time_limit"]) - try: - return cls.model_validate(parsed) - except ValidationError as exc: - detail = "; ".join( - f"{'.'.join(map(str, item['loc']))}: {item['msg']}" - for item in exc.errors(include_url=False, include_input=False) - ) - raise ProjectError(f"invalid recipe resources: {detail}") from exc - - def requirements( - self, workers: dict[str, Any], *, whole_worker: bool = False - ) -> dict[str, float]: - """Choose Dask resource reservations that fit an individual worker. - - Args: - workers: The ``workers`` mapping from Dask's scheduler information. - whole_worker: Reserve a worker's entire CPU and memory budget for - an arbitrary command without declared resource requirements. - - Returns: - Dask's numeric ``CPU`` and ``MEMORY`` resource requirements. - - Raises: - ProjectError: Capacity is unknown, a request cannot fit, or an - unspecified budget is ambiguous across heterogeneous workers. - """ - capacities: set[tuple[float, float]] = set() - for info in workers.values(): - resources = info.get("resources", {}) if isinstance(info, dict) else {} - values = [] - for name in ("CPU", "MEMORY"): - value = resources.get(name) if isinstance(resources, dict) else None - if ( - isinstance(value, bool) - or not isinstance(value, (int, float)) - or not math.isfinite(value) - or value <= 0 - or not float(value).is_integer() - ): - raise ProjectError( - "cluster workers must advertise positive whole CPU and MEMORY budgets; " - "relaunch the cluster with the current Lightcone installation" - ) - values.append(float(value)) - capacities.add((values[0], values[1])) - if not capacities: - raise ProjectError("cluster has no workers available for execution") - if (whole_worker or self.memory_bytes is None) and len(capacities) != 1: - raise ProjectError( - "unspecified task resources require workers with identical CPU and memory budgets" - ) - available_cpus, available_memory = next(iter(capacities)) - requested = { - "CPU": available_cpus if whole_worker else float(self.cpus), - "MEMORY": ( - available_memory - if whole_worker or self.memory_bytes is None - else float(self.memory_bytes) - ), - } - if not any( - cpus >= requested["CPU"] and memory >= requested["MEMORY"] - for cpus, memory in capacities - ): - raise ProjectError( - f"task needs {requested['CPU']:g} CPUs and " - f"{requested['MEMORY'] / 1024**3:g} GiB on one worker; " - "no worker in this cluster can satisfy that request" - ) - return requested - - -def _memory(value: object) -> int: - if not isinstance(value, str) or not ( - match := re.fullmatch(r"([0-9]+(?:\.[0-9]+)?)\s*([KMGTPE]i?B?|kB?|B)", value) - ): - raise ProjectError("recipe memory must include units, e.g. 512Mi, 16Gi, or 8GB") - unit = match[2].lower() - exponent = 0 if unit == "b" else "kmgtpe".index(unit[0]) + 1 - factor: int = (1024 if "i" in unit else 1000) ** exponent - try: - return whole_bytes(match[1], factor) - except ValueError as exc: - raise ProjectError(f"recipe {exc}") from exc - - -def _duration(value: object) -> int: - try: - return duration_seconds(value) - except ValueError as exc: - raise ProjectError(f"recipe time_limit: {exc}") from exc diff --git a/src/lightcone/engine/materialize.py b/src/lightcone/engine/materialize.py index 620e070a..3a6f4eeb 100644 --- a/src/lightcone/engine/materialize.py +++ b/src/lightcone/engine/materialize.py @@ -35,14 +35,13 @@ import functools import json import re -from collections.abc import Iterable, Iterator, Sequence +from collections.abc import Iterator, Sequence from contextlib import contextmanager from dataclasses import asdict, dataclass, field, replace from pathlib import Path from typing import TYPE_CHECKING, Any, Protocol from lightcone.engine import assets, container, dataset, execution, identity, plan, project, worker -from lightcone.engine.execution_resources import TaskResources from lightcone.engine.plan import Graph, Key, Task from lightcone.engine.project import ProjectError @@ -520,7 +519,6 @@ def materialize( scheduler: Scheduler | None = None try: with cluster_for_run(cluster_id) as scheduler: - requirements = scheduler.validate(graph.tasks.values()) _fetch_inputs(root, graph, report) # Materialize is one of the two verbs allowed to build the image (the # other is `lc build`); the probe and the rerun entry point only find @@ -582,7 +580,6 @@ def materialize( foreign[key], *[pending[dep] for dep in task.depends_on], key=_name(key), - resources=requirements[key], ) for result in scheduler.completed(list(pending.values())): @@ -657,18 +654,13 @@ def stopped(self) -> bool: """Whether this invocation positively confirmed all claimed tasks stopped.""" ... - def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: - """Validate all requests and return their Dask resource reservations.""" - ... - - def submit(self, fn: Any, *args: Any, key: str, resources: dict[str, float]) -> Any: + def submit(self, fn: Any, *args: Any, key: str) -> Any: """Schedule a call. Args: fn: The function to run. *args: Its arguments, upstream handles included. key: A display name for the task. - resources: Validated reservations for this task. Returns: A handle to pass to dependents. @@ -694,30 +686,19 @@ class _Dask: client: Any invocation: execution.Invocation output: Forwarder - workers: dict[str, Any] @property def stopped(self) -> bool: """Expose positive cleanup confirmation after the connection context exits.""" return self.invocation.stopped - def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: - """Require each selected task to fit a worker before any task starts.""" - requests = {} - for task in tasks: - try: - requests[task.key] = TaskResources.parse(task.resources).requirements(self.workers) - except ProjectError as exc: - raise ProjectError(f"{_name(task.key)}: {exc}") from exc - return requests - - def submit(self, fn: Any, *args: Any, key: str, resources: dict[str, float]) -> Any: + def submit(self, fn: Any, *args: Any, key: str) -> Any: """Submit an ordinary Dask task with a unique key and forwarded output.""" from lightcone.engine.compute.output import call return self.invocation.submit( call, fn, self.output.topic, key, *args, - key=key, resources=resources, + key=key, ) def completed(self, handles: list[Any]) -> Iterator[worker.TaskResult]: @@ -759,7 +740,7 @@ def cluster_for_run(cluster_id: str) -> Iterator[Scheduler]: forwarding(client, stdout="stderr") as output, execution.invocation(client) as invocation, ): - yield _Dask(client, invocation, output, client.scheduler_info()["workers"]) + yield _Dask(client, invocation, output) def _fetch_inputs(root: Path, graph: Graph, report: MaterializeReport) -> None: diff --git a/src/lightcone/engine/plan.py b/src/lightcone/engine/plan.py index 513f410c..46067016 100644 --- a/src/lightcone/engine/plan.py +++ b/src/lightcone/engine/plan.py @@ -23,10 +23,9 @@ from __future__ import annotations -from dataclasses import dataclass, field +from dataclasses import dataclass from graphlib import CycleError, TopologicalSorter from pathlib import Path -from typing import Any from lightcone.engine import assets, identity from lightcone.engine.project import SPEC_FILENAME, ProjectError @@ -52,8 +51,6 @@ class Task: produced_by: dict[str, Key] decisions: dict[str, str] definition_version: str - #: ASTRA's declaration; executor support is checked only when executing. - resources: dict[str, Any] = field(default_factory=dict) @property def manifest_path(self) -> Path: @@ -334,7 +331,6 @@ def file_of(out: object) -> Path: ) except ValueError as e: raise ProjectError(f"output `{out.id}`: {e}") from e - resources = dict((out.definition.get("recipe") or {}).get("resources") or {}) tasks.append( Task( @@ -348,7 +344,6 @@ def file_of(out: object) -> Path: definition_version=identity.definition_version( recipe=recipe, decisions=out.decisions, fmt=str(out.format) ), - resources=resources, ) ) return tasks diff --git a/src/lightcone/engine/run.py b/src/lightcone/engine/run.py index faeb10df..052d7649 100644 --- a/src/lightcone/engine/run.py +++ b/src/lightcone/engine/run.py @@ -20,7 +20,6 @@ from typing import Any from lightcone.engine import container, execution, sandbox -from lightcone.engine.execution_resources import TaskResources from lightcone.engine.project import ( SPEC_FILENAME, ProjectError, @@ -51,15 +50,12 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. require_uv() paths = input_paths(project, read_spec(project)) with compute.connect(cluster_id) as client: - resources = TaskResources().requirements( - client.scheduler_info()["workers"], whole_worker=True, - ) runtime = container.runtime_for_run(project, build=False) notes = [f"uv: {warning}" for warning in container.converge(runtime)] with forwarding(client) as output, execution.invocation(client) as invocation: future = invocation.submit( call, _probe, output.topic, "probe", runtime, paths, tuple(command), - key="probe", resources=resources, + key="probe", ) try: outcome: sandbox.Outcome = future.result() diff --git a/src/lightcone/engine/units.py b/src/lightcone/engine/units.py deleted file mode 100644 index 2d47130a..00000000 --- a/src/lightcone/engine/units.py +++ /dev/null @@ -1,45 +0,0 @@ -"""Exact byte quantities and explicit walltime units used by execution and compute.""" - -from __future__ import annotations - -import re -from decimal import Decimal, InvalidOperation - - -def whole_bytes(amount: str, unit_bytes: int) -> int: - """Scale a decimal memory amount into positive bytes without rounding. - - Args: - amount: A decimal quantity in the caller's units. - unit_bytes: The number of bytes represented by one unit. - - Raises: - ValueError: The quantity is not positive or exactly representable in bytes. - """ - try: - numerator, denominator = Decimal(amount).as_integer_ratio() - except (InvalidOperation, ValueError, OverflowError) as exc: - raise ValueError("memory must be positive and exactly representable in bytes") from exc - result, remainder = divmod(numerator * unit_bytes, denominator) - if result <= 0 or remainder: - raise ValueError("memory must be positive and exactly representable in bytes") - return result - - -def duration_seconds(value: object) -> int: - """Parse a positive duration such as ``30m``, ``1h30m``, or ``45s``. - - Args: - value: Ordered day, hour, minute, and second components. - - Raises: - ValueError: The duration is malformed, empty, or zero. - """ - if not isinstance(value, str) or not ( - match := re.fullmatch(r"(?:(\d+)d)?(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?", value) - ): - raise ValueError("duration must use explicit units, e.g. 30m, 1h30m, or 45s") - seconds = sum(int(part or 0) * unit for part, unit in zip(match.groups(), (86400, 3600, 60, 1))) - if seconds <= 0: - raise ValueError("duration must be positive") - return seconds diff --git a/src/lightcone/engine/worker.py b/src/lightcone/engine/worker.py index d0d0c8df..68e4aa79 100644 --- a/src/lightcone/engine/worker.py +++ b/src/lightcone/engine/worker.py @@ -36,7 +36,6 @@ from typing import Literal from lightcone.engine import assets, container, dataset, execution, identity, plan, project, sandbox -from lightcone.engine.execution_resources import TaskResources from lightcone.engine.plan import Key, Task from lightcone.engine.project import ( ProjectError, @@ -220,8 +219,7 @@ def execute( committed as part of an output that never produced it. The context's ``env_version`` is checked either side of the recipe, so a mid-run lock edit cannot be recorded as if it had been in force. The subprocess - boundary applies the recipe's time limit and drains owned processes - before reporting completion. CPU and memory scheduling happens upstream. + boundary drains owned processes before reporting completion. Args: root: The project root. @@ -240,7 +238,6 @@ def execute( ExecutionUncertain: Subprocess teardown could not be confirmed. """ execution.check_cancelled() - resources = TaskResources.parse(task.resources) if moved := _gate(root, context.env_version): return TaskResult(task.key, "failed", reason=moved) @@ -272,7 +269,6 @@ def execute( prefix=uv_prefix(root), env=child_env(), output=output, - timeout=resources.time_seconds, cancelled=execution.cancelled, ) finished_at = _now() diff --git a/tests/conftest.py b/tests/conftest.py index 8980a535..25c1cd0a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -5,7 +5,7 @@ import shutil import subprocess import textwrap -from collections.abc import Callable, Iterable, Iterator +from collections.abc import Callable, Iterator from contextlib import contextmanager from pathlib import Path from unittest.mock import MagicMock @@ -15,7 +15,6 @@ from lightcone.engine import dataset, project, templates from lightcone.engine.compute.model import Identity -from lightcone.engine.plan import Key, Task from lightcone.engine.project import _run as _real_run CLUSTER_ID = Identity( @@ -154,12 +153,8 @@ class _Inline: stopped = True # Calls are synchronous; no remote work can survive the fixture. - def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: - """Run fixture tasks without a finite cluster resource envelope.""" - return {task.key: {} for task in tasks} - def submit( - self, fn: Callable[..., object], *args: object, key: str, resources: dict[str, float], + self, fn: Callable[..., object], *args: object, key: str, ) -> object: return fn(*args) @@ -190,7 +185,6 @@ def cluster_id(monkeypatch: pytest.MonkeyPatch) -> Iterator[str]: with LocalCluster( n_workers=1, threads_per_worker=2, processes=False, dashboard_address=None, - resources={"CPU": 2, "MEMORY": 1024**3}, ) as cluster: @contextmanager def connect(value: str) -> Iterator[Client]: diff --git a/tests/test_compute.py b/tests/test_compute.py index dac26101..7e56ab99 100644 --- a/tests/test_compute.py +++ b/tests/test_compute.py @@ -29,7 +29,6 @@ Startup, TimeLimits, UnavailableOfferError, - duration, memory_bytes, validate_name, ) @@ -365,25 +364,6 @@ def test_memory_conversion_does_not_round_fractional_bytes() -> None: memory_bytes("0.000000000931322574615478515625000000000000000001") -@pytest.mark.parametrize( - ("value", "seconds"), - [("1h30m", 5400), ("45s", 45), ("2d3h4m5s", 183845)], -) -def test_allocation_and_recipe_durations_use_the_same_units(value: str, seconds: int) -> None: - from lightcone.engine.execution_resources import TaskResources - - assert duration(value) == seconds - assert TimeLimits(default=value, max="3d").default_seconds == seconds - assert Request.parse("1", "1", time=value).seconds == seconds - assert TaskResources.parse({"time_limit": value}).time_seconds == seconds - - -@pytest.mark.parametrize("value", ["", "0s", "1.5h", "30m1h", "1h30", 60, True]) -def test_allocation_duration_refuses_ambiguous_or_zero_values(value: object) -> None: - with pytest.raises(ComputeError): - duration(value) - - @pytest.mark.parametrize("size", [1, GIB // 2, 8 * GIB, 2**80 + 1]) def test_resource_units_survive_construction_serialization_and_updates(size: int) -> None: whole, fraction = divmod(size, GIB) @@ -757,16 +737,6 @@ def test_cli_resources_dry_run_launch_down(catalog: Path, provider: MagicMock) - provider.terminate.assert_called_once_with(IDENTITY) -def test_cli_resources_preserves_seconds(catalog: Path) -> None: - data = yaml.safe_load(catalog.read_text()) - data["offers"][0]["time"] = {"default": "45s", "max": "1m30s"} - catalog.write_text(yaml.safe_dump(data)) - result = CliRunner().invoke(main, ["compute", "resources"]) - assert result.exit_code == 0, result.output - assert "45s" in result.output - assert "1m30s" in result.output - - def test_cli_launch_name_output_can_be_captured_without_json( catalog: Path, provider: MagicMock, ) -> None: diff --git a/tests/test_execution.py b/tests/test_execution.py index b8649cad..431bc356 100644 --- a/tests/test_execution.py +++ b/tests/test_execution.py @@ -15,8 +15,6 @@ from lightcone.engine import execution from lightcone.engine.worker import TaskResult -_RESOURCES = {"CPU": 1.0, "MEMORY": 1.0} - @pytest.fixture def execution_client(monkeypatch: pytest.MonkeyPatch) -> Iterator[Client]: @@ -25,7 +23,7 @@ def execution_client(monkeypatch: pytest.MonkeyPatch) -> Iterator[Client]: monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.75) with LocalCluster( n_workers=2, threads_per_worker=1, processes=False, - dashboard_address=None, memory_limit=0, resources=_RESOURCES, + dashboard_address=None, memory_limit=0, ) as cluster, Client(cluster) as client: yield client @@ -108,13 +106,13 @@ def test_completed_task_replay_on_another_worker_returns_its_original_receipt( effects = tmp_path / "effects" with execution.invocation(execution_client) as run: assert not run.stopped - original = run.submit(_effect, effects, key="recipe", resources=_RESOURCES).result() + original = run.submit(_effect, effects, key="recipe").result() other = next(address for address in execution_client.scheduler_info()["workers"] if address != original.notes[0]) replay = execution_client.submit( execution._call, run.id, "recipe", _effect, effects, key=f"replay-{uuid4().hex}", workers=[other], allow_other_workers=False, - pure=False, resources=_RESOURCES, + pure=False, ).result() assert replay == original assert effects.read_text() == "executed\n" @@ -127,7 +125,7 @@ def test_worker_loss_recomputes_the_dask_future_without_repeating_completed_effe ) -> None: effects = tmp_path / "effects" with execution.invocation(execution_client) as run: - future = run.submit(_effect, effects, key="recipe", resources=_RESOURCES) + future = run.submit(_effect, effects, key="recipe") original = future.result(timeout=3) lost_worker = original.notes[0] _remove_worker(execution_client, lost_worker) @@ -148,7 +146,7 @@ def test_worker_loss_cannot_replay_effects_while_the_original_execution_still_ru started, stopped = tmp_path / "started", tmp_path / "stopped" with execution.invocation(execution_client) as run: future = run.submit( - _cooperating_effect, started, stopped, key="recipe", resources=_RESOURCES, + _cooperating_effect, started, stopped, key="recipe", ) _wait_for(started.with_suffix(".ready")) lost_worker = started.read_text().strip() @@ -168,13 +166,13 @@ def test_duplicate_running_attempt_cannot_execute_or_finish_the_original_claim( ) with execution.invocation(execution_client) as run: original = run.submit( - _wait_for_release, started, release, key="recipe", resources=_RESOURCES, + _wait_for_release, started, release, key="recipe", ) try: _wait_for(started) duplicate = execution_client.submit( execution._call, run.id, "recipe", _effect, duplicate_effect, - key=f"duplicate-{uuid4().hex}", pure=False, resources=_RESOURCES, + key=f"duplicate-{uuid4().hex}", pure=False, ) with pytest.raises(execution.ExecutionUncertain, match="previous attempt"): duplicate.result(timeout=3) @@ -211,7 +209,7 @@ def test_missing_scheduler_state_refuses_both_replay_and_unstarted_tasks( ) -> None: effects = tmp_path / "effects" with execution.invocation(execution_client) as run: - run.submit(_effect, effects, key="recipe", resources=_RESOURCES).result() + run.submit(_effect, effects, key="recipe").result() execution_client.run_on_scheduler(_lose_state) for key in ("recipe", "new-recipe"): future = execution_client.submit( @@ -231,7 +229,7 @@ def test_missing_scheduler_state_stops_running_work_without_claiming_confirmed_c with pytest.raises(execution.ExecutionUncertain, match="confirm execution stopped"): with execution.invocation(execution_client) as run: future = run.submit( - _cooperate, started, stopped, key="recipe", resources=_RESOURCES, + _cooperate, started, stopped, key="recipe", ) _wait_for(started) execution_client.run_on_scheduler(_lose_state) @@ -256,7 +254,7 @@ def lose_claim_response(*args: Any, **kwargs: Any) -> Any: monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.1) with pytest.raises(execution.ExecutionUncertain, match="unconfirmed tasks: recipe"): with execution.invocation(execution_client) as run: - future = run.submit(_effect, effects, key="recipe", resources=_RESOURCES) + future = run.submit(_effect, effects, key="recipe") with pytest.raises(TimeoutError, match="reply lost"): future.result(timeout=3) assert execution._rpc(execution_client, run.id, "pending").running == ("recipe",) @@ -286,7 +284,7 @@ def test_invocation_exit_cancels_the_future_and_waits_for_task_cooperation( ) -> None: started, stopped = tmp_path / "started", tmp_path / "stopped" with execution.invocation(execution_client) as run: - future = run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + future = run.submit(_cooperate, started, stopped, key="recipe") _wait_for(started) assert future.cancelled() assert stopped.exists() @@ -300,7 +298,7 @@ def test_confirmed_cleanup_survives_an_error_in_the_invoking_driver( started, stopped = tmp_path / "started", tmp_path / "stopped" with pytest.raises(ValueError, match="driver failed to commit"): with execution.invocation(execution_client) as run: - run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + run.submit(_cooperate, started, stopped, key="recipe") _wait_for(started) raise ValueError("driver failed to commit") assert stopped.exists() @@ -314,7 +312,7 @@ def test_driver_disconnect_revokes_execution_even_while_an_observer_remains_conn with Client(execution_client.scheduler.address, set_as_default=False) as owner: run = execution.Invocation(owner) execution._rpc(owner, run.id, "register", value=owner.id) - run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + run.submit(_cooperate, started, stopped, key="recipe") _wait_for(started) _wait_for(stopped) assert not execution._rpc(execution_client, run.id, "heartbeat") @@ -330,7 +328,7 @@ def test_a_returned_failed_recipe_result_is_cached_without_rerunning( ) -> None: failed = TaskResult(("universe", "output"), "failed", reason="recipe exited 2") with execution.invocation(execution_client) as run: - assert run.submit(lambda: failed, key="recipe", resources=_RESOURCES).result() == failed + assert run.submit(lambda: failed, key="recipe").result() == failed replay = execution_client.submit( execution._call, run.id, "recipe", _raise, AssertionError("must not execute"), pure=False, @@ -342,7 +340,7 @@ def test_function_exception_confirms_stop_but_does_not_authorize_reexecution( execution_client: Client, ) -> None: with execution.invocation(execution_client) as run: - future = run.submit(_raise, ValueError("recipe failed"), key="recipe", resources=_RESOURCES) + future = run.submit(_raise, ValueError("recipe failed"), key="recipe") with pytest.raises(ValueError, match="recipe failed"): future.result(timeout=3) assert execution._rpc(execution_client, run.id, "pending").running == () @@ -362,7 +360,7 @@ def test_uncertain_task_remains_unresolved_after_context_exit( with execution.invocation(execution_client) as run: future = run.submit( _raise, execution.ExecutionUncertain("container may still be alive"), - key="recipe", resources=_RESOURCES, + key="recipe", ) with pytest.raises(execution.ExecutionUncertain, match="container may still"): future.result(timeout=3) @@ -372,7 +370,7 @@ def test_uncertain_task_remains_unresolved_after_context_exit( assert not run.stopped -def test_uncertain_cleanup_prevents_independent_tasks_from_using_released_dask_resources( +def test_uncertain_cleanup_prevents_later_tasks_from_starting( execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.1) @@ -381,13 +379,13 @@ def test_uncertain_cleanup_prevents_independent_tasks_from_using_released_dask_r with execution.invocation(execution_client) as run: first = run.submit( _raise, execution.ExecutionUncertain("container may still consume memory"), - key="first", resources=_RESOURCES, + key="first", ) with pytest.raises(execution.ExecutionUncertain, match="container may still"): first.result(timeout=3) - # Dask has released the first task's reservations, but its external - # work may survive. Authorization must close before another task runs. - later = run.submit(_effect, effects, key="later", resources=_RESOURCES) + # Dask reported the first task's error, but its external work may + # survive. Authorization must close before another task runs. + later = run.submit(_effect, effects, key="later") error = later.exception(timeout=3) assert isinstance(error, execution.ExecutionCancelled) assert not effects.exists() @@ -410,7 +408,7 @@ def fail_once(*args: Any, **kwargs: Any) -> Any: monkeypatch.setattr(execution, "_rpc", fail_once) with execution.invocation(execution_client) as run: - future = run.submit(_stay_authorized, 0.8, key="recipe", resources=_RESOURCES) + future = run.submit(_stay_authorized, 0.8, key="recipe") assert future.result(timeout=3) == "completed" assert calls["heartbeat"] > 2 assert calls["active"] > 2 @@ -434,7 +432,7 @@ def lose_monitor(*args: Any, **kwargs: Any) -> Any: monkeypatch.setattr(execution, "_rpc", lose_monitor) started, stopped = tmp_path / "started", tmp_path / "stopped" with execution.invocation(execution_client) as run: - future = run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + future = run.submit(_cooperate, started, stopped, key="recipe") with pytest.raises(execution.ExecutionCancelled, match="cancelled"): future.result(timeout=3) assert failed_polls > 1 # A single failed RPC did not revoke valid authorization. @@ -454,7 +452,7 @@ def lose_cleanup(*args: Any, **kwargs: Any) -> Any: return request(*args, **kwargs) with execution.invocation(execution_client) as run: - result = run.submit(_effect, effects, key="recipe", resources=_RESOURCES).result(timeout=3) + result = run.submit(_effect, effects, key="recipe").result(timeout=3) monkeypatch.setattr(execution, "_rpc", lose_cleanup) assert result.status == "ok" assert effects.read_text() == "executed\n" @@ -482,7 +480,7 @@ def lose_revoke(*args: Any, **kwargs: Any) -> Any: with pytest.raises(execution.ExecutionUncertain, match="confirm execution stopped"): with execution.invocation(execution_client) as run: with pytest.raises(TimeoutError, match="handle was lost"): - run.submit(_effect, effects, key="recipe", resources=_RESOURCES) + run.submit(_effect, effects, key="recipe") assert not run.futures assert effects.read_text() == "executed\n" assert not run.stopped @@ -504,7 +502,7 @@ def lose_revoke(*args: Any, **kwargs: Any) -> Any: with execution.invocation(execution_client) as run: if submit: run.submit( - _effect, tmp_path / "effects", key="recipe", resources=_RESOURCES, + _effect, tmp_path / "effects", key="recipe", ).result() monkeypatch.setattr(execution, "_rpc", lose_revoke) raise failure @@ -527,7 +525,7 @@ def lose_forget(*args: Any, **kwargs: Any) -> Any: monkeypatch.setattr(execution, "_rpc", lose_forget) with pytest.raises(execution.ExecutionInterrupted) as caught: with execution.invocation(execution_client) as run: - run.submit(_cooperate, started, stopped, key="recipe", resources=_RESOURCES) + run.submit(_cooperate, started, stopped, key="recipe") _wait_for(started) raise interruption assert caught.value.__cause__ is interruption @@ -550,7 +548,7 @@ def refuse_polling(*args: Any, **kwargs: Any) -> Any: with execution.invocation(execution_client) as run: future = run.submit( _raise, execution.ExecutionUncertain("container still unaccounted for"), - key="recipe", resources=_RESOURCES, + key="recipe", ) with pytest.raises(execution.ExecutionUncertain, match="unaccounted"): future.result(timeout=3) diff --git a/tests/test_execution_resources.py b/tests/test_execution_resources.py deleted file mode 100644 index 3ce96793..00000000 --- a/tests/test_execution_resources.py +++ /dev/null @@ -1,136 +0,0 @@ -"""Parse recipe requests and refuse impossible execution before submitting work.""" - -from __future__ import annotations - -from typing import Any - -import pytest -from pydantic import ValidationError - -from lightcone.engine.execution_resources import TaskResources -from lightcone.engine.project import ProjectError - -GIB = 1024**3 - - -def _workers(*capacities: tuple[int, int]) -> dict[str, Any]: - return { - f"worker-{index}": {"resources": {"CPU": cpus, "MEMORY": memory}} - for index, (cpus, memory) in enumerate(capacities) - } - - -@pytest.mark.parametrize( - ("memory", "expected"), - [("512Mi", 512 * 1024**2), ("1.5GiB", 3 * GIB // 2), ("8GB", 8_000_000_000), - ("1B", 1), ("2 Ti", 2 * 1024**4), ("1000kB", 1_000_000)], -) -def test_memory_units_have_explicit_decimal_or_binary_meaning(memory: str, expected: int) -> None: - assert TaskResources.parse({"memory": memory}).memory_bytes == expected - - -@pytest.mark.parametrize( - ("duration", "expected"), - [("1h30m", 5400), ("30m", 1800), ("2d3h4m5s", 183845), ("45s", 45)], -) -def test_task_walltime_accepts_compound_durations(duration: str, expected: int) -> None: - assert TaskResources.parse({"time_limit": duration}).time_seconds == expected - - -def test_integral_astra_float_cpu_count_is_accepted_without_rounding() -> None: - assert TaskResources.parse({"cpus": 4.0}).cpus == 4 - with pytest.raises(ProjectError, match="fractional CPUs"): - TaskResources.parse({"cpus": 0.5}) - - -@pytest.mark.parametrize( - "declaration", - [ - {"cpus": 0}, {"cpus": True}, {"cpus": "4"}, {"cpus": None}, - {"memory": "0Gi"}, {"memory": "0.1B"}, {"memory": "16"}, {"memory": 16}, - {"memory": "400m"}, {"memory": None}, - {"time_limit": "0m"}, {"time_limit": ""}, {"time_limit": "5m2h"}, - {"time_limit": "unlimited"}, {"gpus": 1}, {"disk": "10Gi"}, {"ram": "1Gi"}, - ], -) -def test_invalid_or_unhonored_declarations_are_not_silently_ignored( - declaration: dict[str, Any], -) -> None: - with pytest.raises(ProjectError): - TaskResources.parse(declaration) - - -def test_internal_resource_models_remain_validated() -> None: - with pytest.raises(ValidationError): - TaskResources(memory_bytes=-1) - with pytest.raises(ValidationError): - TaskResources(time_seconds=0) - - -def test_recipe_memory_uses_exact_bytes_without_decimal_context_rounding() -> None: - one_byte = "0.000000000931322574615478515625" - assert TaskResources.parse({"memory": f"{one_byte}Gi"}).memory_bytes == 1 - with pytest.raises(ProjectError, match="exactly representable"): - TaskResources.parse({"memory": f"{one_byte}00000000000000001Gi"}) - - -def test_declared_requests_reserve_exact_cpu_and_memory_budgets() -> None: - task = TaskResources.parse({"cpus": 4, "memory": "6Gi"}) - assert task.requirements(_workers((8, 16 * GIB))) == {"CPU": 4, "MEMORY": 6 * GIB} - - -def test_missing_memory_reserves_entire_worker_instead_of_guessing() -> None: - assert TaskResources().requirements(_workers((8, 16 * GIB), (8, 16 * GIB))) == { - "CPU": 1, "MEMORY": 16 * GIB, - } - - -def test_probe_reserves_an_entire_worker() -> None: - assert TaskResources().requirements(_workers((8, 16 * GIB)), whole_worker=True) == { - "CPU": 8, "MEMORY": 16 * GIB, - } - - -def test_cpu_count_is_independent_of_dask_execution_threads() -> None: - workers = _workers((8, 16 * GIB)) - workers["worker-0"]["nthreads"] = 1 - assert TaskResources(cpus=8).requirements(workers)["CPU"] == 8 - - -def test_a_task_must_fit_one_worker_not_the_sum_of_the_cluster() -> None: - with pytest.raises(ProjectError, match="on one worker"): - TaskResources(cpus=8, memory_bytes=20 * GIB).requirements( - _workers((4, 16 * GIB), (4, 16 * GIB)) - ) - - -def test_cpu_and_memory_must_fit_on_the_same_worker() -> None: - with pytest.raises(ProjectError, match="no worker"): - TaskResources(cpus=8, memory_bytes=16 * GIB).requirements( - _workers((8, 4 * GIB), (4, 16 * GIB)) - ) - - -def test_explicit_requests_can_select_a_fitting_worker() -> None: - assert TaskResources(cpus=8, memory_bytes=8 * GIB).requirements( - _workers((4, 4 * GIB), (8, 16 * GIB)) - ) == {"CPU": 8, "MEMORY": 8 * GIB} - - -@pytest.mark.parametrize("whole_worker", [False, True]) -def test_missing_budgets_are_not_guessed_for_heterogeneous_workers(whole_worker: bool) -> None: - with pytest.raises(ProjectError, match="identical"): - TaskResources().requirements( - _workers((4, 4 * GIB), (8, 16 * GIB)), whole_worker=whole_worker - ) - - -@pytest.mark.parametrize( - "workers", - [{}, {"worker": {}}, {"worker": {"resources": {"CPU": 1}}}, - {"worker": {"resources": {"CPU": True, "MEMORY": GIB}}}, - {"worker": {"resources": {"CPU": 1, "MEMORY": float("nan")}}}], -) -def test_absent_or_unknown_worker_capacity_refuses_execution(workers: dict[str, Any]) -> None: - with pytest.raises(ProjectError): - TaskResources().requirements(workers) diff --git a/tests/test_materialize.py b/tests/test_materialize.py index ebd2d8fe..743a45fe 100644 --- a/tests/test_materialize.py +++ b/tests/test_materialize.py @@ -474,7 +474,7 @@ def digest(path: Path) -> str: class _Copied(_Inline): def submit( - self, fn: Callable[..., object], *args: object, key: str, resources: dict[str, float], + self, fn: Callable[..., object], *args: object, key: str, ) -> object: return fn(*pickle.loads(pickle.dumps(args))) @@ -1079,128 +1079,6 @@ def test_the_recorded_command_holds_on_a_fresh_clone( # ---- the scheduler seam ---------------------------------------------------- -def _resource_cluster(monkeypatch: pytest.MonkeyPatch, *, workers: int = 1) -> None: - from distributed import Client, LocalCluster - - from lightcone.engine import compute - - @contextmanager - def connect(cluster_id: str) -> Iterator[Any]: - with LocalCluster( - n_workers=workers, threads_per_worker=4, processes=False, - dashboard_address=None, resources={"CPU": 4, "MEMORY": 2 * 1024**3}, - ) as cluster, Client(cluster, set_as_default=False) as client: - yield client - - monkeypatch.setattr(compute, "connect", connect) - - -@pytest.mark.parametrize("resource_spec", ["cpus: 5", "memory: 3Gi"]) -def test_resource_refusal_precedes_preparation_and_all_recipes( - analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, resource_spec: str, -) -> None: - spec = _SPEC.replace( - "command: cat", f"resources: {{{resource_spec}}}\n command: cat" - ) - root = analysis(spec, universes={"baseline": _UNIVERSE}) - before = dataset.head(root) - _resource_cluster(monkeypatch) - - def unexpected(*args: object, **kwargs: object) -> None: - pytest.fail("preparation began before all resource requests were validated") - - monkeypatch.setattr(engine, "_fetch_inputs", unexpected) - monkeypatch.setattr(engine.container, "runtime_for_run", unexpected) - with pytest.raises(ProjectError, match="baseline/second:.*no worker"): - engine.materialize(root, [], cluster_id=CLUSTER_ID) - assert dataset.head(root) == before - assert not dataset.status(root) - assert not (root / "results/baseline/first.txt").exists() - - -@pytest.mark.parametrize("resource_spec", [ - "gpus: 1", "disk: 1Gi", "cpus: 0.5", "cpus: 0", "memory: null", -]) -def test_execution_only_resources_do_not_block_read_only_commands( - analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, resource_spec: str, -) -> None: - spec = _SPEC.replace( - "command: cat", f"resources: {{{resource_spec}}}\n command: cat", - ) - root = analysis(spec, universes={"baseline": _UNIVERSE}) - assert len(engine.status(root).outputs) == 2 - assert set(engine.check(root, []).planned) == {"baseline/first", "baseline/second"} - - _resource_cluster(monkeypatch) - - def unexpected(*args: object, **kwargs: object) -> None: - pytest.fail("preparation began before execution requirements were validated") - - monkeypatch.setattr(engine, "_fetch_inputs", unexpected) - with pytest.raises(ProjectError, match="baseline/second"): - engine.materialize(root, [], cluster_id=CLUSTER_ID) - assert not dataset.status(root) - - -def test_empty_cluster_refuses_before_project_preparation( - root: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - _resource_cluster(monkeypatch, workers=0) - - def unexpected(*args: object, **kwargs: object) -> None: - pytest.fail("preparation began without an available worker") - - monkeypatch.setattr(engine, "_fetch_inputs", unexpected) - with pytest.raises(ProjectError, match="no workers"): - engine.materialize(root, [], cluster_id=CLUSTER_ID) - assert not dataset.status(root) - - -@pytest.mark.parametrize( - ("resource_spec", "expected_parallelism"), - [("cpus: 3, memory: 256Mi", 1), ("cpus: 1, memory: 1Gi", 2)], -) -def test_real_dask_respects_recipe_cpu_and_memory_reservations( - analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, - resource_spec: str, expected_parallelism: int, -) -> None: - # Four Dask threads would run all four subprocesses together without - # resource reservations. Each independent output records its live interval. - spec = 'version: "0.0.13"\nname: analysis\ninputs: []\noutputs:\n' + "".join( - f" - id: task{index}\n" - " type: metric\n" - " format: json\n" - " recipe:\n" - f" resources: {{{resource_spec}}}\n" - " command: python src/work.py {output}\n" - for index in range(4) - ) - root = analysis(spec, files={"src/work.py": """ - import json - import sys - import time - from pathlib import Path - start = time.monotonic() - time.sleep(0.5) - Path(sys.argv[1]).write_text(json.dumps([start, time.monotonic()])) - """}) - _resource_cluster(monkeypatch) - - report = engine.materialize(root, [], cluster_id=CLUSTER_ID) - - assert report.ok and len(report.made) == 4 - events = [] - for path in (root / "results/baseline").glob("task*.json"): - start, finish = json.loads(path.read_text()) - events.extend([(start, 1), (finish, -1)]) - live = peak = 0 - for _, change in sorted(events): - live += change - peak = max(peak, live) - assert peak == expected_parallelism - assert not dataset.status(root) - - def test_a_real_cluster_still_fits_through_the_seam(root: Path, cluster_id: str) -> None: """The one test that starts Dask. The seam is only worth having if the thing it abstracts still goes through it.""" @@ -1226,7 +1104,6 @@ def test_a_processes_cluster_fits_through_the_seam( def processes(cluster_id: str) -> Iterator[Any]: with LocalCluster( n_workers=2, threads_per_worker=1, processes=True, dashboard_address=None, - resources={"CPU": 1, "MEMORY": 1024**3}, ) as cluster: with Client(cluster, set_as_default=False) as client: yield client @@ -1462,56 +1339,3 @@ def test_an_output_the_spec_dropped_is_excluded_and_named(root: Path, inline: No assert any(".second.manifest.json" in w for w in report.warnings) document = (root / "ro-crate-metadata.json").read_text() assert "results/baseline/second.txt" not in document - - -def test_recipe_time_limit_stops_writes_restores_output_and_keeps_cluster_usable( - root: Path, cluster_id: str, capsys: pytest.CaptureFixture[str], -) -> None: - import psutil - - engine.materialize(root, ["first"], cluster_id=cluster_id) - output = root / "results/baseline/first.txt" - manifest = root / "results/baseline/.first.manifest.json" - original_output, original_manifest = output.read_bytes(), manifest.read_bytes() - - script = root / "src/slow.py" - script.parent.mkdir(exist_ok=True) - script.write_text( - "import os, sys, time\n" - "from pathlib import Path\n" - "Path(sys.argv[1]).write_text('partial output')\n" - "print('slow-recipe-pid:', os.getpid(), flush=True)\n" - "time.sleep(30)\n" - "Path(sys.argv[1]).write_text('should never finish')\n" - ) - spec = root / "astra.yaml" - spec.write_text(spec.read_text().replace( - "command: echo {decisions.method} > {output}", - "resources: {time_limit: 1s}\n command: python src/slow.py {output}", - )) - dataset.save(root, [spec, script], "Run the recipe with a walltime limit") - capsys.readouterr() - - failed = engine.materialize(root, ["first"], cluster_id=cluster_id) - - assert failed.failed == ["baseline/first"] - assert any("timed out" in note for note in failed.notes) - pid_line = next( - line for line in capsys.readouterr().err.splitlines() - if line.startswith("slow-recipe-pid:") - ) - assert not psutil.pid_exists(int(pid_line.split(":", 1)[1])) - assert output.read_bytes() == original_output - assert manifest.read_bytes() == original_manifest - assert not dataset.status(root) - - # The fixture holds the same LocalCluster across both invocations; a - # successful new recipe demonstrates that timeout did not terminate it. - spec.write_text(spec.read_text().replace( - "python src/slow.py {output}", "echo recovered > {output}", - )) - dataset.save(root, [spec], "Use a recipe that completes within its limit") - recovered = engine.materialize(root, ["first"], cluster_id=cluster_id) - assert recovered.made == ["baseline/first"] - assert output.read_text() == "recovered\n" - assert not dataset.status(root) diff --git a/tests/test_plan.py b/tests/test_plan.py index 9f069d38..48a994b0 100644 --- a/tests/test_plan.py +++ b/tests/test_plan.py @@ -122,39 +122,6 @@ def test_an_output_addresses_its_own_file(tmp_path: Path) -> None: assert "results/baseline/fit.json" in task.recipe -def test_recipe_resources_survive_graph_resolution(tmp_path: Path) -> None: - spec = _SPEC.replace( - "command: python src/fit.py", - "resources: {cpus: 4, memory: 6Gi, time_limit: 1h30m}\n" - " command: python src/fit.py", - ) - graph = _build(_project(tmp_path, spec)) - task = graph.tasks[("baseline", "fit")] - assert task.resources == {"cpus": 4, "memory": "6Gi", "time_limit": "1h30m"} - assert graph.tasks[("baseline", "report")].resources == {} - - -@pytest.mark.parametrize( - ("declaration", "expected"), - [ - ("gpus: 1", {"gpus": 1}), - ("disk: 10Gi", {"disk": "10Gi"}), - ("cpus: 0.5", {"cpus": 0.5}), - ("cpus: 0", {"cpus": 0}), - ("memory: null", {"memory": None}), - ], -) -def test_execution_support_does_not_limit_graph_construction( - tmp_path: Path, declaration: str, expected: dict[str, object], -) -> None: - spec = _SPEC.replace( - "command: python src/fit.py", - f"resources: {{{declaration}}}\n command: python src/fit.py", - ) - task = _build(_project(tmp_path, spec)).tasks[("baseline", "fit")] - assert task.resources == expected - - def test_a_declared_input_resolves_to_its_source(tmp_path: Path) -> None: task = _build(_project(tmp_path)).tasks[("baseline", "fit")] assert task.inputs == {"catalog": tmp_path / "data" / "catalog.fits"} @@ -313,3 +280,5 @@ def test_an_output_without_a_format_is_refused_by_name(tmp_path: Path) -> None: + + From 62414bb57d578fa31f6900d141fab81875276c44 Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Mon, 28 Sep 2026 20:09:30 -0700 Subject: [PATCH 7/8] Make lease recovery tests tolerate CI scheduling delays --- tests/test_execution.py | 49 ++++++++++++++++++++++++++++++++--------- 1 file changed, 38 insertions(+), 11 deletions(-) diff --git a/tests/test_execution.py b/tests/test_execution.py index 431bc356..46c6fb14 100644 --- a/tests/test_execution.py +++ b/tests/test_execution.py @@ -2,6 +2,7 @@ from __future__ import annotations +import threading import time from collections.abc import Iterator from pathlib import Path @@ -52,9 +53,9 @@ def _wait_for_release(started: Path, release: Path) -> str: return "original" -def _cooperate(started: Path, stopped: Path) -> None: +def _cooperate(started: Path, stopped: Path, timeout: float = 5) -> None: started.touch() - deadline = time.monotonic() + 5 + deadline = time.monotonic() + timeout while not execution.cancelled(): if time.monotonic() > deadline: raise AssertionError("task did not observe its invocation's cancellation") @@ -394,9 +395,13 @@ def test_uncertain_cleanup_prevents_later_tasks_from_starting( def test_transient_heartbeat_and_monitor_failures_preserve_the_confirmed_lease( execution_client: Client, monkeypatch: pytest.MonkeyPatch, ) -> None: - monkeypatch.setattr(execution, "_LEASE", 0.4) + # A CI runner may pause for longer than the old 400 ms test lease. Keep + # actual lease renewal under test without mistaking that pause for an outage. + lease = 3.0 + monkeypatch.setattr(execution, "_LEASE", lease) request = execution._rpc calls = {"heartbeat": 0, "active": 0} + recovered = {operation: threading.Event() for operation in calls} def fail_once(*args: Any, **kwargs: Any) -> Any: operation = args[2] @@ -404,12 +409,17 @@ def fail_once(*args: Any, **kwargs: Any) -> Any: calls[operation] += 1 if calls[operation] == 1: raise TimeoutError("temporary scheduler RPC failure") - return request(*args, **kwargs) + result = request(*args, **kwargs) + if operation in recovered and result: + recovered[operation].set() + return result monkeypatch.setattr(execution, "_rpc", fail_once) with execution.invocation(execution_client) as run: - future = run.submit(_stay_authorized, 0.8, key="recipe") - assert future.result(timeout=3) == "completed" + future = run.submit(_stay_authorized, 2 * lease, key="recipe") + for operation, event in recovered.items(): + assert event.wait(timeout=10), f"{operation} did not recover after its failed RPC" + assert future.result(timeout=10) == "completed" assert calls["heartbeat"] > 2 assert calls["active"] > 2 assert run.stopped @@ -418,23 +428,40 @@ def fail_once(*args: Any, **kwargs: Any) -> Any: def test_unreachable_monitor_expires_its_last_grant_even_when_driver_heartbeats_continue( execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: - monkeypatch.setattr(execution, "_LEASE", 0.3) + monkeypatch.setattr(execution, "_LEASE", 3.0) request = execution._rpc failed_polls = 0 + monitor_ready = threading.Event() + disconnected = threading.Event() + driver_renewed = threading.Event() def lose_monitor(*args: Any, **kwargs: Any) -> Any: nonlocal failed_polls - if args[2] == "active": + operation = args[2] + if operation == "active" and disconnected.is_set(): failed_polls += 1 raise TimeoutError("worker cannot reach scheduler") - return request(*args, **kwargs) + result = request(*args, **kwargs) + if operation == "active" and result: + monitor_ready.set() + if operation == "heartbeat" and disconnected.is_set() and result: + driver_renewed.set() + return result monkeypatch.setattr(execution, "_rpc", lose_monitor) started, stopped = tmp_path / "started", tmp_path / "stopped" with execution.invocation(execution_client) as run: - future = run.submit(_cooperate, started, stopped, key="recipe") + future = run.submit(_cooperate, started, stopped, 15, key="recipe") + _wait_for(started) + # Inject the partition only after startup and a confirmed worker grant. + assert monitor_ready.wait(timeout=10) + disconnected.set() + assert driver_renewed.wait(timeout=10) with pytest.raises(execution.ExecutionCancelled, match="cancelled"): - future.result(timeout=3) + future.result(timeout=10) + # The worker expired its own last grant while the scheduler still + # authorizes the invocation through successful driver heartbeats. + assert request(execution_client, run.id, "active") > 0 assert failed_polls > 1 # A single failed RPC did not revoke valid authorization. assert stopped.exists() From 0eb27e9c24bd9f676e6ac198f1efc346a9f152e4 Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Tue, 29 Sep 2026 03:36:28 -0700 Subject: [PATCH 8/8] Reduce execution safety to a scheduler replay guard Signed-off-by: Francois Lanusse --- CLAUDE.md | 48 +- docs/api/compute.md | 49 +- docs/api/index.md | 1 - docs/api/materialize.md | 12 +- docs/api/sandbox.md | 23 +- docs/api/worker.md | 16 +- docs/architecture.md | 24 +- docs/cli/materialize.md | 14 +- docs/cli/run.md | 11 +- docs/user/cluster.md | 44 +- src/lightcone/cli/commands.py | 28 +- src/lightcone/engine/compute/__init__.py | 8 + src/lightcone/engine/compute/local.py | 55 +- src/lightcone/engine/compute/local_runtime.py | 43 +- src/lightcone/engine/compute/output.py | 7 +- src/lightcone/engine/execution.py | 312 ++------- src/lightcone/engine/materialize.py | 194 +++--- src/lightcone/engine/run.py | 5 +- src/lightcone/engine/sandbox/boundary.py | 84 +-- src/lightcone/engine/sandbox/model.py | 10 - src/lightcone/engine/sandbox/oci.py | 8 +- src/lightcone/engine/sandbox/processes.py | 386 ----------- src/lightcone/engine/worker.py | 42 +- tests/conftest.py | 8 +- tests/test_cli.py | 12 +- tests/test_compute_local.py | 52 +- tests/test_compute_output.py | 76 +-- tests/test_container_smoke.py | 45 +- tests/test_execution.py | 619 ++++-------------- tests/test_execution_processes.py | 378 ----------- tests/test_materialize.py | 43 +- tests/test_run.py | 3 +- tests/test_sandbox_oci.py | 46 +- 33 files changed, 521 insertions(+), 2185 deletions(-) delete mode 100644 src/lightcone/engine/sandbox/processes.py delete mode 100644 tests/test_execution_processes.py diff --git a/CLAUDE.md b/CLAUDE.md index b12f28f9..548f3c12 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -996,16 +996,14 @@ same run created, and whether it did would depend on whether a recipe finished before or after the previous save. Nondeterminism in a provenance field is worse than either answer. -**Ordinary recipe failures are results; execution-safety failures propagate.** +**The worker never raises, and that is enforced at the unit boundary.** It returns `ok`, `current`, `behind`, `failed`, or `blocked`. A task whose upstream did not report — failed, or never finished at all — returns `blocked` without running. Raising would make Dask re-raise in the driver and abort every task in flight, and reporting all independent failures in one run is most of what owning the loop buys. `worker.materialize` wraps the -whole unit, so ordinary failures keep that contract. `ExecutionCancelled` and -`ExecutionUncertain` bypass it: reporting an uncertain writer as an ordinary -failure would let the driver restore files underneath it. The -one inner guard that remains exists because "your recipe failed" and +whole unit, so the contract holds for failure modes nobody enumerated; +the one inner guard that remains exists because "your recipe failed" and "your recipe worked and we could not record it" deserve different words. **`data_version` is computed in the worker, before anything is staged.** @@ -1197,12 +1195,12 @@ crash, or a Ctrl-C would otherwise leave tracked files deleted or half-written — and the next run's refusal would tell the user to commit truncated, manifest-less garbage into `results/`, destroying the one property the layer exists for. So `ok` → `dataset.save`, and `failed` or -`blocked` → `dataset.restore`. On interruption or driver failure, the invocation -first revokes admission and drains its claimed tasks. Restore unconsumed outputs -only with positive `Invocation.stopped` confirmation, independent of exception -type: a later context's error may mask uncertain cleanup. Otherwise leave files -in place and require native verification before repair. No cross-invocation -checkout lock is provided; run one execution invocation per project at a time. +`blocked` → `dataset.restore`. The one exception is an interrupted run: a +task that never reported may still have a recipe writing on the cluster, so +its files are left in place rather than restored underneath it, and the +dirty-tree refusal's `results/` block says to stop the allocation before +discarding them. This is what makes the refusal survivable rather than a +trap. **The run record names declared paths, never resolved ones.** Every declared input under `data/` is an annex symlink, so a `Path.resolve()` @@ -1768,22 +1766,24 @@ Walltime follows Slurm's native overrun and termination-grace policy; Lightcone does not independently guarantee a finite termination deadline for Slurm jobs. **Execution borrows a client and leaves the allocation alive.** Validate native -identity and scheduler readiness. The driver keeps git and convergence. Unique -invocation task keys plus claims and completion receipts in the existing Dask -scheduler prevent uncertain automatic recipe replay; missing state refuses work. -The driver renews authorization, then revokes and drains it before detaching. -Only positively confirmed cleanup permits restoring unconsumed outputs. A command -supervisor watches worker liveness through a pipe, drains the command group on -timeout/cancellation, and verifies native OCI container termination by immutable ID. -Recipes must not daemonize into other sessions. A hard-killed supervisor can leave -external containers alive; do not claim allocation termination proves otherwise. -Cross-invocation checkout locking remains deferred by explicit user decision. +identity and scheduler readiness. The driver keeps git and convergence. Use unique +invocation task keys. `engine.execution` requires an atomic scheduler claim before +each recipe or probe runs. A repeated claim or missing invocation state refuses +execution, including recomputation after a completed result is lost. There are no +completion receipts, leases, heartbeats, or automatic recovery. `retries=0` alone +does not prevent Dask recomputation after worker loss. This is an invocation-local +replay guard, not a project lock or proof that a disconnected command stopped. +Interrupted unreported outputs remain in place because a +client disconnect does not prove remote subprocess termination. Comprehensive +cancellation and simultaneous writers are deferred by explicit user decision. +Local containerized processes can outlive process-group shutdown; do not claim +that `down` or walltime proves an external runtime's containers have stopped. Read-only project validation precedes cluster connection, and a run with no tasks never connects: it only converges the crate. Populate the declared input-hash memo on the driver before serializing it to independent worker tasks. -Any driver failure while tasks are outstanding (a failed commit included) first -drains the invocation. Unconfirmed cleanup raises `ExecutionUncertain` and retains -partial outputs; completing cleanup does not terminate the reusable allocation. +Any driver failure while tasks are outstanding (a failed commit included, not +only a cluster error) carries `compute.UNSTOPPED`, the one wording for "the +allocation was not stopped and unreported tasks may still be running". **One catalog selector, `LC_COMPUTE_CONFIG` (2026-09).** `lc compute --config` was removed: `run` and `materialize` resolve clusters through the catalog too, diff --git a/docs/api/compute.md b/docs/api/compute.md index 8150bcb9..14f5b874 100644 --- a/docs/api/compute.md +++ b/docs/api/compute.md @@ -114,37 +114,28 @@ unintended reuse across commands. There is no worker-selection layer, per-worker preflight orchestration, source fingerprinting, or login-node guard. Driver-side preparation and the existing task runtime/sandbox checks remain in their owners. +`engine.execution` prevents automatic replay of side effects. The driver registers +an invocation in the existing Dask scheduler; before running a recipe or probe, +its worker must receive an atomic claim for that task. A repeated claim, missing +invocation state, or absent driver client fails before the command runs. This also +refuses recomputation when a completed result was lost: Lightcone does not store +or recover results. +`retries=0` alone cannot prevent [Dask's recomputation after worker loss](https://distributed.dask.org/en/stable/resilience.html). +There are no leases, heartbeats, command supervisors, or separate execution service. +This guard prevents automatic replay within an invocation; it does not lock the +project against another invocation or prove that a disconnected worker stopped. +The context removes its record on exit; later invocations prune records left by +disconnected clients. Removing a record cannot authorize a later claim. + `output.py` transports byte chunks through standard Dask events so detached workers' output reaches the invoking CLI. It uses the borrowed client's event topic, which the schedulers lc launches drop as soon as the client disconnects (`runtime.SCHEDULER_CONFIG`), rather than retaining a separate topic for every -command. Output-delivery errors cannot replace an execution-safety exception. -Probes preserve both streams; +command. A driver that exits before every task reports says so with +`UNSTOPPED`: closing a client cannot prove that a remote subprocess stopped. Probes preserve both streams; materialization sends recipe output to stderr to leave stdout for its report. -`engine.execution.invocation` owns a short renewable authorization in the existing -scheduler. Each task claims its logical key before touching files. Completion -receipts preserve the original result if Dask recomputes a lost result; a running -or uncertain claim refuses replay and revokes the invocation. Missing state also -refuses execution. This uses ordinary tasks and `run_on_scheduler`, without a -custom worker, service, project lock, or persistent execution registry. -Driver heartbeats and worker authorization polls retry transient RPC failures -within the last confirmed 15-second lease. A failed RPC does not extend that -lease; explicit revocation, missing state, or expiry stops execution. - -On exit the invocation revokes admission, cancels pending futures, and waits for -claimed tasks to acknowledge cleanup. Dask cancellation alone is insufficient: -running tasks poll authorization and the subprocess boundary stops their commands. -Only a positive `Invocation.stopped` flag permits restoring unconsumed outputs; -an exception from closing another context cannot manufacture that confirmation. -Without positive completion evidence, scheduler loss or an unacknowledged attempt -raises `ExecutionUncertain` and retains outputs. Known terminal uncertainty is -reported immediately. A finished task already confirms command cleanup and receipt -publication, so metadata cleanup failures cannot discard its result. Receipts are -removed best-effort after confirmed cleanup; uncertain records remain -until the allocation ends. They are not a recovery log for a later invocation. - -Local teardown drains the allocation's validated process session rather than +Local teardown drains the allocation's validated process group rather than assuming the owner's exit proves every child stopped. Boot UUID, UID, process session and the exact command containing a random allocation token establish identity without depending on hostname or wall-clock creation time. Discovery @@ -154,11 +145,9 @@ are cleaned up, and incomplete locator directories do not hide healthy allocatio An allocation verified as ended, by `down` or by discovery, is retired: its TLS material, scheduler files and scratch are removed, and a marker lets discovery skip it unread. Its identity record stays, so a full ID still reports `ended`. -Concurrent invocations writing the same project remain unsupported. Command -cleanup covers process groups and native OCI container identities; recipes must -not daemonize into new sessions. Killing the command supervisor can leave an -external runtime's container alive, so an uncertain execution requires native -verification before output repair. +Cancellation and concurrent project writers are not made safe by allocation +management; callers must respect the documented execution limits. Containers +managed outside that process group can survive local teardown. Tests cover deterministic selection, malformed identities and catalogs, partial native failures, acceptance ambiguity, PID reuse, detached local lifetime, standard diff --git a/docs/api/index.md b/docs/api/index.md index 89bc8b47..8674ea0b 100644 --- a/docs/api/index.md +++ b/docs/api/index.md @@ -18,7 +18,6 @@ is responsibility and contract, not every signature. | [`worker`](worker.md) | Making one output; the rerun entry point | impure | | [`materialize`](materialize.md) | The driver: gates, scheduling, the save/restore loop, status | impure | | [`compute`](compute.md) | Resource requests, native allocation lifecycle, borrowed Dask clients | impure | -| [`execution`](compute.md) | Invocation claims, completion receipts, cleanup confirmation | mixed | | [`sandbox`](sandbox.md) | The exec boundary: policy, backends, attestation, denials | mixed | | [`image` & `container`](container.md) | The container hatch: declaration → image → archive → runtime | pure / impure | | [`crate`](crate.md) | The publication view: the repo as an RO-Crate | pure | diff --git a/docs/api/materialize.md b/docs/api/materialize.md index f010dc9e..39450bcb 100644 --- a/docs/api/materialize.md +++ b/docs/api/materialize.md @@ -18,7 +18,7 @@ driver's stderr, independently of success or failure, leaving stdout for the rep | `check(root, targets, *, refresh)` | The same classification without executing, committing, or fetching. Exempt from the dirty refusal. | | `status(root)` | The report: every output's state and provenance commit, plus the mode/image/sandbox header facts. | | `MaterializeReport` / `StatusReport` | The JSON surfaces; `ok` and `up_to_date` first. | -| `cluster_for_run(cluster_id)` | Borrow the cluster; expose submission, completion, and positive cleanup confirmation. | +| `cluster_for_run(cluster_id)` | Borrow the cluster; the submit/completed scheduler seam (`submit`, `completed`). | | `run_record(...)` / `datalad_run_subject(...)` | The commit message `datalad rerun` replays, and the one spelling of its subject line — shared with the foreign-write comparator, because two strings here would drift. | | `_engine_requirement()` | How a record pins its engine: by version for a release, by source commit (hatch-vcs) for a dev build. | @@ -44,11 +44,9 @@ driver's stderr, independently of success or failure, leaving stdout for the rep worse than either answer. The populated input-hash memo travels with each task; independent worker processes do not rehash shared inputs. Unreadable inputs still fail only the tasks that need them. -6. **Save on `ok`, restore reported failures** — on interruption or a driver - error, first revoke and drain the invocation. Restore submitted, unconsumed - outputs only when the scheduler seam reports positive cleanup confirmation. - Otherwise retain them: an exception is not evidence that a writer stopped. - Separate invocations must still not write the same project concurrently. +6. **Save on `ok`, restore reported failures** — unreported outputs are retained + after interruption because their tasks may still be writing. Allocation + management does not provide concurrent-writer or cancellation guarantees. ## What must stay true @@ -76,6 +74,6 @@ driver's stderr, independently of success or failure, leaving stdout for the rep ## Tests `tests/test_materialize.py` — real repositories, real recipes, a real -`LocalCluster` for scheduling and lifecycle checks, real `datalad rerun` for +`LocalCluster` through the seam exactly once, real `datalad rerun` for the record's whole claim. `cluster_for_run` is the one monkeypatch point for allocation-free tests. diff --git a/docs/api/sandbox.md b/docs/api/sandbox.md index 7a706373..afd87882 100644 --- a/docs/api/sandbox.md +++ b/docs/api/sandbox.md @@ -7,7 +7,7 @@ one, runs it, and reports what was actually enforced. `run.py` (the `lc run` engine) and the worker are the two consumers. Source: `src/lightcone/engine/sandbox/` — `model.py`, `policy.py`, -`boundary.py`, `processes.py`, `landlock.py`, `seatbelt.py`, `oci.py`, `denial.py` — +`boundary.py`, `landlock.py`, `seatbelt.py`, `oci.py`, `denial.py` — plus `lightcone/_sandbox_exec.py`, the Landlock shim. ## Key symbols @@ -26,27 +26,6 @@ An optional output receiver gets stdout/stderr byte chunks. Capturing output nev decodes or normalizes stdout; only the retained stderr tail is decoded for denial classification. Without a receiver, stdout remains inherited. -`processes.Command` starts a small supervisor outside the sandbox. The supervisor -owns the wrapped command's process group and watches a control pipe: worker death -closes the pipe and triggers cleanup even when the worker cannot run `finally`. -Timeouts include command startup. Timeouts and cancellation use the same TERM/KILL -cleanup, with one 15-second deadline shared by process and container operations. -After the leader exits, short-lived helpers get up to one second to exit naturally. -A command that still leaves background processes is failed after they are stopped. -The unreaped leader pins the process-group ID until cleanup completes. - -The OCI backend adds `--cidfile` to its command. Its native client runs in the -supervisor's private directory, while the payload's `--workdir` remains the project. -The boundary passes the OCI runtime explicitly; containing a prefix alone does -not imply container lifecycle behavior. Cleanup inspects the immutable container -ID, stops or kills a running container, verifies it stopped, then attempts removal. -An absent or unverifiable identity during interruption is uncertain, not success. -Only confirmed cleanup allows an ordinary result or `ExecutionCancelled`; -unconfirmed cleanup raises `ExecutionUncertain` and retains the temporary home. -Both lifecycle exceptions are defined in `sandbox.model`, independently of Dask. -Recipes must not detach into other process sessions. A hard-killed supervisor -cannot guarantee cleanup of containers managed by an external runtime. - ## What must stay true - **`wrap` stays pure** — no temp files, no FDs, no global state diff --git a/docs/api/worker.md b/docs/api/worker.md index 8397e01d..3d6e255d 100644 --- a/docs/api/worker.md +++ b/docs/api/worker.md @@ -24,21 +24,17 @@ retain direct terminal output. | Symbol | Role | |---|---| -| `materialize(root, task, context, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns ordinary failures as `TaskResult`; propagates execution safety exceptions. | -| `execute(root, task, input_versions, context)` | Run a recipe unconditionally with cancellation checks, then record its payload and manifest. | -| `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, reason, and diagnostic notes. `.usable` is what dependents check. | +| `materialize(task, versions, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns a `TaskResult`, always. | +| `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, and the attestation. `.usable` is what dependents check. | | `main(argv)` | The rerun entry point: guards, converges the project environment from the commit's own lock, resolves its own HEAD and runtime, executes. | | `lc_version()` | The engine version every manifest records. | ## What must stay true -- **Ordinary recipe failures are results.** Independent tasks continue so - the driver can report all their failures. `ExecutionCancelled` and - `ExecutionUncertain` instead propagate and abort the invocation. They must - not enter the ordinary failed-output restore path: cleanup first establishes - that writers have stopped, and uncertainty retains partial outputs. -- **Task completion includes subprocess teardown.** The boundary owns process - and container cleanup and reports uncertain teardown as an exception. +- **The worker never raises** — enforced at the unit boundary, so the + contract holds for failure modes nobody enumerated. Raising would + make Dask abort every task in flight; reporting all independent + failures in one run is most of what owning the loop buys. - **`data_version` is computed here, before anything is staged** — the dependent's argument *is* this return value, so the digest must exist while the files are still unannexed. Deriving it from diff --git a/docs/architecture.md b/docs/architecture.md index 79fb1b7e..c6ca928a 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -44,7 +44,7 @@ lc materialize "$CLUSTER" │ fetch: git annex get (declared inputs not in this clone) │ converge: uv.lock ⇄ .venv (and the image, containerized) ├─► workers: reset output file → sandbox → recipe → hash → manifest - │ (ordinary failures return failed/blocked) + │ (never raise; return ok/current/behind/failed/blocked) └─ driver: consume results in one thread ok → dataset.save (commit + run record) failed → dataset.restore (tree as clean as it started) @@ -60,11 +60,10 @@ The division of labor is strict and load-bearing: - **Dask owns the ordering.** Every task is submitted with its upstream futures as arguments; there is no ready-set loop or hand-rolled topological sort on the execution path. -- **Ordinary failures are task results.** A recipe failure, a gate failure, - or an unreadable manifest returns a state so independent tasks can continue. - Cancellation and uncertain execution propagate instead, aborting the - invocation. Partial outputs are restored only after writers are confirmed - stopped; uncertainty retains those files for inspection. +- **The worker never raises.** A recipe failure, a gate failure, an + unreadable manifest — all come back as a state, so one failure + doesn't abort every task in flight, and a run reports *all* its + independent failures. - **Values are resolved once and handed down.** HEAD, the container runtime, and the foreign-write facts are read by the driver and passed to workers as values — a worker that asked git itself could @@ -167,12 +166,13 @@ material, not a registry. `compute.connect(CLUSTER_ID)` borrows a standard Dask client and closes only that client on exit. Both execution commands require a cluster ID. The materialization -scheduler keeps its `submit`/`completed` seam; -ordinary Dask scheduling places the tasks. Driver preparation and task runtime -checks remain in their existing owners. `engine.execution` holds invocation claims -and receipts in the scheduler, revokes admission on cancellation, and waits for -command cleanup before allowing output restoration. No execution command -implicitly allocates compute, and there is no separate execution service. +scheduler keeps its `submit`/`completed` seam. Driver preparation and existing +task runtime/sandbox checks remain unchanged. Tasks use ordinary Dask scheduling; +there is no separate worker-selection or preflight layer, or site-marker guard. No execution command implicitly allocates compute. +Before a recipe or probe runs, `engine.execution` claims its task once in the +existing scheduler. Repeated claims and missing invocation state refuse execution, +so Dask cannot silently replay side effects after a worker or result is lost. +This adds no heartbeat or execution service and does not guarantee cancellation. See [compute internals](api/compute.md) and [deployment limits](user/cluster.md). ## The publication view diff --git a/docs/cli/materialize.md b/docs/cli/materialize.md index 1dab5729..e75a72b5 100644 --- a/docs/cli/materialize.md +++ b/docs/cli/materialize.md @@ -48,12 +48,14 @@ never touched, under any flag. ## The run's contract - **Starts clean.** A dirty tree is a refusal. A recipe that returns a - failure has its partial work restored. On interruption or a driver failure, - lc cancels outstanding tasks and restores their uncommitted outputs only after - cleanup is positively confirmed. Completed commits remain. If cleanup is - uncertain, outputs stay in place: stop the allocation by its full ID and verify - its commands and containers have stopped before repairing results. See - [execution limits](../user/cluster.md#execution-requirements-and-limits). + failure has its partial work restored. After a cluster interruption, + unreported outputs are retained because tasks may still be running. The + same holds when a commit fails while other recipes are still running: the + error says so. + Stop the allocation with `lc compute down` and its full ID (a name can + already belong to a newer allocation), and confirm its recipes have + stopped before cleaning results. Local containers may need separate + termination through their runtime; see [execution limits](../user/cluster.md#execution-requirements-and-limits). - **Fetches what it needs.** Declared inputs whose annexed content is not in this clone are fetched before anything hashes. - **Commits as it goes.** Each output lands in its own commit, written diff --git a/docs/cli/run.md b/docs/cli/run.md index b6baa96f..1e21c53e 100644 --- a/docs/cli/run.md +++ b/docs/cli/run.md @@ -40,11 +40,12 @@ that variable to an existing cluster. Set command-specific values inside the command, for example `lc run "$CLUSTER" -- env NAME=value python script.py`. Containerized commands use the image's environment and the sandbox overlays. -Interrupting the CLI requests cancellation and waits for command cleanup; the -cluster remains available. If cleanup cannot be confirmed, the error says so. -Stop the allocation using `lc compute down` with its full ID (names can be reused) -and verify its commands and containers have stopped before repairing outputs. -See [execution limits](../user/cluster.md#execution-requirements-and-limits). +Interrupting the CLI detaches its client; the remote command may still be running. +Stop the allocation with `lc compute down` and its full ID (a name can already +belong to a newer allocation) before working with files the interrupted command +could still be writing. Confirm that the command has +stopped; local containers may require separate termination through their runtime +(see [execution limits](../user/cluster.md#execution-requirements-and-limits)). ## What it does diff --git a/docs/user/cluster.md b/docs/user/cluster.md index 15d1b671..886253bd 100644 --- a/docs/user/cluster.md +++ b/docs/user/cluster.md @@ -44,10 +44,9 @@ Local resources are cooperative limits, not an exclusive CPU/RAM reservation. An allocation owns a detached process session and standard `LocalCluster`: one worker process with `task_slots_per_node` threads, and a scheduler that listens on `127.0.0.1` over TLS. Its own logs are discarded; a startup failure is kept -and shown as the reason by `lc compute status`. At its time limit, or on `down`, -the allocation stops its process session with SIGTERM, then SIGKILL if needed. -Shutdown normally allows three seconds; active command supervisors get up to -sixteen seconds to stop their commands and containers before escalation. +and shown as the reason by `lc compute status`. At its time limit the whole +process session is killed with SIGKILL, so a recipe still running stops mid-write. +`down` sends SIGTERM, waits three seconds, then sends SIGKILL. Private process locators are checked against the native boot UUID, UID, process session, and exact command containing the allocation's random token before attachment or termination. Hostname changes and clock adjustments do not change @@ -333,24 +332,19 @@ directories and credential files still reject symlinks, retain ownership and ancestor-permission checks, and require modes `0700` and `0600`, respectively. The catalog's location is independent of the private connection files. -Ctrl-C revokes the invocation and waits for its commands to stop. Materialize -keeps completed commits and restores uncommitted outputs only after cleanup is -confirmed. If cleanup cannot be confirmed, partial outputs stay in place and the -error asks you to stop the allocation and verify its commands and containers -before retrying. The allocation remains available after ordinary cancellation. - -The existing Dask scheduler holds invocation claims and completion receipts. -After a worker disappears, a replacement task cannot rerun a recipe whose result -is uncertain. A completed task returns its original receipt. Loss of the client -or expiry of its 15-second heartbeat lease revokes further work; brief RPC -failures are retried within the last confirmed lease. A small command supervisor -stops the command if its worker dies. There is no automatic recovery or replay after an -uncertain execution, and no additional server or checkout state directory. - -Use one execution invocation per project at a time: there is no checkout lock -across invocations. Recipes must finish all their work before returning and must -not detach daemon processes into new sessions. Cleanup covers each command's -process group and its OCI container, whose immutable runtime ID is checked. -If a supervisor itself is forcibly killed, a container managed by an external -runtime can survive; native allocation termination alone cannot prove it stopped. -Inspect and stop such containers before repairing outputs. +Lightcone prevents Dask from automatically rerunning a recipe or probe within the +same invocation: each task must claim permission once in the existing scheduler +before it runs. If a worker or its result is lost, an attempted replay fails +instead. Missing scheduler state also refuses execution. There is no automatic +recovery; inspect the allocation and outputs before starting another invocation. + +Use one execution invocation per project at a time. Concurrent writers, +comprehensive cancellation, and recovery after client/worker loss are not +guaranteed. A lost client does not prove its subprocesses stopped. +Unreported partial outputs are retained after interruption rather than restored +while a task may still write them. End the allocation and establish that work has +stopped before inspecting or repairing that project's outputs. +For local containerized execution, `down` and walltime expiry stop the managed +process group but do not guarantee termination of containers managed by an +external runtime. A Podman container that ignores SIGTERM can survive. Inspect +and stop such containers through the container runtime before cleaning results. diff --git a/src/lightcone/cli/commands.py b/src/lightcone/cli/commands.py index 8d04041c..e9346c34 100644 --- a/src/lightcone/cli/commands.py +++ b/src/lightcone/cli/commands.py @@ -179,7 +179,6 @@ def run(cluster_id: str, command: tuple[str, ...]) -> None: remains available after the command finishes. """ from lightcone.engine import run as engine_run - from lightcone.engine.execution import ExecutionInterrupted from lightcone.engine.project import current_project _require_cluster_id(cluster_id) @@ -189,13 +188,8 @@ def run(cluster_id: str, command: tuple[str, ...]) -> None: raise click.UsageError("A command is required; no interactive shell is opened.") try: outcome = engine_run.probe(current_project(), command, cluster_id=cluster_id) - except KeyboardInterrupt as exc: - if isinstance(exc, ExecutionInterrupted): - click.echo( - "Interrupted; the command has stopped. The cluster remains available.", err=True, - ) - else: - _interrupted(cluster_id, "the remote command may still be running") + except KeyboardInterrupt: + _interrupted(cluster_id, "the remote command may still be running") raise if outcome.notes: click.echo("\n".join(["", *outcome.notes]), err=True) @@ -383,19 +377,11 @@ def materialize( else: try: report = engine.materialize(root, targets, cluster_id=cluster_id, refresh=refresh) - except KeyboardInterrupt as exc: - from lightcone.engine.execution import ExecutionInterrupted - - if isinstance(exc, ExecutionInterrupted): - click.echo( - "Interrupted; recipes have stopped and uncommitted outputs were restored. " - "The cluster remains available.", err=True, - ) - else: - _interrupted( - cluster_id, "remote recipes may still be writing results", - " and confirm they have stopped before cleaning results/", - ) + except KeyboardInterrupt: + _interrupted( + cluster_id, "remote recipes may still be writing results", + " and confirm they have stopped before cleaning results/", + ) raise if as_json: diff --git a/src/lightcone/engine/compute/__init__.py b/src/lightcone/engine/compute/__init__.py index d1635726..0f3c40c6 100644 --- a/src/lightcone/engine/compute/__init__.py +++ b/src/lightcone/engine/compute/__init__.py @@ -39,6 +39,14 @@ def _slurm(connection: Connection) -> Provider: # The lifecycle seam is intentionally small: execution never dispatches on a provider. PROVIDERS: dict[str, ProviderFactory] = {"local": _local, "slurm": _slurm} +#: What a driver leaving early must say: closing a client cannot prove that a +#: remote subprocess has stopped. +UNSTOPPED = ( + "lc did not stop the allocation; tasks that did not report may still be " + "running, and any files they wrote remain" +) + + def validate_id(value: str) -> None: """Reject invalid execution targets before preparing a project.""" if value.startswith("clu_"): diff --git a/src/lightcone/engine/compute/local.py b/src/lightcone/engine/compute/local.py index 0f84e97a..0474d146 100644 --- a/src/lightcone/engine/compute/local.py +++ b/src/lightcone/engine/compute/local.py @@ -382,33 +382,43 @@ def connect(self, identity: Identity, *, timeout: float = 10) -> Iterator[Any]: client.close(timeout=min(timeout, 5)) def terminate(self, identity: Identity) -> None: - """Terminate the validated allocation session even if Dask is wedged.""" + """Terminate the validated allocation process group even if Dask is wedged.""" directory, record = self._record(identity) self._stop(identity, directory, record) self._retire(directory) def _stop(self, identity: Identity, directory: Path, record: dict[str, Any]) -> None: - from lightcone.engine.sandbox.processes import ( - _CLEANUP_TIMEOUT, - has_custodian, - ) - from lightcone.engine.sandbox.processes import members as session_members - process = self._process(identity, directory, record) if process is None: return - # Capture birth identities while the owner establishes session custody. - members = session_members(session=process.pid) - for member in members: + members = [] + for member in psutil.process_iter(): try: - member.terminate() - except psutil.NoSuchProcess: + if ( + member.uids().real == os.getuid() + and os.getpgid(member.pid) == process.pid + and os.getsid(member.pid) == process.pid + ): + # Capture each birth identity while the owner still proves + # this session is ours. psutil's signal methods check reuse. + member.create_time() + members.append(member) + except (psutil.NoSuchProcess, psutil.AccessDenied, ProcessLookupError): + continue + process = self._process(identity, directory, record) + if process is not None: + try: + os.killpg(process.pid, signal.SIGTERM) + except ProcessLookupError: pass + else: + for member in members: + try: + member.terminate() + except psutil.NoSuchProcess: + pass for escalation in (False, True): - grace = ( - max(_STOP_GRACE, _CLEANUP_TIMEOUT + 1) if has_custodian(members) else _STOP_GRACE - ) - deadline = time.monotonic() + grace + deadline = time.monotonic() + _STOP_GRACE while members and time.monotonic() < deadline: living = [] for member in members: @@ -429,13 +439,12 @@ def _stop(self, identity: Identity, directory: Path, record: dict[str, Any]) -> ) process = self._process(identity, directory, record) if process is not None: - # Commands have separate groups inside this session. The - # live owner establishes custody of newly created members. - for member in session_members(session=process.pid): - try: - member.kill() - except psutil.NoSuchProcess: - pass + try: + # A live, verified owner also covers children created while + # stopping. Its finalizer provides the same group-wide kill. + os.killpg(process.pid, signal.SIGKILL) + except ProcessLookupError: + pass for member in members: try: # The owner may have exited first. Never signal its old PGID diff --git a/src/lightcone/engine/compute/local_runtime.py b/src/lightcone/engine/compute/local_runtime.py index 515555df..dbb360e6 100644 --- a/src/lightcone/engine/compute/local_runtime.py +++ b/src/lightcone/engine/compute/local_runtime.py @@ -11,8 +11,6 @@ from pathlib import Path from types import FrameType -import psutil - from lightcone.engine.compute.runtime import ( SCHEDULER_CONFIG, create_security, @@ -20,43 +18,6 @@ read_private_json, write_private_json, ) -from lightcone.engine.sandbox.processes import _CLEANUP_TIMEOUT, has_custodian, members - - -def _stop_session() -> None: - """Give command custodians time to drain, then kill the allocation session.""" - owner = os.getpid() - remaining: list[psutil.Process] = [] - started = time.monotonic() - try: - remaining = [member for member in members(session=owner) if member.pid != owner] - for member in remaining: - try: - member.terminate() - except psutil.NoSuchProcess: - pass - while remaining: - grace = _CLEANUP_TIMEOUT if has_custodian(remaining) else 2.5 - if time.monotonic() - started >= grace: - break - time.sleep(0.05) - remaining = [member for member in members(session=owner) if member.pid != owner] - except Exception: - # Enumeration can fail during shutdown. The original group is still - # ours; let its custodians drain their command groups before hard kill. - os.killpg(owner, signal.SIGTERM) - time.sleep(max(0.0, started + _CLEANUP_TIMEOUT - time.monotonic())) - finally: - try: - for member in remaining: - try: - member.kill() - except (psutil.NoSuchProcess, psutil.AccessDenied): - pass - finally: - # This also covers workers the last enumeration could not observe. - # The owner was verified as session/group leader before startup. - os.killpg(owner, signal.SIGKILL) def main() -> None: @@ -73,12 +34,12 @@ def stop(_signum: int, _frame: FrameType | None) -> None: def expire(_signum: int, _frame: FrameType | None) -> None: # This bound does not depend on the scheduler loop or graceful Dask close. - _stop_session() + os.killpg(os.getpgrp(), signal.SIGKILL) # Register before importing Dask so its multiprocessing finalizers run first. # A recipe can ignore SIGTERM and outlive its worker: keep custody of the # session and walltime timer until every member has been sent SIGKILL. - atexit.register(_stop_session) + atexit.register(os.killpg, os.getpgrp(), signal.SIGKILL) signal.signal(signal.SIGTERM, stop) signal.signal(signal.SIGINT, stop) signal.signal(signal.SIGALRM, expire) diff --git a/src/lightcone/engine/compute/output.py b/src/lightcone/engine/compute/output.py index 3fadd122..c92bbcb9 100644 --- a/src/lightcone/engine/compute/output.py +++ b/src/lightcone/engine/compute/output.py @@ -65,9 +65,4 @@ def output(stream: str, data: bytes) -> None: try: return function(*args, output=output) finally: - # Output delivery must not mask an execution-safety exception. A lost - # final marker is reported by the driver's bounded output wait. - try: - worker.log_event(topic, {"done": task}) - except Exception: - pass + worker.log_event(topic, {"done": task}) diff --git a/src/lightcone/engine/execution.py b/src/lightcone/engine/execution.py index 19ad195f..7aa6d450 100644 --- a/src/lightcone/engine/execution.py +++ b/src/lightcone/engine/execution.py @@ -1,291 +1,91 @@ -"""Bound one invocation's ordinary Dask tasks to their command lifetimes. - -The existing scheduler keeps small claims and completion receipts. Tasks never -create missing invocations: losing the scheduler therefore fails closed, rather -than starting a recipe again. This is not a lock on a project checkout. -""" +"""Refuse repeated execution of side-effecting tasks in the existing Dask scheduler.""" from __future__ import annotations -import threading -import time from collections.abc import Callable, Iterator from contextlib import contextmanager -from contextvars import ContextVar -from dataclasses import dataclass, field -from typing import Any, Literal +from dataclasses import dataclass +from typing import Any from uuid import uuid4 -from lightcone.engine.sandbox.model import ExecutionCancelled as ExecutionCancelled -from lightcone.engine.sandbox.model import ExecutionUncertain as ExecutionUncertain - -_HEARTBEAT = 2.0 -_LEASE = 15.0 -_RPC_TIMEOUT = 5.0 -_STOP_TIMEOUT = 40.0 -_CANCELLED: ContextVar[Callable[[], bool]] = ContextVar( - "execution_cancelled", default=lambda: False, -) - - -Operation = Literal[ - "register", "heartbeat", "active", "claim", "finished", "stopped", "uncertain", - "revoke", "pending", "forget", -] - - -class ExecutionInterrupted(KeyboardInterrupt): - """An interrupt whose invocation has positively confirmed command cleanup.""" - - -@dataclass(frozen=True) -class _Claim: - fresh: bool - result: Any - remaining: float - - -@dataclass(frozen=True) -class _Progress: - running: tuple[str, ...] - uncertain: tuple[str, ...] - - -def cancelled() -> bool: - """Check the current task's authorization without blocking its subprocess loop.""" - return _CANCELLED.get()() +from lightcone.engine.project import ProjectError -def check_cancelled() -> None: - """Refuse further task mutations once cancellation has been observed.""" - if cancelled(): - raise ExecutionCancelled("execution was cancelled") - - -def _state( - invocation: str, operation: Operation, task: str = "", value: Any = None, - *, dask_scheduler: Any, -) -> Any: - # Scheduler callbacks run serially on its event loop. Registration is a - # driver-only operation, never retried implicitly by a task or heartbeat. +def _register(invocation: str, owner: str, *, dask_scheduler: Any) -> None: records = dask_scheduler.extensions.setdefault("lightcone-executions", {}) - now = time.monotonic() - if operation == "register": - if invocation in records: - raise ExecutionUncertain("execution is already registered") - records[invocation] = {"client": value, "deadline": now + _LEASE, - "active": True, "tasks": {}} - return None + if invocation in records: + raise ProjectError("execution is already registered") + # Collect abandoned invocations using Dask's own client membership, without + # a background service or a second liveness protocol. + for key, record in list(records.items()): + if record["client"] not in dask_scheduler.clients: + del records[key] + records[invocation] = {"client": owner, "tasks": set()} + + +def _claim(invocation: str, task: str, *, dask_scheduler: Any) -> None: + records = dask_scheduler.extensions.get("lightcone-executions", {}) record = records.get(invocation) - if record is None: - raise ExecutionUncertain("execution is no longer registered; refusing task replay") - if now > record["deadline"] or record["client"] not in dask_scheduler.clients: - record["active"] = False - if operation == "heartbeat": - if record["active"]: - record["deadline"] = now + _LEASE - return record["active"] - if operation == "active": - return max(0.0, record["deadline"] - now) if record["active"] else 0.0 - if operation == "claim": - if not record["active"]: - raise ExecutionCancelled("execution is no longer active") - previous = record["tasks"].get(task) - if previous is not None: - if previous["state"] == "finished": - return _Claim(False, previous["result"], 0.0) - record["active"] = False - raise ExecutionUncertain(f"{task}: a previous attempt has no confirmed result") - record["tasks"][task] = {"state": "running", "attempt": value} - return _Claim(True, None, record["deadline"] - now) - if operation in {"finished", "stopped", "uncertain"}: - attempt, result = value - if record["tasks"].get(task, {}).get("attempt") != attempt: - raise ExecutionUncertain(f"{task}: completion belongs to another attempt") - record["tasks"][task] = {"state": operation, "attempt": attempt, "result": result} - if operation == "uncertain": - record["active"] = False - return None - if operation == "revoke": - record["active"] = False - if operation in {"revoke", "pending"}: - return _Progress( - tuple(name for name, item in record["tasks"].items() if item["state"] == "running"), - tuple(name for name, item in record["tasks"].items() if item["state"] == "uncertain"), - ) - if operation == "forget": - del records[invocation] - return None - raise ValueError(f"unknown execution operation: {operation}") - + if record is None or record["client"] not in dask_scheduler.clients: + raise ProjectError("execution is no longer registered or its client disconnected") + # This synchronous callback runs atomically on the scheduler's event loop. + # Claims survive task failure, worker loss, and Dask forgetting task results. + if task in record["tasks"]: + raise ProjectError(f"{task}: already claimed; refusing duplicate execution") + record["tasks"].add(task) -def _rpc( - client: Any, invocation: str, operation: Operation, task: str = "", value: Any = None, -) -> Any: - return client.sync( - client.run_on_scheduler, _state, invocation, operation, task, value, - callback_timeout=_RPC_TIMEOUT, - ) +def _forget(invocation: str, *, dask_scheduler: Any) -> None: + dask_scheduler.extensions.get("lightcone-executions", {}).pop(invocation, None) -def _call(invocation: str, task: str, function: Callable[..., Any], *args: Any) -> Any: - from distributed import get_client - client = get_client() - # A lost claim reply is ambiguous. Do not execute unless it was received. - attempt = uuid4().hex - requested_at = time.monotonic() - claim: _Claim = _rpc(client, invocation, "claim", task, attempt) - if not claim.fresh: - return claim.result - deadline = requested_at + claim.remaining - stopped = threading.Event() - revoked = threading.Event() +def _rpc(client: Any, function: Callable[..., None], *args: Any) -> None: + try: + client.sync(client.run_on_scheduler, function, *args, callback_timeout=5) + except ProjectError: + raise + except Exception as exc: + raise ProjectError(f"cannot contact the Dask execution guard: {exc}") from exc - def monitor() -> None: - nonlocal deadline - while not stopped.wait(_HEARTBEAT): - requested_at = time.monotonic() - if requested_at >= deadline: - revoked.set() - return - try: - remaining = _rpc(client, invocation, "active") - except ExecutionUncertain: - revoked.set() - return - except Exception: - # An unavailable RPC is not a revocation. Keep the last grant, - # without extending it, while retrying within its deadline. - continue - if not remaining or time.monotonic() >= deadline: - revoked.set() - return - deadline = requested_at + remaining - def is_cancelled() -> bool: - if time.monotonic() >= deadline: - revoked.set() - return revoked.is_set() +def _call(invocation: str, task: str, function: Callable[..., Any], *args: Any) -> Any: + from distributed import get_client - watcher = threading.Thread(target=monitor, daemon=True) - watcher.start() - token = _CANCELLED.set(is_cancelled) - try: - check_cancelled() - result = function(*args) - check_cancelled() - except BaseException as exc: - state: Operation = "uncertain" if isinstance(exc, ExecutionUncertain) else "stopped" - try: - _rpc(client, invocation, state, task, (attempt, None)) - except Exception: - pass - raise - else: - try: - _rpc(client, invocation, "finished", task, (attempt, result)) - except Exception as exc: - raise ExecutionUncertain( - f"{task}: could not record completion; refusing replay" - ) from exc - return result - finally: - _CANCELLED.reset(token) - stopped.set() + # A lost claim reply is ambiguous. Execute only after acknowledgment, and + # never recreate missing state or retry the claim on a worker's behalf. + _rpc(get_client(), _claim, invocation, task) + return function(*args) -@dataclass +@dataclass(frozen=True) class Invocation: - """Submit tasks whose side effects must not be replayed automatically.""" + """Submit tasks that may begin at most once within this invocation.""" client: Any - id: str = field(default_factory=lambda: uuid4().hex) - futures: list[Any] = field(default_factory=list) - stopped: bool = False - _submissions: int = 0 + id: str - def submit( - self, function: Callable[..., Any], *args: Any, key: str, - ) -> Any: - """Claim each logical task inside its worker before it can mutate files.""" - # A submit failure may follow native acceptance. Count it before the - # call so even a caught exception cannot manufacture full completion. - self._submissions += 1 - future = self.client.submit( + def submit(self, function: Callable[..., Any], *args: Any, key: str) -> Any: + """Guard each task before its first side effect, including Dask recomputation.""" + return self.client.submit( _call, self.id, key, function, *args, key=f"lc-{self.id}-{key}", pure=False, retries=0, ) - self.futures.append(future) - return future @contextmanager def invocation(client: Any) -> Iterator[Invocation]: - """Own authorization and wait for running commands to stop before detaching.""" - run = Invocation(client) - registered_at = time.monotonic() - _rpc(client, run.id, "register", value=client.id) - stopped = threading.Event() - - def heartbeat() -> None: - deadline = registered_at + _LEASE - while not stopped.wait(_HEARTBEAT): - requested_at = time.monotonic() - if requested_at >= deadline: - return - try: - if not _rpc(client, run.id, "heartbeat"): - return - except ExecutionUncertain: - return - except Exception: - continue - deadline = requested_at + _LEASE - - threading.Thread(target=heartbeat, daemon=True).start() - failure: BaseException | None = None + """Register claims for one borrowed client; never stop its running commands. + + Missing records refuse admission, so forgetting an invocation also prevents + late tasks from starting. Cleanup failures cannot discard received results; + abandoned records are collected when another invocation registers. + """ + run = Invocation(client, uuid4().hex) + _rpc(client, _register, run.id, client.id) try: yield run - except BaseException as exc: - failure = exc - raise finally: - stopped.set() - # A finished wrapper has already acknowledged command cleanup and - # published its result. Losing metadata-cleanup RPCs cannot undo that. - run.stopped = ( - run._submissions == len(run.futures) - and all(future.status == "finished" for future in run.futures) - ) try: - pending = _rpc(client, run.id, "revoke") - unfinished = [future for future in run.futures if not future.done()] - if unfinished: - client.sync(client.cancel, unfinished, callback_timeout=_RPC_TIMEOUT) - deadline = time.monotonic() + _STOP_TIMEOUT - while pending.running and not pending.uncertain and time.monotonic() < deadline: - time.sleep(0.1) - pending = _rpc(client, run.id, "pending") - if pending.running or pending.uncertain: - run.stopped = False - raise ExecutionUncertain( - "unconfirmed tasks: " + ", ".join((*pending.running, *pending.uncertain)) - ) - run.stopped = True - except Exception as exc: - if not run.stopped: - raise ExecutionUncertain( - f"could not confirm execution stopped: {exc}; partial outputs were retained. " - "Stop the allocation and verify its commands/containers have ended " - "before retrying" - ) from exc - else: - # Revoked admission plus no unfinished claims already proves stop. - # Forgetting receipts is metadata cleanup, not another safety gate. - try: - _rpc(client, run.id, "forget") - except Exception: - pass - if isinstance(failure, KeyboardInterrupt) and run.stopped: - raise ExecutionInterrupted() from failure + _rpc(client, _forget, run.id) + except ProjectError: + pass diff --git a/src/lightcone/engine/materialize.py b/src/lightcone/engine/materialize.py index 3a6f4eeb..24c67631 100644 --- a/src/lightcone/engine/materialize.py +++ b/src/lightcone/engine/materialize.py @@ -14,9 +14,8 @@ **It owns git, alone.** Workers execute and return; the driver commits, in one thread, as results arrive. That is not a preference: concurrent git operations on one repository race on the index lock. The same loop -restores what a completed failed task left behind. On interruption it restores -unreported outputs only after confirming their writers stopped; otherwise it -retains those partial files. +restores what a completed failed task left behind. Unreported tasks may +still be writing, so interruptions retain their partial files. One consequence, checked rather than assumed: a dependent starts as soon as its upstream's *worker* returns, which is milliseconds before the @@ -515,97 +514,94 @@ def materialize( # maintainer. Nothing is submitted, so no allocation is needed. _converge_crate(root, report, full, dsid) return report - unreported: set[Key] = set() - scheduler: Scheduler | None = None - try: - with cluster_for_run(cluster_id) as scheduler: - _fetch_inputs(root, graph, report) - # Materialize is one of the two verbs allowed to build the image (the - # other is `lc build`); the probe and the rerun entry point only find - # one. Resolved once, then handed to every task — the HEAD discipline. - runtime = container.runtime_for_run(root, build=True) - # Converge the environment: workers pass `--no-sync`, so this is the - # only place on a run's path where it is made to match the lock. (A - # rerun does not come through here; its entry point converges too.) - report.warnings.extend(f"uv: {w}" for w in container.converge(runtime)) - # The run's driver-resolved facts, each read once: HEAD because the - # driver commits as outputs land and a per-task read would stamp - # later manifests with a commit this run created; the uv probe - # because attestation is a fact about the run (and empty is an - # answer, not a failure); one content-hash memo because a declared - # input shared by several outputs is the same bytes every time. - versions = assets.Versions() - for path in { - path - for task in graph.tasks.values() - for name, path in task.inputs.items() - if name not in task.produced_by - }: - try: - versions.of(path) - except Exception: - # Keep unreadable-input failures inside the tasks that need them. - pass - context = worker.RunContext( - env_version=env_version, - head=dataset.head(root), - versions=versions, - runtime=runtime, - uv_version=project.uv_version(root), + with cluster_for_run(cluster_id) as scheduler: + _fetch_inputs(root, graph, report) + # Materialize is one of the two verbs allowed to build the image (the + # other is `lc build`); the probe and the rerun entry point only find + # one. Resolved once, then handed to every task — the HEAD discipline. + runtime = container.runtime_for_run(root, build=True) + # Converge the environment: workers pass `--no-sync`, so this is the + # only place on a run's path where it is made to match the lock. (A + # rerun does not come through here; its entry point converges too.) + report.warnings.extend(f"uv: {w}" for w in container.converge(runtime)) + # The run's driver-resolved facts, each read once: HEAD because the + # driver commits as outputs land and a per-task read would stamp + # later manifests with a commit this run created; the uv probe + # because attestation is a fact about the run (and empty is an + # answer, not a failure); one content-hash memo because a declared + # input shared by several outputs is the same bytes every time. + versions = assets.Versions() + for path in { + path + for task in graph.tasks.values() + for name, path in task.inputs.items() + if name not in task.produced_by + }: + try: + versions.of(path) + except Exception: + # Keep unreadable-input failures inside the tasks that need them. + pass + context = worker.RunContext( + env_version=env_version, + head=dataset.head(root), + versions=versions, + runtime=runtime, + uv_version=project.uv_version(root), + ) + # The history question is the driver's to answer — workers have no + # git, by design — so each task is told up front whether its + # directory was last written by something other than its own run + # record. A foreign write contradicts the manifest, and a worker that + # trusted the recorded digest would skip the output forever. Guarded + # on the manifest's presence, as `_classified` is: without one the + # answer is dead — the output is remade regardless — and each ask is + # a git process. + foreign = { + key: _foreign_write(root, task) if task.manifest_path.is_file() else None + for key, task in graph.tasks.items() + } + pending: dict[Key, Any] = {} + # Futures retain dependency ordering; task placement belongs to the + # selected cluster, while commits stay in this one driver thread. + for key in graph.order(): + task = graph.tasks[key] + pending[key] = scheduler.submit( + worker.materialize, + root, + task, + context, + refresh, + foreign[key], + *[pending[dep] for dep in task.depends_on], + key=_name(key), ) - # The history question is the driver's to answer — workers have no - # git, by design — so each task is told up front whether its - # directory was last written by something other than its own run - # record. A foreign write contradicts the manifest, and a worker that - # trusted the recorded digest would skip the output forever. Guarded - # on the manifest's presence, as `_classified` is: without one the - # answer is dead — the output is remade regardless — and each ask is - # a git process. - foreign = { - key: _foreign_write(root, task) if task.manifest_path.is_file() else None - for key, task in graph.tasks.items() - } - pending: dict[Key, Any] = {} - # Futures retain dependency ordering; task placement belongs to the - # selected cluster, while commits stay in this one driver thread. - for key in graph.order(): - task = graph.tasks[key] - unreported.add(key) - pending[key] = scheduler.submit( - worker.materialize, - root, - task, - context, - refresh, - foreign[key], - *[pending[dep] for dep in task.depends_on], - key=_name(key), - ) - - for result in scheduler.completed(list(pending.values())): + from lightcone.engine.compute import UNSTOPPED + + # An unreported task can still have a running subprocess. Leave its + # partial files in place on interruption rather than restoring over it. + outstanding = len(pending) + for result in scheduler.completed(list(pending.values())): + outstanding -= 1 + try: _consume(root, graph.tasks[result.key], result, dsid, runtime, report) - unreported.discard(result.key) - # The tree was clean at the start-of-run refusal and save/restore - # keeps `results/` clean, so anything dirty *now* was edited while - # the graph ran — and every manifest records the starting commit, - # which no longer describes that code. A warning, never a manifest - # field: the driver does not rewrite files the worker owns. - if edited := dataset.status(root): - names = ", ".join(sorted(path for _, path in edited)) - report.warnings.append( - f"edited while the run was in flight: {names} — the manifests " - "record the starting commit, which no longer describes this code" - ) - _converge_crate(root, report, full, dsid) - return report - - except BaseException: - # An outer context's error can mask an uncertain cleanup. Require a - # positive acknowledgment, independent of which exception reached us. - if scheduler is not None and scheduler.stopped: - for key in unreported: - dataset.restore(root, _owned(root, graph.tasks[key])) - raise + except Exception as exc: + if not outstanding: + raise + raise ProjectError(f"{exc}. {UNSTOPPED}") from exc + # The tree was clean at the start-of-run refusal and save/restore + # keeps `results/` clean, so anything dirty *now* was edited while + # the graph ran — and every manifest records the starting commit, + # which no longer describes that code. A warning, never a manifest + # field: the driver does not rewrite files the worker owns. + if edited := dataset.status(root): + names = ", ".join(sorted(path for _, path in edited)) + report.warnings.append( + f"edited while the run was in flight: {names} — the manifests " + "record the starting commit, which no longer describes this code" + ) + _converge_crate(root, report, full, dsid) + return report def _consume( @@ -649,11 +645,6 @@ class Scheduler(Protocol): they land, keeping the commit logic independent of the provider. """ - @property - def stopped(self) -> bool: - """Whether this invocation positively confirmed all claimed tasks stopped.""" - ... - def submit(self, fn: Any, *args: Any, key: str) -> Any: """Schedule a call. @@ -687,11 +678,6 @@ class _Dask: invocation: execution.Invocation output: Forwarder - @property - def stopped(self) -> bool: - """Expose positive cleanup confirmation after the connection context exits.""" - return self.invocation.stopped - def submit(self, fn: Any, *args: Any, key: str) -> Any: """Submit an ordinary Dask task with a unique key and forwarded output.""" from lightcone.engine.compute.output import call @@ -719,7 +705,9 @@ def completed(self, handles: list[Any]) -> Iterator[worker.TaskResult]: except ProjectError: raise except Exception as exc: - raise ProjectError(f"cluster execution failed: {exc}") from exc + from lightcone.engine.compute import UNSTOPPED + + raise ProjectError(f"cluster execution failed: {exc}. {UNSTOPPED}") from exc @contextmanager diff --git a/src/lightcone/engine/run.py b/src/lightcone/engine/run.py index 052d7649..be44b7d1 100644 --- a/src/lightcone/engine/run.py +++ b/src/lightcone/engine/run.py @@ -62,7 +62,9 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. except ProjectError: raise except Exception as exc: - raise ProjectError(f"cluster execution failed: {exc}") from exc + raise ProjectError( + f"cluster execution failed: {exc}. {compute.UNSTOPPED}" + ) from exc if not output.wait("probe"): notes.append("remote output forwarding did not finish before its deadline") if warning := uv_scrub_warning(): @@ -80,7 +82,6 @@ def _probe( outcome = sandbox.run( container.backend(runtime), policy, command, cwd=runtime.root, prefix=uv_prefix(runtime.root), env=child_env(), output=output, - cancelled=execution.cancelled, ) return outcome diff --git a/src/lightcone/engine/sandbox/boundary.py b/src/lightcone/engine/sandbox/boundary.py index 4a66c7d3..7b87ea4f 100644 --- a/src/lightcone/engine/sandbox/boundary.py +++ b/src/lightcone/engine/sandbox/boundary.py @@ -10,6 +10,7 @@ from __future__ import annotations import shutil +import subprocess import sys import threading from collections import deque @@ -20,14 +21,7 @@ from typing import IO from lightcone.engine.sandbox import policy as policy_module -from lightcone.engine.sandbox.model import ( - Attestation, - Backend, - Capability, - ExecutionUncertain, - Policy, -) -from lightcone.engine.sandbox.processes import Command +from lightcone.engine.sandbox.model import Attestation, Backend, Capability, Policy #: How much of the child's stderr to keep for the denial classifier. The #: denial is in the last few lines of a traceback, and a recipe that @@ -123,15 +117,10 @@ def scope(policy: Policy) -> Iterator[Policy]: Yields: The same policy, with its ``tmp_home`` removed on exit. """ - cleanup = True try: yield policy - except ExecutionUncertain: - cleanup = False - raise finally: - if cleanup: - shutil.rmtree(policy.tmp_home, ignore_errors=True) + shutil.rmtree(policy.tmp_home, ignore_errors=True) def run( @@ -143,8 +132,6 @@ def run( env: dict[str, str], prefix: Sequence[str] = (), output: Callable[[str, bytes], None] | None = None, - cancelled: Callable[[], bool] | None = None, - timeout: float | None = None, ) -> Outcome: """Run a command through a backend, and explain it if it fails. @@ -167,8 +154,6 @@ def run( host-resolved ``env``. output: Optional receiver for unchanged stdout/stderr bytes, used when the caller forwards a remote command's output to its own terminal. - cancelled: A stop predicate polled while the command runs. - timeout: Maximum command runtime in seconds, or no bound. Returns: The exit code, what was actually enforced, and any lines the @@ -179,11 +164,6 @@ def run( else: wrapped = [*prefix, *backend.wrap(policy, [*env_argv(policy), *argv])] attestation = backend.attest(policy) - oci_runtime = ( - attestation.mechanism - if attestation.mechanism in {"podman", "docker", "podman-hpc"} - else None - ) # `policy.env` is deliberately **not** merged here: it went inside # the wrap, above, via :func:`env_argv`. Everything *outside* the # rewrite has to keep the real environment — `uv` resolves its cache @@ -196,33 +176,33 @@ def run( if backend.capability.kind == "none": notes.append(_downgrade_note(backend.capability)) - with Command( - wrapped, cwd=cwd, env=child_env, capture=output is not None, timeout=timeout, - oci_runtime=oci_runtime, - ) as command: - proc = command.process - assert proc.stderr is not None # Popen was given PIPE - tail = _Tail(proc.stderr, output) - tail.start() - stdout: threading.Thread | None = None - if output is not None: - assert proc.stdout is not None - - def forward() -> None: - assert proc.stdout is not None and output is not None - while chunk := proc.stdout.read(64 * 1024): - output("stdout", chunk) - - stdout = threading.Thread(target=forward, daemon=True) - stdout.start() - try: - returncode, lifecycle_note = command.wait(cancelled) - finally: - tail.join(timeout=5) - if stdout is not None: - stdout.join(timeout=5) - if lifecycle_note: - notes.append(lifecycle_note) + proc = subprocess.Popen( + wrapped, + cwd=cwd, + env=child_env, + stdin=subprocess.DEVNULL if output is not None else None, + stdout=subprocess.PIPE if output is not None else None, + stderr=subprocess.PIPE, + bufsize=0, + ) + assert proc.stderr is not None # Popen was given PIPE + tail = _Tail(proc.stderr, output) + tail.start() + stdout: threading.Thread | None = None + if output is not None: + assert proc.stdout is not None + + def forward() -> None: + assert proc.stdout is not None and output is not None + while chunk := proc.stdout.read(64 * 1024): + output("stdout", chunk) + + stdout = threading.Thread(target=forward, daemon=True) + stdout.start() + returncode = proc.wait() + tail.join(timeout=5) + if stdout is not None: + stdout.join(timeout=5) # Imported here, not at module scope: `sandbox/__init__` loads this # module eagerly, and the shim drags ctypes in for one integer. @@ -239,7 +219,7 @@ def forward() -> None: "lc could not set up the sandbox (see above) — this is an lc " "problem, not your command's" ) - elif returncode == 125 and oci_runtime is not None: + elif returncode == 125 and backend.contains_prefix: # The runtimes reserve 125 for their own failures (a bad flag, a # vanished mount source): the command never ran, so the denial # heuristics have nothing to say about it. @@ -248,7 +228,7 @@ def forward() -> None: f"above, `{attestation.mechanism}` exit 125) — this is a " "runtime problem, not your command's" ) - elif returncode != 0 and attestation.mechanism != "none" and not lifecycle_note: + elif returncode != 0 and attestation.mechanism != "none": from lightcone.engine.sandbox import denial explanation = denial.explain(tail.text(), policy, cwd=cwd) diff --git a/src/lightcone/engine/sandbox/model.py b/src/lightcone/engine/sandbox/model.py index 9389e34d..b8ad809b 100644 --- a/src/lightcone/engine/sandbox/model.py +++ b/src/lightcone/engine/sandbox/model.py @@ -26,16 +26,6 @@ from pathlib import Path from typing import Literal, Protocol -from lightcone.engine.project import ProjectError - - -class ExecutionUncertain(ProjectError): # noqa: N818 - """Command cleanup is unconfirmed; retain any files it could still write.""" - - -class ExecutionCancelled(ProjectError): # noqa: N818 - """Execution was revoked and its command has stopped.""" - #: Bumped when the meaning of the exec allowlist changes. It is recorded #: in the attestation, so a run stays interpretable after the list #: grows — the allowlist is a maintained policy surface. diff --git a/src/lightcone/engine/sandbox/oci.py b/src/lightcone/engine/sandbox/oci.py index dcc44451..0acbf4e9 100644 --- a/src/lightcone/engine/sandbox/oci.py +++ b/src/lightcone/engine/sandbox/oci.py @@ -24,7 +24,6 @@ from lightcone.engine.sandbox.boundary import SANDBOX_ENV from lightcone.engine.sandbox.model import Attestation, Capability, Policy -from lightcone.engine.sandbox.processes import CIDFILE #: The runtimes this backend can speak for — the one statement of the #: set, so the type does not get hand-copied out of step at its uses. @@ -84,12 +83,7 @@ def wrap(self, policy: Policy, argv: Sequence[str]) -> list[str]: mounts += [f"--volume={path.resolve()}:{path}:rw" for path in policy.write] overlay = [f"--env={k}={v}" for k, v in sorted(policy.env.items())] return [ - # The custodian retains the native record until it has inspected - # the immutable container ID and confirmed the payload stopped. - self.runtime, "run", - # The supervisor runs this native client in a private directory; - # --workdir below independently sets the payload's project cwd. - "--cidfile", CIDFILE, + self.runtime, "run", "--rm", "--entrypoint", "", # The rootfs is read-only so a write outside the declared set # is a loud denial rather than bytes vanishing with the diff --git a/src/lightcone/engine/sandbox/processes.py b/src/lightcone/engine/sandbox/processes.py deleted file mode 100644 index 921a6e2d..00000000 --- a/src/lightcone/engine/sandbox/processes.py +++ /dev/null @@ -1,386 +0,0 @@ -"""Keep custody of command processes independently of their Dask worker. - -The small child process owns the command's process group. Its control pipe -closes if the worker dies, so cleanup does not depend on a Python finally block -in that worker. Groups stay in the allocation's session for native shutdown. -""" - -from __future__ import annotations - -import json -import os -import re -import select -import signal -import subprocess -import sys -import tempfile -import time -from collections.abc import Callable, Sequence -from pathlib import Path -from types import FrameType, TracebackType -from typing import Any, Self - -import psutil - -_GRACE = 1.0 -_EXIT_GRACE = 1.0 -_CLEANUP_TIMEOUT = 15.0 -_REPORT_GRACE = 1.0 -CIDFILE = "container.cid" - - -def members(*, group: int | None = None, session: int | None = None) -> list[psutil.Process]: - """Find live owned processes, retaining their birth identities for signalling.""" - found = [] - for process in psutil.process_iter(): - try: - if process.uids().real != os.getuid(): - continue - except (psutil.NoSuchProcess, psutil.AccessDenied): - continue - try: - if process.status() == psutil.STATUS_ZOMBIE: - continue - if group is not None and os.getpgid(process.pid) != group: - continue - if session is not None and os.getsid(process.pid) != session: - continue - process.create_time() - found.append(process) - except (psutil.NoSuchProcess, ProcessLookupError): - continue - return found - - -def has_custodian(processes: Sequence[psutil.Process]) -> bool: - """Check whether allocation shutdown must allow command cleanup to finish.""" - for process in processes: - try: - if str(Path(__file__)) in process.cmdline(): - return True - except (psutil.NoSuchProcess, psutil.AccessDenied): - continue - return False - - -def _drain(process: subprocess.Popen[bytes], *, deadline: float | None = None) -> bool: - """Stop the whole command group while its unreaped leader pins the group ID.""" - if deadline is None: - deadline = time.monotonic() + _CLEANUP_TIMEOUT - for sig in (signal.SIGTERM, signal.SIGKILL): - if not members(group=process.pid): - process.wait() - return True - try: - os.killpg(process.pid, sig) - except ProcessLookupError: - pass - except PermissionError: - # Darwin reports EPERM when only zombies remain. Accept that race - # only after confirming no live group member still needs stopping. - if members(group=process.pid): - raise - process.wait() - return True - until = min(deadline, time.monotonic() + _GRACE) if sig == signal.SIGTERM else deadline - while members(group=process.pid): - if time.monotonic() >= until: - break - time.sleep(0.025) - else: - process.wait() - return True - return False - - -class Command: - """Launch a custodian and collect its verified completion report. - - Args: - argv: Fully wrapped command. - cwd: Command working directory. - env: Command environment. - capture: Whether to pipe stdout and disable stdin. - timeout: Maximum command lifetime including startup, or no bound, in seconds. - oci_runtime: The native OCI runtime, or None for a host command. - """ - - def __init__( - self, argv: Sequence[str], *, cwd: Path, env: dict[str, str], capture: bool, - timeout: float | None = None, oci_runtime: str | None = None, - ) -> None: - self._control, control_write = os.pipe() - status_read, self._status = os.pipe() - self._writer = os.fdopen(control_write, "wb", buffering=0) - self._reader = os.fdopen(status_read, "rb") - execution_deadline = time.monotonic() + timeout if timeout is not None else None - self._deadline = ( - execution_deadline + _CLEANUP_TIMEOUT + _REPORT_GRACE - if execution_deadline is not None else None - ) - try: - try: - self.process = subprocess.Popen( - [sys.executable, "-P", str(Path(__file__)), - str(self._control), str(self._status)], - pass_fds=(self._control, self._status), - stdin=subprocess.DEVNULL if capture else None, - stdout=subprocess.PIPE if capture else None, - stderr=subprocess.PIPE, - bufsize=0, - ) - finally: - # Only the custodian owns these ends. In particular, keeping - # its report writer here would prevent EOF during failed setup. - os.close(self._control) - os.close(self._status) - self._writer.write(json.dumps({ - "argv": list(argv), "cwd": str(cwd), "env": env, - "deadline": execution_deadline, "oci_runtime": oci_runtime, - }).encode() + b"\n") - except BaseException: - self._writer.close() - if hasattr(self, "process"): - try: - self.wait() - except Exception as cleanup_error: - from lightcone.engine.sandbox.model import ExecutionCancelled - - if not isinstance(cleanup_error, ExecutionCancelled): - raise - else: - self._reader.close() - raise - - def __enter__(self) -> Self: - return self - - def __exit__( - self, exc_type: type[BaseException] | None, exc: BaseException | None, - traceback: TracebackType | None, - ) -> None: - from lightcone.engine.sandbox.model import ExecutionCancelled - - if not self._reader.closed: - self._writer.close() - try: - self.wait() - except ExecutionCancelled: - pass - - def wait(self, cancelled: Callable[[], bool] | None = None) -> tuple[int, str]: - """Wait for command completion; cancellation includes verified cleanup.""" - from lightcone.engine.sandbox.model import ExecutionCancelled, ExecutionUncertain - - requested = self._writer.closed - deadline = ( - time.monotonic() + _CLEANUP_TIMEOUT + _REPORT_GRACE if requested else self._deadline - ) - try: - while self.process.poll() is None: - if not requested and cancelled is not None and cancelled(): - self._writer.close() - requested = True - deadline = time.monotonic() + _CLEANUP_TIMEOUT + _REPORT_GRACE - if deadline is not None and time.monotonic() >= deadline: - raise ExecutionUncertain("command cleanup did not finish") - time.sleep(0.025) - except BaseException: - self._writer.close() - try: - remaining = ( - _CLEANUP_TIMEOUT + _REPORT_GRACE if deadline is None - else min( - _CLEANUP_TIMEOUT + _REPORT_GRACE, - max(0.0, deadline - time.monotonic()), - ) - ) - self.process.wait(timeout=remaining) - except subprocess.TimeoutExpired as exc: - self._reader.close() - raise ExecutionUncertain("command cleanup did not finish") from exc - report = self._report() - if report.get("error"): - raise ExecutionUncertain(str(report["error"])) - raise - finally: - self._writer.close() - report = self._report() - if report.get("error"): - raise ExecutionUncertain(str(report["error"])) - if report.get("start_error"): - raise OSError(str(report["start_error"])) - if report.get("cancelled"): - raise ExecutionCancelled("command cancelled; its processes have stopped") - return int(report["returncode"]), str(report.get("note", "")) - - def _report(self) -> dict[str, Any]: - from lightcone.engine.sandbox.model import ExecutionUncertain - - try: - with self._reader: - result = json.load(self._reader) - if not isinstance(result, dict) or "returncode" not in result: - raise ValueError("missing completion record") - return result - except (OSError, ValueError) as exc: - raise ExecutionUncertain( - "command custodian ended without confirming that its processes stopped" - ) from exc - - -def _container_cleanup( - runtime: str, cidfile: Path, env: dict[str, str], *, deadline: float, -) -> None: - """Stop, inspect and remove exactly the container created by this command.""" - try: - identity = cidfile.read_text().strip() - except OSError as exc: - raise RuntimeError("container creation ended before its identity was recorded") from exc - if not re.fullmatch(r"[0-9a-f]{64}", identity): - raise RuntimeError("container runtime did not publish a valid immutable container ID") - - def run(*args: str) -> subprocess.CompletedProcess[str]: - remaining = deadline - time.monotonic() - if remaining <= 0: - raise TimeoutError("command cleanup deadline expired") - return subprocess.run( - [runtime, *args], env=env, stdin=subprocess.DEVNULL, - capture_output=True, text=True, timeout=remaining, - ) - - def running() -> bool: - result = run("inspect", "--format", "{{.State.Running}}", identity) - if result.returncode or result.stdout.strip() not in ("true", "false"): - raise RuntimeError(f"cannot establish whether container {identity} stopped") - return result.stdout.strip() == "true" - - try: - if running(): - try: - run("stop", "--time", "1", identity) - except subprocess.TimeoutExpired: - pass - if running(): - run("kill", identity) - while running(): - if time.monotonic() >= deadline: - raise RuntimeError(f"container {identity} remains running after kill") - time.sleep(0.025) - except BaseException: - # Inspection can fail while native termination still works. Attempt - # cleanup without converting that lack of evidence into success. - try: - run("kill", identity) - except (OSError, subprocess.SubprocessError): - pass - raise - # A stopped container is safe even when the runtime cannot remove its metadata. - try: - run("rm", identity) - except (OSError, subprocess.SubprocessError): - pass - - -def _supervise(control: int, status: int) -> None: - stopped = False - - def stop(_signum: int, _frame: FrameType | None) -> None: - nonlocal stopped - stopped = True - - signal.signal(signal.SIGTERM, stop) - signal.signal(signal.SIGINT, stop) - process: subprocess.Popen[bytes] | None = None - cleanup_deadline: float | None = None - report: dict[str, Any] = {"returncode": 125} - with os.fdopen(control, "rb", buffering=0) as channel, tempfile.TemporaryDirectory( - prefix="lc-command-" - ) as directory: - try: - config = json.loads(channel.readline()) - argv = config["argv"] - cidfile = Path(directory) / CIDFILE - oci_runtime = config["oci_runtime"] - if stopped or select.select([channel], [], [], 0)[0]: - report["cancelled"] = True - return - if config["deadline"] is not None and time.monotonic() >= config["deadline"]: - report.update(returncode=124, note="command timed out before startup") - return - process = subprocess.Popen( - argv, cwd=directory if oci_runtime else config["cwd"], - env=config["env"], process_group=0, - ) - reason = "" - # Keep the leader unreaped until group cleanup is complete, pinning - # its PID/PGID against reuse. Unlike waitid(WNOWAIT), psutil also - # supports macOS with Python 3.11 and 3.12. - leader = psutil.Process(process.pid) - while leader.status() != psutil.STATUS_ZOMBIE: - if stopped or select.select([channel], [], [], 0.025)[0]: - reason = "cancelled" - break - if ( - config["deadline"] is not None - and time.monotonic() >= config["deadline"] - ): - reason = "timed out" - break - # Helpers such as multiprocessing's resource_tracker finish after - # their parent closes its pipe. Give normal teardown a short grace. - exit_deadline = time.monotonic() + _EXIT_GRACE - while not reason and members(group=process.pid): - if stopped or select.select([channel], [], [], 0.025)[0]: - reason = "cancelled" - elif ( - config["deadline"] is not None - and time.monotonic() >= config["deadline"] - ): - reason = "timed out" - elif time.monotonic() >= exit_deadline: - reason = "left background processes running" - cleanup_deadline = time.monotonic() + _CLEANUP_TIMEOUT - container_stopped = False - if oci_runtime: - # A runtime startup failure with no CID has not published a - # container; interruption in that window cannot prove the same. - if cidfile.exists() or reason: - _container_cleanup( - oci_runtime, cidfile, config["env"], deadline=cleanup_deadline, - ) - # Podman removes its cidfile together with the container. - # Keep the verified outcome, not the file's later existence. - container_stopped = True - if not _drain(process, deadline=cleanup_deadline): - raise RuntimeError("command processes remain alive after SIGKILL") - if oci_runtime and not container_stopped and process.returncode != 125: - raise RuntimeError("container runtime exited without recording its identity") - report["returncode"] = process.returncode - if reason == "cancelled": - report["cancelled"] = True - report["returncode"] = 130 - elif reason: - report["returncode"] = 124 if reason == "timed out" else 1 - report["note"] = f"command {reason}; its processes have stopped" - except BaseException as exc: - if process is None: - report["start_error"] = f"command could not start: {str(exc)[:2048]}" - else: - try: - _drain(process, deadline=cleanup_deadline) - except BaseException: - pass - report["error"] = f"cannot confirm command cleanup: {str(exc)[:2048]}" - finally: - with os.fdopen(status, "wb", buffering=0) as result: - try: - result.write(json.dumps(report).encode()) - except BrokenPipeError: - # The worker can die before receiving the cleanup report. - pass - - -if __name__ == "__main__": - _supervise(int(sys.argv[1]), int(sys.argv[2])) diff --git a/src/lightcone/engine/worker.py b/src/lightcone/engine/worker.py index 68e4aa79..af1be88a 100644 --- a/src/lightcone/engine/worker.py +++ b/src/lightcone/engine/worker.py @@ -18,11 +18,11 @@ Keep this module cheap to import: no click, no rich. It is on the ``python -m`` path of every rerun, and of every task in every run. -Recipe failures return results, so independent tasks can finish and report -their own failures. Cancellation and uncertain execution propagate instead: -the driver must stop the invocation and establish that its writers have stopped -before it can restore partial outputs. Cluster task resource reservations belong -to the scheduler; the subprocess boundary enforces each recipe's time limit. +Nothing here writes to git, and nothing here raises. A task that fails +returns a result saying so, because Dask propagates an exception to every +dependent and "who actually failed" would stop being answerable — +reporting every independent failure in one run is most of the point of +owning the loop. """ from __future__ import annotations @@ -35,7 +35,7 @@ from pathlib import Path from typing import Literal -from lightcone.engine import assets, container, dataset, execution, identity, plan, project, sandbox +from lightcone.engine import assets, container, dataset, identity, plan, project, sandbox from lightcone.engine.plan import Key, Task from lightcone.engine.project import ( ProjectError, @@ -58,7 +58,7 @@ @dataclass(frozen=True) class TaskResult: - """A completed task's outcome, passed to its dependents.""" + """What one task did. Returned, never raised, and handed to dependents.""" key: Key status: Literal["ok", "current", "behind", "failed", "blocked"] @@ -126,9 +126,10 @@ def materialize( ) -> TaskResult: """Make *task* if it needs making. What Dask submits, once per task. - Ordinary failures become task results so independent outputs can continue. - Cancellation and uncertain execution abort the invocation: treating either - as an ordinary failure could restore files while a subprocess still writes. + Where "the worker never raises" is enforced. Dask re-raises a task's + exception in the driver, which would abort every other task in flight, + so the contract is absolute — and one assembled from individually + guarded call sites is only as true as the last person to add one. Args: root: The project root. @@ -145,17 +146,11 @@ def materialize( output: Optional receiver forwarding recipe stdout and stderr bytes. Returns: - The output's result, including ordinary recipe failures. - - Raises: - ExecutionCancelled: The invocation revoked this task's authorization. - ExecutionUncertain: A writer may still be running; retain its outputs. + What happened. Never raises. """ try: return _materialize(root, task, context, refresh, foreign, upstream, output) - except (execution.ExecutionCancelled, execution.ExecutionUncertain): - raise - except Exception as e: # Ordinary recipe failures remain per-output results. + except Exception as e: # the contract is that this function returns return TaskResult(task.key, "failed", reason=f"{type(e).__name__}: {e}") @@ -218,8 +213,7 @@ def execute( left from a previous run would otherwise enter the content hash and be committed as part of an output that never produced it. The context's ``env_version`` is checked either side of the recipe, so a mid-run - lock edit cannot be recorded as if it had been in force. The subprocess - boundary drains owned processes before reporting completion. + lock edit cannot be recorded as if it had been in force. Args: root: The project root. @@ -232,12 +226,7 @@ def execute( Returns: ``ok`` with the output's ``data_version``, or ``failed``. Commits nothing and never touches git beyond reading HEAD. - - Raises: - ExecutionCancelled: The invocation revoked this task's authorization. - ExecutionUncertain: Subprocess teardown could not be confirmed. """ - execution.check_cancelled() if moved := _gate(root, context.env_version): return TaskResult(task.key, "failed", reason=moved) @@ -269,12 +258,9 @@ def execute( prefix=uv_prefix(root), env=child_env(), output=output, - cancelled=execution.cancelled, ) finished_at = _now() - execution.check_cancelled() - if outcome.returncode != 0: return TaskResult( task.key, diff --git a/tests/conftest.py b/tests/conftest.py index 25c1cd0a..2a4d164a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -151,11 +151,7 @@ class _Inline: are the upstream results themselves, exactly what the worker expects. """ - stopped = True # Calls are synchronous; no remote work can survive the fixture. - - def submit( - self, fn: Callable[..., object], *args: object, key: str, - ) -> object: + def submit(self, fn: Callable[..., object], *args: object, key: str) -> object: return fn(*args) def completed(self, handles: list[object]) -> Iterator[object]: @@ -184,7 +180,7 @@ def cluster_id(monkeypatch: pytest.MonkeyPatch) -> Iterator[str]: from lightcone.engine import compute with LocalCluster( - n_workers=1, threads_per_worker=2, processes=False, dashboard_address=None, + n_workers=1, threads_per_worker=2, processes=False, dashboard_address=None ) as cluster: @contextmanager def connect(value: str) -> Iterator[Client]: diff --git a/tests/test_cli.py b/tests/test_cli.py index 1f6be9a4..2fb60cc7 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -457,29 +457,21 @@ def test_check_explains_that_a_cluster_id_is_not_a_target(runner: CliRunner) -> @pytest.mark.parametrize("command", ["run", "materialize"]) @pytest.mark.parametrize("cluster", [CLUSTER_ID, "analysis"]) -@pytest.mark.parametrize("stopped", [False, True]) def test_execution_interrupt_explains_how_to_stop_remote_work( runner: CliRunner, project: Path, monkeypatch: pytest.MonkeyPatch, command: str, - cluster: str, stopped: bool, + cluster: str, ) -> None: from lightcone.engine import materialize as engine_materialize from lightcone.engine import run as engine_run def interrupt(*args: object, **kwargs: object) -> None: - from lightcone.engine.execution import ExecutionInterrupted - - raise ExecutionInterrupted() if stopped else KeyboardInterrupt() + raise KeyboardInterrupt monkeypatch.setattr(engine_run, "probe", interrupt) monkeypatch.setattr(engine_materialize, "materialize", interrupt) args = [command, cluster, "--", "true"] if command == "run" else [command, cluster] result = runner.invoke(main, args) assert result.exit_code != 0 - if stopped: - assert "stopped" in result.output - assert "remains available" in result.output - assert "lc compute down" not in result.output - return target = cluster if cluster == CLUSTER_ID else "" assert f"lc compute down {target}" in result.output if cluster != CLUSTER_ID: diff --git a/tests/test_compute_local.py b/tests/test_compute_local.py index 31b9358b..0da9d92a 100644 --- a/tests/test_compute_local.py +++ b/tests/test_compute_local.py @@ -367,7 +367,7 @@ def test_owner_shutdown_drains_recipes_without_a_waiting_cli(provider: LocalProv _ready(provider, identity) child = _ignoring_recipe(provider, identity) os.kill(int(identity.native_id), signal.SIGTERM) - _ended(provider, identity, timeout=20) + _ended(provider, identity, timeout=10) deadline = time.monotonic() + 3 while child.is_running() and child.status() != psutil.STATUS_ZOMBIE: assert time.monotonic() < deadline @@ -378,53 +378,6 @@ def test_owner_shutdown_drains_recipes_without_a_waiting_cli(provider: LocalProv child.kill() -@pytest.mark.parametrize("after_capture", [False, True]) -def test_owner_walltime_still_hard_kills_when_process_enumeration_fails( - after_capture: bool, -) -> None: - script = f""" -import os, signal, subprocess, sys, time -from lightcone.engine.compute import local_runtime - -after_capture = {after_capture!r} -called = False -def fail(**kwargs): - global called - if after_capture and not called: - called = True - return [local_runtime.psutil.Process(child.pid)] - raise RuntimeError('process enumeration unavailable') - -child = subprocess.Popen([sys.executable, '-c', - "import signal,time; signal.signal(signal.SIGTERM, signal.SIG_IGN); " - "print('ready',flush=True); time.sleep(60)"], stdout=subprocess.PIPE, - process_group=0 if after_capture else os.getpgrp()) -assert child.stdout.readline() == b'ready\\n' -print(child.pid, flush=True) -local_runtime.members = fail -local_runtime._CLEANUP_TIMEOUT = .1 -signal.signal(signal.SIGTERM, signal.SIG_IGN) -signal.signal(signal.SIGALRM, lambda *_: local_runtime._stop_session()) -signal.setitimer(signal.ITIMER_REAL, .1) -time.sleep(60) -""" - owner = subprocess.Popen( - [sys.executable, "-c", script], start_new_session=True, stdout=subprocess.PIPE, - ) - assert owner.stdout is not None - child = psutil.Process(int(owner.stdout.readline())) - try: - assert owner.wait(timeout=5) == -signal.SIGKILL - assert not child.is_running() or child.status() == psutil.STATUS_ZOMBIE - finally: - if owner.poll() is None: - owner.kill() - owner.wait(timeout=5) - if child.is_running(): - child.kill() - owner.stdout.close() - - def test_termination_escalates_captured_children_when_owner_exits_first( provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, ) -> None: @@ -434,7 +387,7 @@ def test_termination_escalates_captured_children_when_owner_exits_first( import signal, subprocess, sys, time child = subprocess.Popen([sys.executable, '-c', "import signal,time; signal.signal(signal.SIGTERM, signal.SIG_IGN); " - "print('ready',flush=True); time.sleep(120)"], stdout=subprocess.PIPE, process_group=0) + "print('ready',flush=True); time.sleep(120)"], stdout=subprocess.PIPE) assert child.stdout.readline() == b'ready\\n' print(child.pid, flush=True) signal.signal(signal.SIGTERM, lambda *_: sys.exit(0)) @@ -445,7 +398,6 @@ def test_termination_escalates_captured_children_when_owner_exits_first( ) assert owner.stdout is not None child = psutil.Process(int(owner.stdout.readline())) - assert os.getpgid(child.pid) != owner.pid process = psutil.Process(owner.pid) identity = Identity( namespace=provider.connection.namespace, native_id=str(owner.pid), token=uuid4().hex, diff --git a/tests/test_compute_output.py b/tests/test_compute_output.py index b1a2bf82..7a9ebf5e 100644 --- a/tests/test_compute_output.py +++ b/tests/test_compute_output.py @@ -10,10 +10,8 @@ import time from collections.abc import Callable, Iterator from pathlib import Path -from types import SimpleNamespace from uuid import uuid4 -import psutil import pytest from lightcone.engine import sandbox @@ -93,90 +91,38 @@ def test_probe_uses_allocation_environment_and_accepts_explicit_command_variable assert result.stdout == expected -@pytest.mark.parametrize( - "stop_signal", [signal.SIGINT, signal.SIGKILL], ids=["interrupt", "client-loss"], -) -def test_interrupt_or_client_loss_stops_command_and_keeps_cluster_usable( - analysis: Callable[..., Path], detached_cluster: str, stop_signal: signal.Signals, +def test_interrupt_warns_that_the_remote_command_may_still_run( + analysis: Callable[..., Path], detached_cluster: str, ) -> None: root = analysis("version: '0.0.13'\nname: analysis\ninputs: []\noutputs: []\n") started = root / "results/started" - cli = [sys.executable, "-c", "from lightcone.cli.commands import main; main()", - "run", detached_cluster, "--"] process = subprocess.Popen( [ - *cli, "python", "-c", - "from pathlib import Path; import os, signal, time; " - "signal.signal(signal.SIGTERM, signal.SIG_IGN); " - "Path('results/started').write_text(str(os.getpid())); time.sleep(60)", + sys.executable, "-c", "from lightcone.cli.commands import main; main()", + "run", detached_cluster, "--", "python", "-c", + "from pathlib import Path; import time; " + "Path('results/started').touch(); time.sleep(30)", ], cwd=root, stdout=subprocess.PIPE, stderr=subprocess.PIPE, ) - remote: psutil.Process | None = None try: deadline = time.monotonic() + 15 - while not started.exists() or not started.read_text(): + while not started.exists(): if process.poll() is not None: pytest.fail(process.communicate()[1].decode(errors="replace")) if time.monotonic() >= deadline: pytest.fail("remote command did not start") time.sleep(0.05) - remote = psutil.Process(int(started.read_text())) - remote.create_time() - process.send_signal(stop_signal) - _, stderr = process.communicate(timeout=30) + process.send_signal(signal.SIGINT) + _, stderr = process.communicate(timeout=15) assert process.returncode != 0 - if stop_signal == signal.SIGINT: - assert b"the command has stopped" in stderr - assert b"may still be running" not in stderr - deadline = time.monotonic() + (0 if stop_signal == signal.SIGINT else 30) - while True: - try: - alive = remote.is_running() and remote.status() != psutil.STATUS_ZOMBIE - except psutil.NoSuchProcess: - alive = False - if not alive: - break - assert time.monotonic() < deadline, "the departed CLI left its command running" - time.sleep(0.05) + assert b"may still be running" in stderr + assert f"lc compute down {detached_cluster}".encode() in stderr assert Compute().status(detached_cluster).phase == "active" - again = subprocess.run( - [*cli, "python", "-c", "print('still usable')"], - cwd=root, capture_output=True, timeout=30, - ) - assert again.returncode == 0, again.stderr.decode(errors="replace") - assert again.stdout == b"still usable\n" finally: if process.poll() is None: process.kill() process.wait(timeout=5) - if remote is not None and remote.is_running(): - try: - remote.kill() - except psutil.NoSuchProcess: - pass - - -def test_failed_output_completion_marker_preserves_execution_uncertainty( - monkeypatch: pytest.MonkeyPatch, -) -> None: - import distributed - - from lightcone.engine.compute.output import call - from lightcone.engine.execution import ExecutionUncertain - - uncertainty = ExecutionUncertain("container termination could not be confirmed") - - def execute(*, output: Callable[[str, bytes], None]) -> None: - raise uncertainty - - def log_event(*args: object) -> None: - raise OSError("scheduler disconnected") - - monkeypatch.setattr(distributed, "get_worker", lambda: SimpleNamespace(log_event=log_event)) - with pytest.raises(ExecutionUncertain) as raised: - call(execute, "topic", "probe") - assert raised.value is uncertainty def test_dask_cleans_output_history_after_the_borrowed_client_disconnects( diff --git a/tests/test_container_smoke.py b/tests/test_container_smoke.py index 7eb3c876..8d023438 100644 --- a/tests/test_container_smoke.py +++ b/tests/test_container_smoke.py @@ -19,17 +19,14 @@ import subprocess import sys from collections.abc import Callable -from dataclasses import replace from pathlib import Path -from uuid import uuid4 import pytest -from lightcone.engine import assets, container, dataset, image, sandbox +from lightcone.engine import assets, container, dataset, image from lightcone.engine import materialize as engine from lightcone.engine import run as engine_run -from lightcone.engine.project import ProjectError, child_env, uv_prefix -from lightcone.engine.sandbox.oci import OCIBackend +from lightcone.engine.project import ProjectError, child_env REQUIRED_ENV = "LC_CONTAINER_TESTS_REQUIRED" @@ -213,44 +210,6 @@ def test_the_probe_and_its_boundary(runtime: str, cproject: Path, cluster_id: st assert loopback.returncode == 0 -def test_timeout_stops_and_removes_a_sigterm_ignoring_container( - runtime: str, cproject: Path, -) -> None: - resolved, _ = container.build(cproject) - container.converge(resolved) - backend = container.backend(resolved) - assert isinstance(backend, OCIBackend) - name = f"lc-timeout-test-{uuid4().hex}" - backend = replace(backend, user_flags=(*backend.user_flags, "--name", name)) - try: - with sandbox.scope(container.policy_for(resolved, [])) as policy: - outcome = sandbox.run( - backend, policy, - ["python", "-c", ( - "import signal,time; from pathlib import Path; " - "signal.signal(signal.SIGTERM, signal.SIG_IGN); " - "Path('results/timeout-started').touch(); time.sleep(60)" - )], cwd=cproject, env=child_env(), prefix=uv_prefix(cproject), timeout=5, - ) - assert (cproject / "results/timeout-started").exists(), "container command never started" - assert outcome.returncode == 124 - assert any("timed out" in note for note in outcome.notes) - inspected = subprocess.run( - [runtime, "inspect", name], capture_output=True, timeout=10, - ) - assert inspected.returncode != 0, "timed-out container was not removed" - # A failed inspection alone could mean the runtime is unavailable. - remaining = subprocess.run( - [runtime, "ps", "--all", "--quiet", "--filter", f"name={name}"], - capture_output=True, check=True, timeout=10, - ) - assert not remaining.stdout.strip() - finally: - subprocess.run( - [runtime, "rm", "--force", name], capture_output=True, timeout=15, - ) - - # ---- lc materialize --------------------------------------------------------- diff --git a/tests/test_execution.py b/tests/test_execution.py index 46c6fb14..0092a3aa 100644 --- a/tests/test_execution.py +++ b/tests/test_execution.py @@ -1,10 +1,9 @@ -"""Ordinary Dask tasks must not repeat effects or outlive their invocation silently.""" +"""One scheduler claim prevents Dask from replaying a side-effecting task.""" from __future__ import annotations -import threading import time -from collections.abc import Iterator +from collections.abc import Callable, Iterator from pathlib import Path from types import SimpleNamespace from typing import Any @@ -14,14 +13,11 @@ from distributed import Client, LocalCluster, get_worker from lightcone.engine import execution -from lightcone.engine.worker import TaskResult +from lightcone.engine.project import ProjectError @pytest.fixture -def execution_client(monkeypatch: pytest.MonkeyPatch) -> Iterator[Client]: - monkeypatch.setattr(execution, "_HEARTBEAT", 0.05) - monkeypatch.setattr(execution, "_RPC_TIMEOUT", 2.0) - monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.75) +def execution_client() -> Iterator[Client]: with LocalCluster( n_workers=2, threads_per_worker=1, processes=False, dashboard_address=None, memory_limit=0, @@ -29,554 +25,203 @@ def execution_client(monkeypatch: pytest.MonkeyPatch) -> Iterator[Client]: yield client -def _wait_for(path: Path) -> None: - deadline = time.monotonic() + 5 - while not path.exists(): - if time.monotonic() > deadline: - pytest.fail(f"worker did not create {path.name}") +def _wait_until(condition: Callable[[], bool]) -> None: + deadline = time.monotonic() + 10 + while not condition(): + assert time.monotonic() < deadline, "Dask task did not reach the expected state" time.sleep(0.01) -def _effect(path: Path) -> TaskResult: +def _effect(path: Path, fail: bool = False) -> str: + address = get_worker().address with path.open("a") as stream: - stream.write("executed\n") - return TaskResult(("universe", "output"), "ok", notes=(get_worker().address,)) + stream.write(address + "\n") + if fail: + raise ValueError("recipe failed after writing") + return address -def _wait_for_release(started: Path, release: Path) -> str: - started.touch() - deadline = time.monotonic() + 5 - while not release.exists(): - if time.monotonic() > deadline: - raise AssertionError("test did not release its running task") - time.sleep(0.01) - return "original" - - -def _cooperate(started: Path, stopped: Path, timeout: float = 5) -> None: - started.touch() - deadline = time.monotonic() + timeout - while not execution.cancelled(): - if time.monotonic() > deadline: - raise AssertionError("task did not observe its invocation's cancellation") - time.sleep(0.01) - stopped.touch() - execution.check_cancelled() - - -def _cooperating_effect(started: Path, stopped: Path) -> None: - with started.open("a") as stream: - stream.write(get_worker().address + "\n") - # Publish readiness after closing the append, never during file creation. - started.with_suffix(".ready").touch() - _cooperate(started, stopped) +def _running_effect(path: Path, release: Path, finished: Path) -> None: + _effect(path) + # Publish readiness after the append is closed, including on slow CI hosts. + path.with_suffix(".ready").touch() + try: + _wait_until(release.exists) + finally: + finished.touch() def _remove_worker(client: Client, address: str) -> None: worker = next(worker for worker in client.cluster.workers.values() if worker.address == address) - # close(), unlike close_gracefully(), discards this worker's task data. - # Keeping its executor alive also models a partitioned task that can still - # write even though the scheduler has reassigned its Dask key elsewhere. - client.cluster.sync(worker.close, executor_wait=False, timeout=0.5) - + # Discard its data while keeping an executing thread alive: scheduler loss + # does not itself prove that the original command has stopped writing. + client.cluster.sync(worker.close, executor_wait=False, timeout=1) -def _raise(error: Exception) -> None: - raise error - -def _stay_authorized(seconds: float) -> str: - deadline = time.monotonic() + seconds - while time.monotonic() < deadline: - execution.check_cancelled() - time.sleep(0.01) - return "completed" - - -def _records(*, dask_scheduler: Any) -> dict[str, Any]: - return dask_scheduler.extensions.get("lightcone-executions", {}) +def _replay(client: Client, invocation: str, task: str, path: Path) -> Any: + return client.submit( + execution._call, invocation, task, _effect, path, + key=f"replay-{uuid4().hex}", pure=False, retries=0, + ) def _lose_state(*, dask_scheduler: Any) -> None: dask_scheduler.extensions.pop("lightcone-executions", None) -def test_completed_task_replay_on_another_worker_returns_its_original_receipt( - execution_client: Client, tmp_path: Path, -) -> None: - effects = tmp_path / "effects" - with execution.invocation(execution_client) as run: - assert not run.stopped - original = run.submit(_effect, effects, key="recipe").result() - other = next(address for address in execution_client.scheduler_info()["workers"] - if address != original.notes[0]) - replay = execution_client.submit( - execution._call, run.id, "recipe", _effect, effects, - key=f"replay-{uuid4().hex}", workers=[other], allow_other_workers=False, - pure=False, - ).result() - assert replay == original - assert effects.read_text() == "executed\n" - assert run.stopped - assert run.id not in execution_client.run_on_scheduler(_records) - - -def test_worker_loss_recomputes_the_dask_future_without_repeating_completed_effects( +def test_completed_worker_loss_refuses_recomputation( execution_client: Client, tmp_path: Path, ) -> None: effects = tmp_path / "effects" with execution.invocation(execution_client) as run: future = run.submit(_effect, effects, key="recipe") - original = future.result(timeout=3) - lost_worker = original.notes[0] - _remove_worker(execution_client, lost_worker) - deadline = time.monotonic() + 3 - while True: - holders = execution_client.who_has([future])[future.key] - if holders and lost_worker not in holders: - break - assert time.monotonic() < deadline, "Dask did not recompute the lost result" - time.sleep(0.01) - assert future.result(timeout=3) == original - assert effects.read_text() == "executed\n" - - -def test_worker_loss_cannot_replay_effects_while_the_original_execution_still_runs( - execution_client: Client, tmp_path: Path, -) -> None: - started, stopped = tmp_path / "started", tmp_path / "stopped" - with execution.invocation(execution_client) as run: - future = run.submit( - _cooperating_effect, started, stopped, key="recipe", - ) - _wait_for(started.with_suffix(".ready")) - lost_worker = started.read_text().strip() - _remove_worker(execution_client, lost_worker) - with pytest.raises(execution.ExecutionUncertain, match="previous attempt"): - future.result(timeout=3) - _wait_for(stopped) - assert started.read_text().splitlines() == [lost_worker] - assert run.id not in execution_client.run_on_scheduler(_records) - - -def test_duplicate_running_attempt_cannot_execute_or_finish_the_original_claim( + address = future.result(timeout=10) + _remove_worker(execution_client, address) + # A previously finished Future can still contain its old result until + # the scheduler tells this client that recomputation has failed. + _wait_until(lambda: future.status == "error") + with pytest.raises(ProjectError, match="already claimed"): + future.result(timeout=10) + assert effects.read_text().splitlines() == [address] + + +def test_running_worker_loss_refuses_replay_while_original_can_still_write( execution_client: Client, tmp_path: Path, ) -> None: - started, release, duplicate_effect = ( - tmp_path / "started", tmp_path / "release", tmp_path / "duplicate" - ) + effects, release, finished = (tmp_path / name for name in ("effects", "release", "finished")) with execution.invocation(execution_client) as run: - original = run.submit( - _wait_for_release, started, release, key="recipe", - ) + future = run.submit(_running_effect, effects, release, finished, key="recipe") try: - _wait_for(started) - duplicate = execution_client.submit( - execution._call, run.id, "recipe", _effect, duplicate_effect, - key=f"duplicate-{uuid4().hex}", pure=False, - ) - with pytest.raises(execution.ExecutionUncertain, match="previous attempt"): - duplicate.result(timeout=3) - assert execution._rpc(execution_client, run.id, "pending").running == ("recipe",) - assert not duplicate_effect.exists() + _wait_until(effects.with_suffix(".ready").exists) + address = effects.read_text().strip() + _remove_worker(execution_client, address) + with pytest.raises(ProjectError, match="already claimed"): + future.result(timeout=10) + assert not finished.exists() + assert effects.read_text().splitlines() == [address] finally: release.touch() - try: - assert original.result(timeout=3) == "original" - except execution.ExecutionCancelled: - # Rejecting the ambiguous duplicate may revoke the invocation before - # the original completes. Only that original may acknowledge its stop. - pass - assert execution._rpc(execution_client, run.id, "pending").running == () + _wait_until(finished.exists) -def test_late_dispatch_after_invocation_exit_cannot_recreate_authorization( +def test_task_failure_does_not_release_its_claim( execution_client: Client, tmp_path: Path, ) -> None: effects = tmp_path / "effects" with execution.invocation(execution_client) as run: - pass - late = execution_client.submit( - execution._call, run.id, "late", _effect, effects, pure=False, - ) - with pytest.raises(execution.ExecutionUncertain, match="no longer registered"): - late.result(timeout=3) - assert not effects.exists() - assert run.id not in execution_client.run_on_scheduler(_records) + with pytest.raises(ValueError, match="recipe failed"): + run.submit(_effect, effects, True, key="recipe").result(timeout=10) + with pytest.raises(ProjectError, match="already claimed"): + _replay(execution_client, run.id, "recipe", effects).result(timeout=10) + assert len(effects.read_text().splitlines()) == 1 -def test_missing_scheduler_state_refuses_both_replay_and_unstarted_tasks( +def test_dask_forgetting_and_recreating_a_task_does_not_forget_its_claim( execution_client: Client, tmp_path: Path, ) -> None: effects = tmp_path / "effects" with execution.invocation(execution_client) as run: - run.submit(_effect, effects, key="recipe").result() - execution_client.run_on_scheduler(_lose_state) - for key in ("recipe", "new-recipe"): - future = execution_client.submit( - execution._call, run.id, key, _effect, effects, - key=f"lost-state-{uuid4().hex}", pure=False, - ) - with pytest.raises(execution.ExecutionUncertain, match="no longer registered"): - future.result(timeout=3) - assert effects.read_text() == "executed\n" - assert run.stopped # The only admitted task returned its confirmed completion. - - -def test_missing_scheduler_state_stops_running_work_without_claiming_confirmed_cleanup( - execution_client: Client, tmp_path: Path, + original = run.submit(_effect, effects, key="recipe") + original.result(timeout=10) + key = original.key + original.release() + _wait_until(lambda: execution_client.run_on_scheduler( + lambda dask_scheduler: key not in dask_scheduler.tasks, + )) + with pytest.raises(ProjectError, match="already claimed"): + run.submit(_effect, effects, key="recipe").result(timeout=10) + assert len(effects.read_text().splitlines()) == 1 + + +@pytest.mark.parametrize( + "state_lost", [False, True], ids=["context-exited", "scheduler-state-lost"], +) +def test_missing_invocation_refuses_replays_and_late_new_tasks( + execution_client: Client, tmp_path: Path, state_lost: bool, ) -> None: - started, stopped = tmp_path / "started", tmp_path / "stopped" - with pytest.raises(execution.ExecutionUncertain, match="confirm execution stopped"): - with execution.invocation(execution_client) as run: - future = run.submit( - _cooperate, started, stopped, key="recipe", - ) - _wait_for(started) + effects = tmp_path / "effects" + with execution.invocation(execution_client) as run: + run.submit(_effect, effects, key="recipe").result(timeout=10) + if state_lost: execution_client.run_on_scheduler(_lose_state) - with pytest.raises(execution.ExecutionCancelled, match="cancelled"): - future.result(timeout=3) - assert stopped.exists() + with pytest.raises(ProjectError): + _replay(execution_client, run.id, "new-recipe", effects).result(timeout=10) + for key in ("recipe", "new-recipe"): + with pytest.raises(ProjectError): + _replay(execution_client, run.id, key, effects).result(timeout=10) + assert len(effects.read_text().splitlines()) == 1 -def test_lost_claim_response_never_starts_the_recipe_and_retains_its_unresolved_claim( +def test_lost_claim_reply_never_executes_and_cannot_be_retried( execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: effects = tmp_path / "effects" request = execution._rpc - def lose_claim_response(*args: Any, **kwargs: Any) -> Any: - result = request(*args, **kwargs) - if args[2] == "claim": - raise TimeoutError("claim accepted but reply lost") + def lose_reply(client: Any, function: Any, *args: Any) -> Any: + result = request(client, function, *args) + if function is execution._claim: + raise ProjectError("claim accepted but reply lost") return result - monkeypatch.setattr(execution, "_rpc", lose_claim_response) - monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.1) - with pytest.raises(execution.ExecutionUncertain, match="unconfirmed tasks: recipe"): - with execution.invocation(execution_client) as run: - future = run.submit(_effect, effects, key="recipe") - with pytest.raises(TimeoutError, match="reply lost"): - future.result(timeout=3) - assert execution._rpc(execution_client, run.id, "pending").running == ("recipe",) - assert not effects.exists() - - -@pytest.mark.parametrize("loss", ["lease", "client"]) -def test_expired_or_disconnected_invocations_cannot_be_revived_by_heartbeat(loss: str) -> None: - scheduler = SimpleNamespace(extensions={}, clients={"driver": object()}) - execution._state("invocation", "register", value="driver", dask_scheduler=scheduler) - record = scheduler.extensions["lightcone-executions"]["invocation"] - if loss == "lease": - record["deadline"] = 0 - else: - scheduler.clients.clear() - assert not execution._state("invocation", "heartbeat", dask_scheduler=scheduler) - # Neither a fresh connection nor a late heartbeat can resurrect permission. - scheduler.clients["driver"] = object() - record["deadline"] = time.monotonic() + 60 - assert not execution._state("invocation", "heartbeat", dask_scheduler=scheduler) - with pytest.raises(execution.ExecutionCancelled, match="no longer active"): - execution._state("invocation", "claim", "recipe", dask_scheduler=scheduler) - - -def test_invocation_exit_cancels_the_future_and_waits_for_task_cooperation( - execution_client: Client, tmp_path: Path, -) -> None: - started, stopped = tmp_path / "started", tmp_path / "stopped" with execution.invocation(execution_client) as run: - future = run.submit(_cooperate, started, stopped, key="recipe") - _wait_for(started) - assert future.cancelled() - assert stopped.exists() - assert run.stopped - assert run.id not in execution_client.run_on_scheduler(_records) - - -def test_confirmed_cleanup_survives_an_error_in_the_invoking_driver( - execution_client: Client, tmp_path: Path, -) -> None: - started, stopped = tmp_path / "started", tmp_path / "stopped" - with pytest.raises(ValueError, match="driver failed to commit"): - with execution.invocation(execution_client) as run: - run.submit(_cooperate, started, stopped, key="recipe") - _wait_for(started) - raise ValueError("driver failed to commit") - assert stopped.exists() - assert run.stopped + with monkeypatch.context() as patch: + patch.setattr(execution, "_rpc", lose_reply) + with pytest.raises(ProjectError, match="reply lost"): + run.submit(_effect, effects, key="recipe").result(timeout=10) + with pytest.raises(ProjectError, match="already claimed"): + _replay(execution_client, run.id, "recipe", effects).result(timeout=10) + assert not effects.exists() -def test_driver_disconnect_revokes_execution_even_while_an_observer_remains_connected( +def test_claims_are_scoped_to_one_invocation( execution_client: Client, tmp_path: Path, ) -> None: - started, stopped = tmp_path / "started", tmp_path / "stopped" - with Client(execution_client.scheduler.address, set_as_default=False) as owner: - run = execution.Invocation(owner) - execution._rpc(owner, run.id, "register", value=owner.id) - run.submit(_cooperate, started, stopped, key="recipe") - _wait_for(started) - _wait_for(stopped) - assert not execution._rpc(execution_client, run.id, "heartbeat") - deadline = time.monotonic() + 3 - while execution._rpc(execution_client, run.id, "pending").running: - assert time.monotonic() < deadline, "task never acknowledged that it stopped" - time.sleep(0.01) - execution._rpc(execution_client, run.id, "forget") - - -def test_a_returned_failed_recipe_result_is_cached_without_rerunning( - execution_client: Client, -) -> None: - failed = TaskResult(("universe", "output"), "failed", reason="recipe exited 2") - with execution.invocation(execution_client) as run: - assert run.submit(lambda: failed, key="recipe").result() == failed - replay = execution_client.submit( - execution._call, run.id, "recipe", _raise, AssertionError("must not execute"), - pure=False, - ) - assert replay.result(timeout=3) == failed - - -def test_function_exception_confirms_stop_but_does_not_authorize_reexecution( - execution_client: Client, -) -> None: - with execution.invocation(execution_client) as run: - future = run.submit(_raise, ValueError("recipe failed"), key="recipe") - with pytest.raises(ValueError, match="recipe failed"): - future.result(timeout=3) - assert execution._rpc(execution_client, run.id, "pending").running == () - replay = execution_client.submit( - execution._call, run.id, "recipe", _raise, AssertionError("must not execute"), - pure=False, - ) - with pytest.raises(execution.ExecutionUncertain, match="previous attempt"): - replay.result(timeout=3) - - -def test_uncertain_task_remains_unresolved_after_context_exit( - execution_client: Client, monkeypatch: pytest.MonkeyPatch, -) -> None: - monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.1) - with pytest.raises(execution.ExecutionUncertain, match="unconfirmed tasks: recipe"): - with execution.invocation(execution_client) as run: - future = run.submit( - _raise, execution.ExecutionUncertain("container may still be alive"), - key="recipe", - ) - with pytest.raises(execution.ExecutionUncertain, match="container may still"): - future.result(timeout=3) - assert execution._rpc(execution_client, run.id, "pending").uncertain == ("recipe",) - assert execution._rpc(execution_client, run.id, "pending").uncertain == ("recipe",) - assert not execution._rpc(execution_client, run.id, "heartbeat") - assert not run.stopped - - -def test_uncertain_cleanup_prevents_later_tasks_from_starting( - execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - monkeypatch.setattr(execution, "_STOP_TIMEOUT", 0.1) effects = tmp_path / "effects" - with pytest.raises(execution.ExecutionUncertain, match="unconfirmed tasks: first"): + for _ in range(2): with execution.invocation(execution_client) as run: - first = run.submit( - _raise, execution.ExecutionUncertain("container may still consume memory"), - key="first", - ) - with pytest.raises(execution.ExecutionUncertain, match="container may still"): - first.result(timeout=3) - # Dask reported the first task's error, but its external work may - # survive. Authorization must close before another task runs. - later = run.submit(_effect, effects, key="later") - error = later.exception(timeout=3) - assert isinstance(error, execution.ExecutionCancelled) - assert not effects.exists() - - -def test_transient_heartbeat_and_monitor_failures_preserve_the_confirmed_lease( - execution_client: Client, monkeypatch: pytest.MonkeyPatch, -) -> None: - # A CI runner may pause for longer than the old 400 ms test lease. Keep - # actual lease renewal under test without mistaking that pause for an outage. - lease = 3.0 - monkeypatch.setattr(execution, "_LEASE", lease) - request = execution._rpc - calls = {"heartbeat": 0, "active": 0} - recovered = {operation: threading.Event() for operation in calls} - - def fail_once(*args: Any, **kwargs: Any) -> Any: - operation = args[2] - if operation in calls: - calls[operation] += 1 - if calls[operation] == 1: - raise TimeoutError("temporary scheduler RPC failure") - result = request(*args, **kwargs) - if operation in recovered and result: - recovered[operation].set() - return result - - monkeypatch.setattr(execution, "_rpc", fail_once) - with execution.invocation(execution_client) as run: - future = run.submit(_stay_authorized, 2 * lease, key="recipe") - for operation, event in recovered.items(): - assert event.wait(timeout=10), f"{operation} did not recover after its failed RPC" - assert future.result(timeout=10) == "completed" - assert calls["heartbeat"] > 2 - assert calls["active"] > 2 - assert run.stopped + run.submit(_effect, effects, key="recipe").result(timeout=10) + assert len(effects.read_text().splitlines()) == 2 -def test_unreachable_monitor_expires_its_last_grant_even_when_driver_heartbeats_continue( - execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - monkeypatch.setattr(execution, "_LEASE", 3.0) - request = execution._rpc - failed_polls = 0 - monitor_ready = threading.Event() - disconnected = threading.Event() - driver_renewed = threading.Event() - - def lose_monitor(*args: Any, **kwargs: Any) -> Any: - nonlocal failed_polls - operation = args[2] - if operation == "active" and disconnected.is_set(): - failed_polls += 1 - raise TimeoutError("worker cannot reach scheduler") - result = request(*args, **kwargs) - if operation == "active" and result: - monitor_ready.set() - if operation == "heartbeat" and disconnected.is_set() and result: - driver_renewed.set() - return result +def test_disconnected_owner_cannot_admit_tasks_and_registration_prunes_stale_records() -> None: + scheduler = SimpleNamespace(extensions={}, clients={"owner": object()}) + execution._register("first", "owner", dask_scheduler=scheduler) + execution._claim("first", "recipe", dask_scheduler=scheduler) + with pytest.raises(ProjectError): + execution._register("first", "owner", dask_scheduler=scheduler) + with pytest.raises(ProjectError, match="already claimed"): + execution._claim("first", "recipe", dask_scheduler=scheduler) + scheduler.clients.clear() + with pytest.raises(ProjectError): + execution._claim("first", "late-recipe", dask_scheduler=scheduler) + scheduler.clients["new-owner"] = object() + execution._register("second", "new-owner", dask_scheduler=scheduler) + assert "first" not in scheduler.extensions["lightcone-executions"] + execution._claim("second", "recipe", dask_scheduler=scheduler) - monkeypatch.setattr(execution, "_rpc", lose_monitor) - started, stopped = tmp_path / "started", tmp_path / "stopped" - with execution.invocation(execution_client) as run: - future = run.submit(_cooperate, started, stopped, 15, key="recipe") - _wait_for(started) - # Inject the partition only after startup and a confirmed worker grant. - assert monitor_ready.wait(timeout=10) - disconnected.set() - assert driver_renewed.wait(timeout=10) - with pytest.raises(execution.ExecutionCancelled, match="cancelled"): - future.result(timeout=10) - # The worker expired its own last grant while the scheduler still - # authorizes the invocation through successful driver heartbeats. - assert request(execution_client, run.id, "active") > 0 - assert failed_polls > 1 # A single failed RPC did not revoke valid authorization. - assert stopped.exists() +def test_unavailable_scheduler_is_reported_as_a_project_error() -> None: + def offline(*args: Any, **kwargs: Any) -> None: + raise OSError("connection lost") -@pytest.mark.parametrize("operation", ["revoke", "forget"]) -def test_completed_results_survive_metadata_cleanup_rpc_failure( - execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, operation: str, -) -> None: - request = execution._rpc - effects = tmp_path / "effects" - - def lose_cleanup(*args: Any, **kwargs: Any) -> Any: - if args[2] == operation: - raise TimeoutError("scheduler unavailable during metadata cleanup") - return request(*args, **kwargs) - - with execution.invocation(execution_client) as run: - result = run.submit(_effect, effects, key="recipe").result(timeout=3) - monkeypatch.setattr(execution, "_rpc", lose_cleanup) - assert result.status == "ok" - assert effects.read_text() == "executed\n" - assert run.stopped + client = SimpleNamespace(sync=offline, run_on_scheduler=None) + with pytest.raises(ProjectError, match="Dask execution guard"): + execution._rpc(client, execution._claim, "invocation", "recipe") -def test_caught_submit_failure_cannot_manufacture_completion_from_an_empty_future_list( +def test_forget_failure_does_not_hide_a_successful_result( execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - request, submit = execution._rpc, execution_client.submit - effects = tmp_path / "effects" - - def accept_then_fail(*args: Any, **kwargs: Any) -> Any: - future = submit(*args, **kwargs) - future.result(timeout=3) - raise TimeoutError("submission accepted but its handle was lost") - - def lose_revoke(*args: Any, **kwargs: Any) -> Any: - if args[2] == "revoke": - raise TimeoutError("cannot query accepted submissions") - return request(*args, **kwargs) - - monkeypatch.setattr(execution_client, "submit", accept_then_fail) - monkeypatch.setattr(execution, "_rpc", lose_revoke) - with pytest.raises(execution.ExecutionUncertain, match="confirm execution stopped"): - with execution.invocation(execution_client) as run: - with pytest.raises(TimeoutError, match="handle was lost"): - run.submit(_effect, effects, key="recipe") - assert not run.futures - assert effects.read_text() == "executed\n" - assert not run.stopped - - -@pytest.mark.parametrize("submit", [False, True]) -def test_positive_task_completion_preserves_driver_error_during_cleanup_outage( - execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, submit: bool, ) -> None: request = execution._rpc - failure = ValueError("driver could not commit") - - def lose_revoke(*args: Any, **kwargs: Any) -> Any: - if args[2] == "revoke": - raise TimeoutError("scheduler unavailable during cleanup") - return request(*args, **kwargs) - - with pytest.raises(ValueError, match="could not commit") as caught: - with execution.invocation(execution_client) as run: - if submit: - run.submit( - _effect, tmp_path / "effects", key="recipe", - ).result() - monkeypatch.setattr(execution, "_rpc", lose_revoke) - raise failure - assert caught.value is failure - assert run.stopped + def fail_forget(client: Any, function: Any, *args: Any) -> Any: + if function is execution._forget: + raise ProjectError("scheduler disconnected during cleanup") + return request(client, function, *args) -def test_confirmed_revocation_and_drain_remain_valid_when_forgetting_receipts_fails( - execution_client: Client, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - request = execution._rpc - started, stopped = tmp_path / "started", tmp_path / "stopped" - interruption = KeyboardInterrupt() - - def lose_forget(*args: Any, **kwargs: Any) -> Any: - if args[2] == "forget": - raise TimeoutError("receipt cleanup reply lost") - return request(*args, **kwargs) - - monkeypatch.setattr(execution, "_rpc", lose_forget) - with pytest.raises(execution.ExecutionInterrupted) as caught: - with execution.invocation(execution_client) as run: - run.submit(_cooperate, started, stopped, key="recipe") - _wait_for(started) - raise interruption - assert caught.value.__cause__ is interruption - assert stopped.exists() - assert run.stopped - - -def test_known_terminal_uncertainty_is_reported_without_polling_for_a_different_result( - execution_client: Client, monkeypatch: pytest.MonkeyPatch, -) -> None: - request = execution._rpc - - def refuse_polling(*args: Any, **kwargs: Any) -> Any: - if args[2] == "pending": - pytest.fail("a terminal uncertain receipt cannot become confirmed by waiting") - return request(*args, **kwargs) - - monkeypatch.setattr(execution, "_rpc", refuse_polling) - with pytest.raises(execution.ExecutionUncertain, match="unconfirmed tasks: recipe"): - with execution.invocation(execution_client) as run: - future = run.submit( - _raise, execution.ExecutionUncertain("container still unaccounted for"), - key="recipe", - ) - with pytest.raises(execution.ExecutionUncertain, match="unaccounted"): - future.result(timeout=3) - assert not run.stopped + monkeypatch.setattr(execution, "_rpc", fail_forget) + with execution.invocation(execution_client) as run: + result = run.submit(_effect, tmp_path / "effects", key="recipe").result(timeout=10) + assert result diff --git a/tests/test_execution_processes.py b/tests/test_execution_processes.py deleted file mode 100644 index 67b99089..00000000 --- a/tests/test_execution_processes.py +++ /dev/null @@ -1,378 +0,0 @@ -"""Real command ownership: timeout, interruption, background children and worker loss.""" - -from __future__ import annotations - -import os -import signal -import subprocess -import sys -import time -from pathlib import Path -from unittest.mock import Mock - -import psutil -import pytest - -from lightcone.engine.sandbox import Policy, Unavailable, run -from lightcone.engine.sandbox.model import ExecutionCancelled, ExecutionUncertain -from lightcone.engine.sandbox.processes import CIDFILE, Command - - -def test_finished_group_is_reaped_without_signalling(monkeypatch: pytest.MonkeyPatch) -> None: - from lightcone.engine.sandbox import processes - - process = Mock(spec=subprocess.Popen, pid=1234) - signal_group = Mock(side_effect=PermissionError("no live signalable processes")) - monkeypatch.setattr(processes, "members", lambda **kwargs: []) - monkeypatch.setattr(processes.os, "killpg", signal_group) - assert processes._drain(process) - signal_group.assert_not_called() - process.wait.assert_called_once_with() - - -def test_permission_error_after_last_group_member_exits_is_safe( - monkeypatch: pytest.MonkeyPatch, -) -> None: - from lightcone.engine.sandbox import processes - - process = Mock(spec=subprocess.Popen, pid=1234) - monkeypatch.setattr(processes, "members", Mock(side_effect=[[object()], []])) - monkeypatch.setattr( - processes.os, "killpg", Mock(side_effect=PermissionError("no live signalable processes")), - ) - assert processes._drain(process) - process.wait.assert_called_once_with() - - -def test_permission_error_with_a_live_group_member_remains_an_error( - monkeypatch: pytest.MonkeyPatch, -) -> None: - from lightcone.engine.sandbox import processes - - process = Mock(spec=subprocess.Popen, pid=1234) - monkeypatch.setattr(processes, "members", lambda **kwargs: [object()]) - monkeypatch.setattr( - processes.os, "killpg", Mock(side_effect=PermissionError("not permitted")), - ) - with pytest.raises(PermissionError, match="not permitted"): - processes._drain(process) - process.wait.assert_not_called() - - -def test_kill_confirmation_uses_the_remaining_cleanup_budget( - monkeypatch: pytest.MonkeyPatch, -) -> None: - from lightcone.engine.sandbox import processes - - clock = [0.0] - process = Mock(spec=subprocess.Popen, pid=1234) - signal_group = Mock() - monkeypatch.setattr(processes.time, "monotonic", lambda: clock[0]) - monkeypatch.setattr( - processes.time, "sleep", lambda seconds: clock.__setitem__(0, clock[0] + seconds), - ) - monkeypatch.setattr(processes, "members", lambda **kwargs: [object()] if clock[0] < 2.6 else []) - monkeypatch.setattr(processes.os, "killpg", signal_group) - assert processes._drain(process, deadline=10) - assert [call.args[1] for call in signal_group.call_args_list] == [ - signal.SIGTERM, signal.SIGKILL, - ] - assert 2.6 <= clock[0] < 3 - process.wait.assert_called_once_with() - - -def _policy(root: Path) -> Policy: - return Policy(read=(root,), write=(root,), execute=(), tmp_home=root) - - -def _gone(pid: int) -> bool: - try: - return psutil.Process(pid).status() == psutil.STATUS_ZOMBIE - except psutil.NoSuchProcess: - return True - - -def test_failed_configuration_handoff_reaps_custodian_without_waiting_for_pipe_eof( - tmp_path: Path, -) -> None: - # Bound the reproduction in another process: the old constructor retained - # the report pipe's writer and blocked forever while reading its child reply. - result = subprocess.run( - [sys.executable, "-c", """ -import sys -from pathlib import Path -import psutil -from lightcone.engine.sandbox.processes import Command - -try: - Command([sys.executable, '-c', "open('started', 'w').close()"], - cwd=Path(sys.argv[1]), env={'INVALID': object()}, capture=False) -except OSError: - pass -else: - raise AssertionError('unserializable configuration was accepted') -assert not psutil.Process().children(), 'custodian was not reaped' -""", str(tmp_path)], capture_output=True, text=True, timeout=5, - ) - assert result.returncode == 0, result.stderr - assert not (tmp_path / "started").exists() - - -def test_custodian_spawn_failure_closes_all_pipe_descriptors( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - from lightcone.engine.sandbox import processes - - pipe = os.pipe - descriptors: list[int] = [] - - def record_pipe() -> tuple[int, int]: - pair = pipe() - descriptors.extend(pair) - return pair - - monkeypatch.setattr(processes.os, "pipe", record_pipe) - monkeypatch.setattr(processes.subprocess, "Popen", Mock(side_effect=OSError("spawn failed"))) - with pytest.raises(OSError, match="spawn failed"): - Command([sys.executable], cwd=tmp_path, env=dict(os.environ), capture=False) - for descriptor in descriptors: - with pytest.raises(OSError): - os.fstat(descriptor) - - -def test_timeout_escalates_ignoring_command(tmp_path: Path) -> None: - pidfile = tmp_path / "pid" - outcome = run( - Unavailable(), _policy(tmp_path), - [sys.executable, "-c", ( - "import os,signal,time; from pathlib import Path; " - "signal.signal(signal.SIGTERM, signal.SIG_IGN); " - f"Path({str(pidfile)!r}).write_text(str(os.getpid())); time.sleep(60)" - )], cwd=tmp_path, env=dict(os.environ), timeout=1, - ) - assert outcome.returncode == 124 - assert any("timed out" in note for note in outcome.notes) - assert _gone(int(pidfile.read_text())) - - -def test_cancel_stops_only_its_command(tmp_path: Path) -> None: - unrelated = subprocess.Popen([sys.executable, "-c", "import time; time.sleep(60)"]) - started = time.monotonic() - try: - with pytest.raises(ExecutionCancelled, match="processes have stopped"): - run( - Unavailable(), _policy(tmp_path), - [sys.executable, "-c", "import time; time.sleep(60)"], - cwd=tmp_path, env=dict(os.environ), - cancelled=lambda: time.monotonic() - started > 0.2, - ) - assert unrelated.poll() is None - finally: - unrelated.kill() - unrelated.wait() - - -def test_interrupt_cleanup_does_not_wait_for_the_original_task_timeout( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - command = Command( - [sys.executable, "-c", "import time; time.sleep(60)"], cwd=tmp_path, - env=dict(os.environ), capture=False, timeout=60, - ) - wait = Mock(wraps=command.process.wait) - monkeypatch.setattr(command.process, "poll", Mock(side_effect=KeyboardInterrupt)) - monkeypatch.setattr(command.process, "wait", wait) - with pytest.raises(KeyboardInterrupt): - command.wait() - assert wait.call_args.kwargs["timeout"] <= 16 - - -def test_successful_leader_cannot_leave_a_background_writer(tmp_path: Path) -> None: - pidfile = tmp_path / "child" - child = ( - "import os,signal,time; from pathlib import Path; " - "signal.signal(signal.SIGTERM, signal.SIG_IGN); " - f"Path({str(pidfile)!r}).write_text(str(os.getpid())); time.sleep(60)" - ) - outcome = run( - Unavailable(), _policy(tmp_path), - [sys.executable, "-c", ( - "import subprocess,sys,time; from pathlib import Path; " - f"subprocess.Popen([sys.executable, '-c', {child!r}]); " - f"\nwhile not Path({str(pidfile)!r}).exists(): time.sleep(.01)" - )], cwd=tmp_path, env=dict(os.environ), - ) - assert outcome.returncode == 1 - assert any("background processes" in note for note in outcome.notes) - assert _gone(int(pidfile.read_text())) - - -def test_short_lived_helpers_can_finish_after_the_command(tmp_path: Path) -> None: - helper = ( - "import time; from pathlib import Path; " - "Path('ready').touch(); time.sleep(.2); Path('finished').touch()" - ) - outcome = run( - Unavailable(), _policy(tmp_path), - [sys.executable, "-c", ( - "import subprocess,sys,time; from pathlib import Path; " - f"subprocess.Popen([sys.executable, '-c', {helper!r}]); " - "\nwhile not Path('ready').exists(): time.sleep(.01)" - )], cwd=tmp_path, env=dict(os.environ), - ) - assert outcome.returncode == 0 - assert (tmp_path / "finished").exists() - assert not any("background processes" in note for note in outcome.notes) - - -def test_worker_sigkill_closes_custody_pipe_and_stops_child(tmp_path: Path) -> None: - pidfile = tmp_path / "child" - command = ( - "import os,signal,time; from pathlib import Path; " - "signal.signal(signal.SIGTERM, signal.SIG_IGN); " - f"Path({str(pidfile)!r}).write_text(str(os.getpid())); time.sleep(60)" - ) - worker = subprocess.Popen( - [sys.executable, "-c", ( - "import os,sys; from pathlib import Path; " - "from lightcone.engine.sandbox.processes import Command; " - f"c=Command([sys.executable, '-c', {command!r}], cwd=Path({str(tmp_path)!r}), " - "env=dict(os.environ), capture=False); c.wait()" - )], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, - ) - pid = None - try: - deadline = time.monotonic() + 5 - while not pidfile.exists(): - assert worker.poll() is None - assert time.monotonic() < deadline - time.sleep(0.02) - pid = int(pidfile.read_text()) - worker.kill() - worker.wait() - while not _gone(pid): - assert time.monotonic() < deadline - time.sleep(0.02) - finally: - if worker.poll() is None: - worker.kill() - worker.wait() - if pid is not None and not _gone(pid): - os.kill(pid, signal.SIGKILL) - - -def test_stdout_bytes_are_unchanged_by_custody(tmp_path: Path) -> None: - received: list[bytes] = [] - outcome = run( - Unavailable(), _policy(tmp_path), - [sys.executable, "-c", "import os; os.write(1, b'\\xff\\r\\n')"], - cwd=tmp_path, env=dict(os.environ), - output=lambda stream, value: received.append(value) if stream == "stdout" else None, - ) - assert outcome.returncode == 0 - assert b"".join(received) == b"\xff\r\n" - - -@pytest.mark.parametrize("inspection_delay", [0, 2.2]) -def test_container_timeout_uses_its_immutable_runtime_id( - tmp_path: Path, inspection_delay: float, -) -> None: - runtime = tmp_path / "runtime" - runtime.write_text(f"#!{sys.executable}\n" + ''' -import json, os, signal, subprocess, sys, time -from pathlib import Path -import psutil -root = Path(os.environ['STATE_ROOT']) -identity = 'a' * 64 -argv = sys.argv[1:] -with (root / 'calls').open('a') as log: - log.write(json.dumps(argv) + '\\n') -if argv[0] == 'run': - process = subprocess.Popen([sys.executable, '-c', - 'import signal,time; signal.signal(signal.SIGTERM, signal.SIG_IGN); time.sleep(60)'], - start_new_session=True) - (root / 'payload').write_text(str(process.pid)) - Path(argv[argv.index('--cidfile') + 1]).write_text(identity) - (root / 'cidfile').write_text(str(Path(argv[argv.index('--cidfile') + 1]).resolve())) - process.wait() -elif argv[0] == 'inspect': - time.sleep(float(os.environ['INSPECTION_DELAY'])) - try: - alive = psutil.Process(int((root / 'payload').read_text())).status() != psutil.STATUS_ZOMBIE - except psutil.NoSuchProcess: - alive = False - print('true' if alive else 'false') -elif argv[0] == 'kill': - os.kill(int((root / 'payload').read_text()), signal.SIGKILL) -elif argv[0] == 'rm': - Path((root / 'cidfile').read_text()).unlink() -''') - runtime.chmod(0o700) - command = Command( - [str(runtime), "run", "--cidfile", CIDFILE, "image"], cwd=tmp_path, - env={**os.environ, "STATE_ROOT": str(tmp_path), "INSPECTION_DELAY": str(inspection_delay)}, - capture=False, oci_runtime=str(runtime), timeout=1, - ) - try: - code, note = command.wait() - assert code == 124 - assert "timed out" in note - assert _gone(int((tmp_path / "payload").read_text())) - import json - - calls = [json.loads(line) for line in (tmp_path / "calls").read_text().splitlines()] - assert ["stop", "--time", "1", "a" * 64] in calls - assert ["kill", "a" * 64] in calls - assert ["rm", "a" * 64] in calls - finally: - path = tmp_path / "payload" - if path.exists() and not _gone(int(path.read_text())): - os.kill(int(path.read_text()), signal.SIGKILL) - - -def test_killed_custodian_reports_uncertainty(tmp_path: Path) -> None: - command = Command( - [sys.executable, "-c", "pass"], cwd=tmp_path, - env=dict(os.environ), capture=False, - ) - command.process.kill() - with pytest.raises(ExecutionUncertain, match="without confirming"): - command.wait() - - -def test_uncertain_cleanup_retains_sandbox_home(tmp_path: Path) -> None: - from lightcone.engine.sandbox import scope - - home = tmp_path / "home" - home.mkdir() - policy = Policy(read=(), write=(), execute=(), tmp_home=home) - with pytest.raises(ExecutionUncertain): - with scope(policy): - raise ExecutionUncertain("worker lost") - assert home.is_dir() - - -def test_stream_setup_failure_stops_command_before_propagating( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - from lightcone.engine.sandbox import boundary - - pidfile = tmp_path / "pid" - - def fail(_self: object) -> None: - deadline = time.monotonic() + 5 - while not pidfile.exists(): - assert time.monotonic() < deadline - time.sleep(0.01) - raise RuntimeError("cannot start stream thread") - - monkeypatch.setattr(boundary._Tail, "start", fail) - with pytest.raises(RuntimeError, match="stream thread"): - run( - Unavailable(), _policy(tmp_path), - [sys.executable, "-c", ( - "import os,time; from pathlib import Path; " - f"Path({str(pidfile)!r}).write_text(str(os.getpid())); time.sleep(60)" - )], cwd=tmp_path, env=dict(os.environ), - ) - assert _gone(int(pidfile.read_text())) diff --git a/tests/test_materialize.py b/tests/test_materialize.py index 743a45fe..c9b9a29b 100644 --- a/tests/test_materialize.py +++ b/tests/test_materialize.py @@ -440,7 +440,7 @@ def unexpected(*args: object) -> None: @pytest.mark.parametrize("failing", ["baseline/first", "baseline/second"]) -def test_a_failed_commit_restores_unconsumed_outputs_after_confirmed_stop( +def test_a_failed_commit_says_whether_remote_tasks_may_still_run( root: Path, monkeypatch: pytest.MonkeyPatch, failing: str, ) -> None: _cluster(monkeypatch, _Inline()) @@ -455,10 +455,8 @@ def fail(root: Path, paths: list[Path], message: str) -> None: monkeypatch.setattr(dataset, "save", fail) with pytest.raises(ProjectError, match="git commit failed") as raised: engine.materialize(root, [], cluster_id=CLUSTER_ID) - assert str(raised.value) == "git commit failed" - assert not dataset.status(root) - assert (root / "results/baseline/first.txt").exists() == (failing == "baseline/second") - assert not (root / "results/baseline/second.txt").exists() + # Only the first output leaves another one outstanding. + assert ("did not stop the allocation" in str(raised.value)) == (failing == "baseline/first") def test_shared_inputs_are_hashed_once_before_task_serialization( analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, @@ -473,9 +471,7 @@ def digest(path: Path) -> str: return real(path) class _Copied(_Inline): - def submit( - self, fn: Callable[..., object], *args: object, key: str, - ) -> object: + def submit(self, fn: Callable[..., object], *args: object, key: str) -> object: return fn(*pickle.loads(pickle.dumps(args))) monkeypatch.setattr(assets, "data_version", digest) @@ -733,42 +729,27 @@ def test_a_rebuild_that_fails_puts_the_previous_output_back( assert not dataset.status(root) -@pytest.mark.parametrize("failure", ["confirmed", "uncertain", "masked", "second_interrupt"]) -def test_interrupted_outputs_are_restored_only_after_confirmed_stop( - root: Path, inline: None, monkeypatch: pytest.MonkeyPatch, failure: str, +def test_an_interrupted_run_retains_what_never_reported( + root: Path, inline: None, monkeypatch: pytest.MonkeyPatch ) -> None: - """A completed sibling keeps its commit; unconfirmed writers retain their files.""" - from lightcone.engine.execution import ExecutionUncertain - - uncertain = failure != "confirmed" - + """A sibling that already saved keeps its commit; the output still in + flight may still have a writer, so its partial files are retained.""" engine.materialize(root, [], cluster_id=CLUSTER_ID) (root / "astra.yaml").write_text(_SPEC.replace("echo {decisions.method}", "echo changed")) dataset.save(root, [root], "edit both recipes") class _Interrupted(_Inline): - stopped = not uncertain - def completed(self, handles: list[Any]) -> Iterator[TaskResult]: yield handles[0] - if failure == "masked": - raise RuntimeError("client close masked uncertain cleanup") - if failure == "uncertain": - raise ExecutionUncertain("worker disappeared without acknowledging stop") raise KeyboardInterrupt _cluster(monkeypatch, _Interrupted()) - expected = {"masked": RuntimeError, "uncertain": ExecutionUncertain}.get( - failure, KeyboardInterrupt, - ) - with pytest.raises(expected): + with pytest.raises(KeyboardInterrupt): engine.materialize(root, [], cluster_id=CLUSTER_ID) - assert bool(dataset.status(root)) is uncertain - assert (root / "results/baseline/second.txt").read_text() == ( - "changed\n" if uncertain else "alpha\n" - ) + assert dataset.status(root) + assert (root / "results/baseline/second.txt").read_text() == "changed\n" # ---- the commit message ---------------------------------------------------- @@ -1103,7 +1084,7 @@ def test_a_processes_cluster_fits_through_the_seam( @contextmanager def processes(cluster_id: str) -> Iterator[Any]: with LocalCluster( - n_workers=2, threads_per_worker=1, processes=True, dashboard_address=None, + n_workers=2, threads_per_worker=1, processes=True, dashboard_address=None ) as cluster: with Client(cluster, set_as_default=False) as client: yield client diff --git a/tests/test_run.py b/tests/test_run.py index 03a0cabe..8412faa5 100644 --- a/tests/test_run.py +++ b/tests/test_run.py @@ -309,8 +309,7 @@ def test_a_remote_task_exception_is_an_engine_error_and_leaves_compute_available monkeypatch.setattr(container, "backend", lambda _: Unavailable()) with pytest.raises(ProjectError, match="cluster execution failed") as raised: engine_run.probe(project, ["true"], cluster_id=cluster_id) - assert "may still be running" not in str(raised.value) - assert "Stop the allocation" not in str(raised.value) + assert "did not stop the allocation" in str(raised.value) with compute.connect(cluster_id) as client: assert client.scheduler_info()["workers"] diff --git a/tests/test_sandbox_oci.py b/tests/test_sandbox_oci.py index b4074ff5..cf8cdccd 100644 --- a/tests/test_sandbox_oci.py +++ b/tests/test_sandbox_oci.py @@ -7,6 +7,7 @@ from __future__ import annotations +import subprocess from pathlib import Path from typing import Any @@ -220,37 +221,26 @@ def test_the_attestation_is_derived_from_the_flags(root: Path, policy: Policy) - class _Recorder: - """A custodian stand-in recording the sandbox's fully wrapped argv.""" + """A Popen stand-in that records the argv and exits as told.""" def __init__(self, returncode: int = 0) -> None: self.argv: list[str] | None = None - self.options: dict[str, Any] = {} self.returncode = returncode def __call__(self, argv: list[str], **kwargs: Any) -> Any: self.argv = list(argv) - self.options = kwargs code = self.returncode class _Proc: import io - stderr = io.BytesIO(b"") + stderr = io.StringIO("") returncode = code - class _Command: - process = _Proc() + def wait(self) -> int: + return code - def __enter__(self) -> _Command: - return self - - def __exit__(self, *args: Any) -> None: - pass - - def wait(self, cancelled: Any) -> tuple[int, str]: - return code, "" - - return _Command() + return _Proc() def test_a_world_backend_takes_the_prefix_inside( @@ -260,7 +250,7 @@ def test_a_world_backend_takes_the_prefix_inside( is part of the world being entered, so it lands after the image in the argv rather than in front of the runtime.""" recorder = _Recorder() - monkeypatch.setattr(boundary, "Command", recorder) + monkeypatch.setattr(subprocess, "Popen", recorder) boundary.run( _backend(root), @@ -273,8 +263,6 @@ def test_a_world_backend_takes_the_prefix_inside( assert recorder.argv is not None assert recorder.argv[0] == "podman" - assert recorder.options["oci_runtime"] == "podman" - assert "--cidfile" in recorder.argv assert recorder.argv[recorder.argv.index(_IMAGE_ID) + 1 :] == [ "uv", "run", "--locked", "--no-sync", "--project", str(root), "--", "bash", "-c", "true", @@ -287,7 +275,7 @@ def test_a_host_backend_keeps_the_prefix_outside( """The existing composition, pinned: uv's config and caches are trusted plumbing outside a host mechanism's rewrite.""" recorder = _Recorder() - monkeypatch.setattr(boundary, "Command", recorder) + monkeypatch.setattr(subprocess, "Popen", recorder) boundary.run( Unavailable(), @@ -300,20 +288,6 @@ def test_a_host_backend_keeps_the_prefix_outside( assert recorder.argv is not None assert recorder.argv[:3] == ["uv", "run", "--"] - assert recorder.options["oci_runtime"] is None - - -def test_a_world_backend_does_not_implicitly_get_oci_lifecycle( - root: Path, policy: Policy, monkeypatch: pytest.MonkeyPatch, -) -> None: - recorder = _Recorder(returncode=125) - monkeypatch.setattr(boundary, "Command", recorder) - outcome = boundary.run( - Unavailable(contains_prefix=True), policy, ["true"], cwd=root, env={}, - ) - assert recorder.options["oci_runtime"] is None - assert "--cidfile" not in recorder.argv - assert not any("runtime failed before the command ran" in note for note in outcome.notes) def test_exit_97_is_the_shims_only_under_landlock( @@ -322,7 +296,7 @@ def test_exit_97_is_the_shims_only_under_landlock( """97 is the shim's reserved code, and there is no shim in a container — a recipe legitimately exiting 97 must not be told lc could not set up the sandbox.""" - monkeypatch.setattr(boundary, "Command", _Recorder(returncode=97)) + monkeypatch.setattr(subprocess, "Popen", _Recorder(returncode=97)) outcome = boundary.run(_backend(root), policy, ["true"], cwd=root, env={}) assert not any("could not set up" in note for note in outcome.notes) @@ -333,7 +307,7 @@ def test_exit_125_names_the_runtime_not_the_command( """The runtimes reserve 125 for their own failures — the command never ran, so neither the denial heuristics nor the trailer should point at it.""" - monkeypatch.setattr(boundary, "Command", _Recorder(returncode=125)) + monkeypatch.setattr(subprocess, "Popen", _Recorder(returncode=125)) outcome = boundary.run(_backend(root), policy, ["true"], cwd=root, env={}) assert any("runtime failed before the command ran" in note for note in outcome.notes) assert not any("ran under the lc sandbox" in note for note in outcome.notes)