From aa3254984716b4f3c2ce98268f1930ea9b22a4a9 Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Tue, 29 Sep 2026 03:56:29 -0700 Subject: [PATCH 1/3] Honor recipe resources and allocate typed NVIDIA GPUs Signed-off-by: Francois Lanusse --- CLAUDE.md | 46 +++++ docs/api/compute.md | 76 ++++++- docs/api/index.md | 2 + docs/api/materialize.md | 5 +- docs/api/plan.md | 18 +- docs/api/sandbox.md | 14 +- docs/api/worker.md | 18 +- docs/architecture.md | 29 ++- docs/cli/compute.md | 28 ++- docs/cli/materialize.md | 7 + docs/cli/run.md | 10 +- docs/user/cluster.md | 139 ++++++++++++- src/lightcone/cli/compute.py | 24 ++- src/lightcone/engine/compute/__init__.py | 11 +- src/lightcone/engine/compute/catalog.py | 29 ++- src/lightcone/engine/compute/local.py | 37 +++- src/lightcone/engine/compute/local_runtime.py | 11 + src/lightcone/engine/compute/model.py | 109 ++++++++-- src/lightcone/engine/compute/slurm.py | 68 +++++- .../engine/compute/slurm_bootstrap.py | 16 +- src/lightcone/engine/container.py | 7 +- src/lightcone/engine/execution_resources.py | 162 +++++++++++++++ src/lightcone/engine/gpu.py | 141 +++++++++++++ src/lightcone/engine/materialize.py | 29 ++- src/lightcone/engine/plan.py | 7 +- src/lightcone/engine/run.py | 18 +- src/lightcone/engine/sandbox/oci.py | 17 +- src/lightcone/engine/sandbox/policy.py | 12 +- src/lightcone/engine/units.py | 45 ++++ src/lightcone/engine/worker.py | 12 +- tests/conftest.py | 14 +- tests/test_compute.py | 192 ++++++++++++++++- tests/test_compute_local.py | 153 ++++++++++++++ tests/test_compute_slurm.py | 160 +++++++++++++- tests/test_execution_resources.py | 195 ++++++++++++++++++ tests/test_gpu.py | 144 +++++++++++++ tests/test_gpu_execution.py | 106 ++++++++++ tests/test_materialize.py | 144 ++++++++++++- tests/test_plan.py | 35 +++- tests/test_sandbox_oci.py | 36 ++++ tests/test_sandbox_policy.py | 23 +++ 41 files changed, 2244 insertions(+), 105 deletions(-) create mode 100644 src/lightcone/engine/execution_resources.py create mode 100644 src/lightcone/engine/gpu.py create mode 100644 src/lightcone/engine/units.py create mode 100644 tests/test_execution_resources.py create mode 100644 tests/test_gpu.py create mode 100644 tests/test_gpu_execution.py diff --git a/CLAUDE.md b/CLAUDE.md index 961880e6..e8927406 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1727,6 +1727,10 @@ and use it for every native ownership check and filter. **Local compute needs no setup.** An absent implicit `~/.lightcone/compute.yaml` selects a built-in local catalog: one CPU, 1 GiB, one node, fast startup, 30-minute default and two-hour maximum lifetime. It writes no catalog and starts no cluster. +Linux CUDA discovery adds GPU offers grouped by native model (`local-gpu`, or +`local-gpu-1`, etc. for mixed models), retaining those CPU/RAM defaults. Discovery +failure must not disable the CPU offer. Explicit local allocations remain +cooperative; they do not reserve a device against other host programs/allocations. Configured catalogs replace it completely; missing explicit paths and invalid files are errors. Execution still requires an explicitly launched cluster's name or ID. @@ -1753,6 +1757,17 @@ Connection names live only in the catalog's mapping keys. Use validated `replace for updates and the explicit `as_dict` allowlists for public output. Preserve duplicate-key rejection in YAML; providers validate their own `launch` and `config`. +**Allocation syntax follows SkyPilot without depending on SkyPilot.** Compute +CPU/memory requests support exact quantities or `+` minimums. Compute memory +uses binary units: bare `32`, `32GB`, and `32GiB` agree. Catalog resources use +one `accelerators: NAME[:COUNT]` or a one-entry mapping; CLI `--gpus A100:4`, +`A100`, or generic `GPU:4` selects an exact positive whole count, while `0` +means CPU only. Type matching is case-insensitive; no GPU `+`, fractions, or +global model alias registry. Local discovery reports native names with whitespace +and punctuation normalized to hyphens (e.g. `NVIDIA-A100-SXM4-80GB`). Named Slurm offers must map +their public label to the site's GRES type through `config.gpu_type`; generic +`GPU` offers may omit that setting. Preserve native evidence in observations. + **Configured compute roots may be filesystem aliases.** Resolve connection and scratch roots before appending managed namespace, submission, or attempt paths. Keep symlink rejection within those managed paths and enforce private directory @@ -1779,6 +1794,37 @@ Any driver failure while tasks are outstanding (a failed commit included, not only a cluster error) carries `compute.UNSTOPPED`, the one wording for "the allocation was not stopped and unreported tasks may still be running". +**Recipe resources use standard Dask admission.** Preserve ASTRA `recipe.resources` +in `plan.Task` as raw mappings so `status` and `--check` remain independent of +executor support. Parse `TaskResources` at execution admission: whole CPUs, memory +bytes, and whole GPU counts. Validate the whole selected graph before preparation +or submission, then pass reservations explicitly to submission. Recipe `time_limit` +is unsupported and must fail explicitly; allocation walltime remains supported. +Workers advertise CPU/MEMORY/GPU; tasks reserve their declarations, with omitted RAM +reserving a whole worker's memory and probes reserving all whole-worker budgets. +GPU recipes reserve the worker's full GPU budget, one GPU recipe at a time, while +their commands see only their requested count. Recipe `gpus` defaults to zero and +does not select a model. Recipe memory retains ASTRA units (`8Gi` binary, +`8GB` decimal, no bare quantities), independently of compute's SkyPilot units. +Thread slots remain a separate concurrency cap. Reservations are cooperative, not +per-command OS CPU/RAM limits; unsupported disk/model requests and fractional +CPU/GPU counts fail explicitly. + +**GPU discovery and visibility stay native and per-command.** `engine.gpu` probes +the CUDA Driver API in an isolated stdlib process, respecting native masks and +returning UUIDs/model names without initializing CUDA in a reusable worker. No +new Python GPU dependency or custom Dask worker. Whole NVIDIA GPUs on Linux only; +MIG/fractional GPUs and other vendors are unsupported. Verify requested devices +before deleting outputs. Set the command policy's CUDA mask; never mutate the +shared worker environment. CPU commands get an empty mask, and probes expose only +their reserved count, even if the native mask is larger. Direct policies grant +known NVIDIA character nodes; native permissions/cgroups still govern access. +OCI uses Docker's explicit UUID `--gpus` (CSV quoting for multiple devices), Podman +CDI UUID device names, or podman-hpc `--gpu` plus the CUDA mask. CPU containers +override `NVIDIA_VISIBLE_DEVICES=void` to defeat image defaults. The CUDA mask is +cooperative visibility, not a security claim. Physical GPU execution remains +unvalidated; tests simulate inventory and check real subprocess masks/native argv. + **One catalog selector, `LC_COMPUTE_CONFIG` (2026-09).** `lc compute --config` was removed: `run` and `materialize` resolve clusters through the catalog too, and a per-invocation override on one command group launched allocations those diff --git a/docs/api/compute.md b/docs/api/compute.md index c6fa9429..7c97b833 100644 --- a/docs/api/compute.md +++ b/docs/api/compute.md @@ -6,7 +6,7 @@ It owns no service, registry, or saved current-cluster selection. | Symbol | Contract | |---|---| -| `Request.parse(...)` | Common exact/minimum CPU and memory requests, node count, walltime, startup class. | +| `Request.parse(...)` | Exact/minimum CPU and memory requests, exact accelerator type/count, node count, walltime, startup class. | | `Catalog.load(path)` | Ordered fixed shapes and stable connection namespaces; use the built-in local catalog only when the implicit default file is absent. | | `Compute.plan(request, *, name=None)` | Select an eligible offer and freeze its native launch settings and optional name without allocation. | | `Compute.launch(plan)` | Check names across native authorities, generate one if omitted, submit once, and return a self-contained `Identity`. | @@ -17,14 +17,16 @@ It owns no service, registry, or saved current-cluster selection. | `Provider` | `plan`, `launch`, `discover`, `inspect`, `connect`, `terminate`. | `Catalog.load()` defaults to `~/.lightcone/compute.yaml`. When that implicit file -is absent, the built-in catalog exposes one `local` offer: one CPU, 1 GiB, one node, -fast startup, 30-minute default and two-hour maximum lifetime. It creates no -configuration file or allocation. Configured catalogs replace it completely. +is absent, the built-in catalog exposes a `local` offer: one CPU, 1 GiB, one node, +fast startup, 30-minute default and two-hour maximum lifetime. Successful Linux +CUDA discovery adds one GPU offer per native model: `local-gpu` for one model, +or `local-gpu-1`, `local-gpu-2`, etc. GPU discovery failure preserves the CPU offer. +It creates no configuration file or allocation. Configured catalogs replace it completely. Missing paths selected through an argument or `LC_COMPUTE_CONFIG`, unreadable files, and invalid catalogs remain errors. Stable connection namespaces let separate invocations discover and attach to the same local allocations. -`model.py` defines the shared Pydantic models: `Connection`, `Offer`, `Resources`, +`model.py` defines the shared Pydantic models: `Connection`, `Offer`, `Resources`, `Accelerator`, `TimeLimits`, `Startup`, `Request`, `Identity`, `LaunchPlan`, and `Snapshot`. `Catalog` validates YAML directly into these objects, which providers also use. Unknown common fields are rejected; schema errors identify paths such as @@ -39,6 +41,14 @@ keeps the configured `default` and `max` duration strings and exposes `default_seconds` and `max_seconds`. `Startup.class_` corresponds to YAML `class`. Connection names exist only as catalog mapping keys, referenced by `Offer.connection`. +Compute memory accepts bare GiB quantities and SkyPilot-style binary units: +`32`, `32GB`, and `32GiB` agree. CPU and memory requests accept a trailing `+`. +`Accelerator` accepts one `NAME[:COUNT]` or one-entry mapping, such as `A100:4` +or `{A100: 4}`, and serializes to that mapping. Counts are exact positive integers; +type matching is case-insensitive, and the generic name `GPU` accepts any model. +No accelerator registry or model alias expansion is maintained. Slurm's +`config.gpu_type` maps a named catalog accelerator to its native GRES type. + Model constructors take keyword arguments. `replace(...)` validates updates; `model_dump()` and `model_validate()` support internal roundtrips without changing units. Keep the explicit `as_dict()` methods for public CLI output so internal @@ -113,6 +123,62 @@ Dask chooses the workers and handles dependencies; invocation-specific keys prev unintended reuse across commands. There is no worker-selection layer, per-worker preflight orchestration, source fingerprinting, or login-node guard. Driver-side preparation and the existing task runtime/sandbox checks remain in their owners. + +Workers advertise standard Dask `CPU`, `MEMORY`, and `GPU` resources; memory is measured +in bytes. `engine.execution_resources.TaskResources` validates ASTRA's +`recipe.resources` into whole CPUs, bytes, and a whole GPU count at +execution admission. `plan.Task` preserves the ASTRA mapping so read-only +classification does not impose executor restrictions. `requirements(workers)` +checks that one worker can satisfy it and returns the resource dictionary used +by `Client.submit`. +An omitted memory request reserves the full homogeneous worker budget; +`whole_worker=True` reserves CPU, memory, and GPUs for a probe. Recipe GPU counts +default to zero; a GPU recipe reserves the full GPU budget of a fitting worker, +while exposing only the requested device count. This serializes GPU recipes per +worker without a device-assignment service. Unsupported disk/type requests and +fractional CPU/GPU counts fail before execution. + +The materialize scheduler validates every selected task before preparation or +submission, preventing earlier tasks from starting before a later impossible +request is discovered, then passes each task's reservation explicitly to +submission. Allocation and task requests share byte conversion utilities; their +models remain distinct because allocation selection supports minimum quantities +and node counts. Standard Dask scheduling accounts for +concurrent CPU, memory, and GPU reservations; Dask execution-thread counts remain a +separate concurrency cap. Reservations do not impose hard limits on recipe +subprocesses. Recipe `time_limit` is unsupported and explicitly refused before +preparation or execution; allocation walltime remains supported. + +Recipe memory remains ASTRA-style: `8Gi` is binary, `8GB` is decimal, and units +are required. Allocation memory follows the compute convention above; keep the +two parsers' contracts explicit even though they share exact byte arithmetic. + +## GPU discovery and visibility + +`engine.gpu.inventory()` uses a short isolated stdlib process to query the CUDA +Driver API for native UUIDs and model names, respecting `CUDA_VISIBLE_DEVICES`. +`visible_devices()` returns those UUIDs; `device_paths()` lists NVIDIA character +device nodes for direct sandbox grants. CUDA is never initialized in the reusable +worker, and no additional GPU Python dependency or custom Dask worker is needed. +Missing drivers produce an empty inventory. Discovery rejects invalid identities, +driver failures, and partitioned MIG devices rather than inventing capacity. + +Local launch selects devices matching the offer, freezes their UUID mask in the +allocation environment, and checks the visible count before starting Dask. Slurm +validates native GPU capacity and CUDA visibility before advertising the offer's +GPU budget. Workers check recipe visibility before resetting output files. +Slurm bootstrap sets `CUDA_DEVICE_ORDER=PCI_BUS_ID` before discovery so CUDA +interprets native numeric masks in Slurm/NVML's device order. +Probes receive the reserved GPU count explicitly and never expose excess native +devices. CPU commands receive an empty CUDA mask without GPU discovery. + +The sandbox applies masks per command, preserving the worker's shared environment. +OCI backends request native device injection by UUID; direct policies grant known +NVIDIA device nodes. Visibility is cooperative, with native permissions and cgroups +still authoritative. See [GPU deployment requirements](../user/cluster.md#gpu-allocations). + +## Execution output and teardown + `output.py` transports byte chunks through standard Dask events so detached workers' output reaches the invoking CLI. It uses the borrowed client's event topic, which the schedulers lc launches drop as soon as the client disconnects diff --git a/docs/api/index.md b/docs/api/index.md index 8674ea0b..a202a896 100644 --- a/docs/api/index.md +++ b/docs/api/index.md @@ -18,6 +18,8 @@ is responsibility and contract, not every signature. | [`worker`](worker.md) | Making one output; the rerun entry point | impure | | [`materialize`](materialize.md) | The driver: gates, scheduling, the save/restore loop, status | impure | | [`compute`](compute.md) | Resource requests, native allocation lifecycle, borrowed Dask clients | impure | +| [`execution_resources`](compute.md) | Task resource admission | pure | +| [`gpu`](compute.md#gpu-discovery-and-visibility) | Native CUDA inventory and NVIDIA device paths | impure | | [`sandbox`](sandbox.md) | The exec boundary: policy, backends, attestation, denials | mixed | | [`image` & `container`](container.md) | The container hatch: declaration → image → archive → runtime | pure / impure | | [`crate`](crate.md) | The publication view: the repo as an RO-Crate | pure | diff --git a/docs/api/materialize.md b/docs/api/materialize.md index 39450bcb..a7419172 100644 --- a/docs/api/materialize.md +++ b/docs/api/materialize.md @@ -18,7 +18,7 @@ driver's stderr, independently of success or failure, leaving stdout for the rep | `check(root, targets, *, refresh)` | The same classification without executing, committing, or fetching. Exempt from the dirty refusal. | | `status(root)` | The report: every output's state and provenance commit, plus the mode/image/sandbox header facts. | | `MaterializeReport` / `StatusReport` | The JSON surfaces; `ok` and `up_to_date` first. | -| `cluster_for_run(cluster_id)` | Borrow the cluster; the submit/completed scheduler seam (`submit`, `completed`). | +| `cluster_for_run(cluster_id)` | Borrow the cluster; expose resource validation, submission, and completion. | | `run_record(...)` / `datalad_run_subject(...)` | The commit message `datalad rerun` replays, and the one spelling of its subject line — shared with the foreign-write comparator, because two strings here would drift. | | `_engine_requirement()` | How a record pins its engine: by version for a release, by source commit (hatch-vcs) for a dev build. | @@ -27,7 +27,8 @@ driver's stderr, independently of success or failure, leaving stdout for the rep 1. **Read-only project checks before connecting** — tool, committer, dirty-tree, spec and lock errors do not require a reachable cluster to report. 2. **Explicit cluster before preparing the environment** — validate native - allocation identity and connect before fetching inputs or building an image. + allocation identity, connect, and validate every selected task's CPU/memory/GPU + request before fetching inputs or building an image. The dirty refusal has already run: in containerized mode the converge can commit an image archive, and `dataset.save` commits the whole index; on a dirty tree the user's diff --git a/docs/api/plan.md b/docs/api/plan.md index 93cf8e35..abc1641e 100644 --- a/docs/api/plan.md +++ b/docs/api/plan.md @@ -3,8 +3,8 @@ The spec, read as a graph of tasks. `astra.yaml` × `universes/*.yaml` gives one task per `(universe, output)` pair that has a recipe; a task carries everything executing it needs — the rendered command, where its -bytes go, what it reads, its decisions, its `definition_version` — and -nothing about *how* it will be executed. +bytes go, what it reads, its decisions, its `definition_version`, and its +resource requirements — and nothing about *how* it will be executed. Source: `src/lightcone/engine/plan.py`. @@ -14,7 +14,7 @@ Source: `src/lightcone/engine/plan.py`. |---|---| | `build(root)` | Validate the spec with ASTRA's own validators, resolve every universe, return the `Graph`. | | `Graph` | Tasks keyed on `(universe_id, output_id)`; `order()` for the read-only topological walk, `resolve(targets)` for what a user typed, `closure(keys)` to narrow a run. | -| `Task` | One output in one universe, frozen. | +| `Task` | One output in one universe, frozen, retaining ASTRA's resource declaration in `resources`. | | `declared_path(root, path)` | The one rule that names a path: project-relative inside the tree, absolute outside, never resolved. | ## What must stay true @@ -31,6 +31,14 @@ Source: `src/lightcone/engine/plan.py`. schema, file, and universe validators before resolving anything — resolution answers what a *valid* spec means and does not re-check that it is one. +- **Resource declarations survive resolution.** `build` reads + `recipe.resources` from ASTRA's resolved output definition and preserves the + mapping. A valid declaration remains readable by `status` and + `materialize --check` even when this executor cannot honor it. Execution + validates supported requirements through `TaskResources.parse` and checks + cluster capacity before materialize prepares the project or submits any + task. No worker placement or executor-specific resource validation belongs + in this module. - **The layout is flat and path-addressed.** `results//.`, and the path in a rendered recipe *is* the path on disk — no staging, no relocation. @@ -50,8 +58,8 @@ Source: `src/lightcone/engine/plan.py`. ## Tests `tests/test_plan.py` — pure; tests what lc *adds* (directories, edges, -versions, the validation gate), never what a spec means — that -coverage lives in astra-tools' own suite, and re-asserting it here +versions, resource preservation, the validation gate), never what a spec means — +that coverage lives in astra-tools' own suite, and re-asserting it here would recreate the second implementation this module deleted. Every fixture must be a spec `astra validate` accepts; the gate enforces it for free. diff --git a/docs/api/sandbox.md b/docs/api/sandbox.md index afd87882..3911b242 100644 --- a/docs/api/sandbox.md +++ b/docs/api/sandbox.md @@ -18,7 +18,7 @@ plus `lightcone/_sandbox_exec.py`, the Landlock shim. | `Capability` | What this host can do — `detect()`'s answer, the only `sys.platform` branch. | | `Attestation` | What was actually enforced, derived from the flags applied — never from what the matrix says should have happened. | | `Backend.wrap(policy, argv)` | The pure rewrite. `contains_prefix` declares whether the uv hop rides inside (a container is a world; a host mechanism trusts host plumbing). | -| `exec_policy(...)` | The one policy: probe and recipe get the same thing. Building it is where the impurity lives (the per-run private `$HOME`); `scope()` owns its cleanup. | +| `exec_policy(...)` | Shared policy builder with caller-supplied write scope and GPU UUIDs. Building it creates a private `$HOME`; `scope()` owns its cleanup. | | `Unavailable` | A real backend that wraps to the same argv and attests `fs: open`. Saying so is the caller's job; pretending is nobody's. | | `denial.explain()` / `denial.trailer()` | Best-guess remedies (allowed to return nothing) and the unconditional trailer on every nonzero sandboxed exit. | @@ -26,6 +26,18 @@ An optional output receiver gets stdout/stderr byte chunks. Capturing output nev decodes or normalizes stdout; only the retained stderr tail is decoded for denial classification. Without a receiver, stdout remains inherited. +`exec_policy(..., gpu_devices=...)` carries allocated CUDA UUIDs into the command's +`CUDA_VISIBLE_DEVICES`, with an empty mask for CPU-only execution. Direct policies +grant existing NVIDIA character device nodes only for GPU commands. Native +permissions still apply; the CUDA mask controls cooperative visibility, not +hostile-code device isolation. No worker-wide environment mutation is involved. + +The pure OCI rewrite requests UUIDs through Docker's `--gpus device=...`, Podman's +`--device=nvidia.com/gpu=UUID`, or podman-hpc's `--gpu` plus the CUDA mask. CPU +containers also set `NVIDIA_VISIBLE_DEVICES=void`, overriding GPU-enabled image +defaults. Runtime prerequisites are documented under +[GPU allocations](../user/cluster.md#gpu-allocations). + ## What must stay true - **`wrap` stays pure** — no temp files, no FDs, no global state diff --git a/docs/api/worker.md b/docs/api/worker.md index 3d6e255d..acd2e628 100644 --- a/docs/api/worker.md +++ b/docs/api/worker.md @@ -18,14 +18,19 @@ Source: `src/lightcone/engine/worker.py`. Cluster execution supplies an output receiver to `materialize`/`execute`, which passes byte chunks from the sandbox back to the invocation. Standalone reruns -retain direct terminal output. +retain direct terminal output. The driver submits each cluster task with its +CPU, memory, and GPU reservations. Before resetting outputs, `execute` validates +resource syntax, checks native GPU visibility, and selects the requested number +of UUIDs. Standalone reruns apply the same checks but do not perform Dask resource +admission. Recipe `time_limit` is explicitly refused. ## Key symbols | Symbol | Role | |---|---| -| `materialize(task, versions, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns a `TaskResult`, always. | -| `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, and the attestation. `.usable` is what dependents check. | +| `materialize(root, task, context, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns a `TaskResult`, always. | +| `execute(root, task, input_versions, context)` | Validate resource syntax and device visibility, run a recipe unconditionally, then record its payload and manifest. | +| `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, reason, and diagnostic notes. `.usable` is what dependents check. | | `main(argv)` | The rerun entry point: guards, converges the project environment from the commit's own lock, resolves its own HEAD and runtime, executes. | | `lc_version()` | The engine version every manifest records. | @@ -35,6 +40,11 @@ retain direct terminal output. contract holds for failure modes nobody enumerated. Raising would make Dask abort every task in flight; reporting all independent failures in one run is most of what owning the loop buys. +- **Device visibility belongs to each command.** CPU recipes receive an empty + `CUDA_VISIBLE_DEVICES`; GPU recipes receive only their selected UUIDs through + the sandbox policy. Never modify the reusable worker's shared environment. + Admission reserves the worker's full GPU budget for one GPU recipe at a time, + even when its requested visible count is smaller. - **`data_version` is computed here, before anything is staged** — the dependent's argument *is* this return value, so the digest must exist while the files are still unannexed. Deriving it from @@ -67,3 +77,5 @@ retain direct terminal output. `tests/test_worker.py` — real recipes through the real boundary against a real repository (the `analysis` fixture): whether gates hold and bytes land are not questions a stub can answer. +`tests/test_gpu_execution.py` checks real subprocess masks and refusal before +output deletion using a simulated GPU inventory; it requires no physical GPU. diff --git a/docs/architecture.md b/docs/architecture.md index 5e55fb62..9e02f404 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -41,6 +41,7 @@ lc materialize "$CLUSTER" │ plan: astra validate + resolve → Graph of Tasks │ (no tasks → converge the crate and stop; nothing connects) │ connect: native identity + Dask readiness + │ admit: each task's CPU, memory, and GPU request fits a worker │ fetch: git annex get (declared inputs not in this clone) │ converge: uv.lock ⇄ .venv (and the image, containerized) ├─► workers: reset output file → sandbox → recipe → hash → manifest @@ -64,6 +65,13 @@ The division of labor is strict and load-bearing: unreadable manifest — all come back as a state, so one failure doesn't abort every task in flight, and a run reports *all* its independent failures. +- **Dask accounts for task resources.** Workers advertise CPU, memory, and GPU + budgets; submissions reserve the recipe's requirements. CPU and memory + reservations coordinate scheduling rather than imposing per-recipe OS limits. + Recipe `time_limit` is explicitly refused; allocation walltime remains supported. + A GPU recipe reserves the worker's full GPU budget and exposes only its + requested devices through a per-command CUDA mask. CPU recipes expose none; + probes reserve and expose the whole worker budget. - **Values are resolved once and handed down.** HEAD, the container runtime, and the foreign-write facts are read by the driver and passed to workers as values — a worker that asked git itself could @@ -135,9 +143,9 @@ Because every backend is a pure argv rewrite, all of them are testable on a host that can't run them, and the manifest's `hermeticity` field records what was *actually* enforced — never what should have been. -There is one policy, `exec_policy`: probe and recipe get exactly the -same thing (tree read-only apart from `results/`), so "works under -`lc run`" and "works as a recipe" stay the same fact. +There is one policy builder, `exec_policy`: probes and recipes share environment +and filesystem rules, with write scope and GPU visibility supplied by the caller. +Probes expose their reserved worker's GPUs; recipes expose their declared count. ## The container hatch @@ -157,17 +165,26 @@ config-blob id, never a tag. `engine.compute` owns allocation lifecycle through a small provider protocol. A YAML catalog supplies ordered resource offers and stable native service namespaces. When the implicit default file is absent, a built-in local catalog -provides one CPU and 1 GiB without setup. An explicit catalog replaces that default; +provides one CPU and 1 GiB without setup, plus GPU offers grouped by model when +Linux CUDA discovery succeeds. An explicit catalog replaces those defaults; missing explicit paths and invalid files remain errors. No catalog is written and no allocation starts until `compute launch` resolves resources and submits once. Slurm queries and validated local OS identities are authoritative for allocations; Dask is the authority for connected workers. Private scheduler/TLS files are connection material, not a registry. +Allocation requests use SkyPilot-style CPU/memory exact or minimum quantities +and one accelerator type/count. Providers translate those requests into native +allocations; Lightcone does not depend on SkyPilot or carry its GPU alias registry. +An isolated stdlib CUDA probe discovers native device UUIDs and model names; +stock Dask workers remain unchanged. GPU container access uses each runtime's +native mechanism. See [GPU setup](user/cluster.md#gpu-allocations). + `compute.connect(CLUSTER_ID)` borrows a standard Dask client and closes only that client on exit. Both execution commands require a cluster ID. The materialization -scheduler keeps its `submit`/`completed` seam. Driver preparation and existing -task runtime/sandbox checks remain unchanged. Tasks use ordinary Dask scheduling; +scheduler validates resource requests, then keeps its `submit`/`completed` seam. +Driver preparation and existing task runtime/sandbox checks remain unchanged. +Tasks use ordinary Dask scheduling; there is no separate worker-selection or preflight layer, or site-marker guard. No execution command implicitly allocates compute. See [compute internals](api/compute.md) and [deployment limits](user/cluster.md). diff --git a/docs/cli/compute.md b/docs/cli/compute.md index e4d697eb..23e05b81 100644 --- a/docs/cli/compute.md +++ b/docs/cli/compute.md @@ -5,7 +5,7 @@ No project is required for these commands. ```text lc compute resources [--json] -lc compute launch --cpus VALUE --memory VALUE +lc compute launch --cpus VALUE --memory VALUE [--gpus NAME[:COUNT]|0] [--name NAME] [--num-nodes N] [--time DURATION] [--startup fast] [--dry-run] [--json] lc compute status [CLUSTER] [--wait] [--timeout SECONDS] [--json] lc compute down CLUSTER [--json] @@ -15,6 +15,10 @@ Without configuration, `resources` exposes a built-in `local` offer: one CPU, 1 GiB, one node, fast startup, and a 30-minute default lifetime (two-hour maximum). Launch it with `lc compute launch --cpus 1 --memory 1`; execution still requires the returned cluster name or its full immutable ID. +On Linux, visible NVIDIA GPUs also produce local GPU offers, grouped by model. +Use the accelerator names and counts shown by `resources`, or `GPU:N` to request +any model with exactly N GPUs per node. GPU discovery failure leaves the CPU +offer available. `~/.lightcone/compute.yaml`, when present, replaces this built-in catalog. `LC_COMPUTE_CONFIG` selects another file for both compute and execution commands, @@ -53,9 +57,21 @@ durable reference to that allocation. Use the full immutable `id` from launch or status JSON to address one allocation directly, including when unrelated connections are unavailable. No name registry is maintained. -CPU quantities are logical CPUs **per node**, memory is **GiB per node**, and -`--num-nodes` defaults to one. Bare quantities are exact; `4+` means at least four. -Time accepts positive whole minutes or hours, such as `30m` or `2h`. Without +Resource quantities are **per node**, and `--num-nodes` defaults to one. CPU and +memory requests follow SkyPilot's exact/minimum convention: `4` is exact and `4+` +means at least four. Compute memory uses binary units: `16`, `16GB`, and `16GiB` +all mean 16 GiB; `16GB+` permits a larger offer. + +`--gpus A100:4` requests exactly four GPUs from an offer whose accelerator type is `A100`; +`--gpus A100` means one, `--gpus GPU:4` accepts any GPU model, and the default +`--gpus 0` selects CPU-only offers. Names match case-insensitively. GPU counts +are positive whole numbers, with no `+` or fractional form. Lightcone does not +maintain SkyPilot's accelerator alias registry: copy local model names from +`resources` or use `GPU:N`. Catalog shapes use `accelerators: A100:4` or +`accelerators: {A100: 4}`. See [GPU allocations](../user/cluster.md#gpu-allocations). + +Time accepts positive durations with day/hour/minute/second units, such as `30m`, +`1h30m`, or `45s`. Without `--time`, the chosen offer's default applies. `fast` is a service class, not a queue-time promise. Limits apply to each allocation; aggregate quotas remain with the native backend. @@ -86,7 +102,9 @@ scheduler credentials: `phase` is `pending`, `active`, `stopping`, `ended`, or `unknown`. `allocation` holds `num_nodes`, per-node `resources`, and their `evidence` (`configured`, -`requested`, or `unknown`). `dask` is observed separately: `observation` +`requested`, or `unknown`). Resource objects contain `cpus`, `memory` in GiB, +and `accelerators` as a one-entry type/count mapping, or `null` for CPU-only shapes. +`dask` is observed separately: `observation` (`unverified`, `reachable`, or `unreachable`), `ready`, and `workers`. A discovery that partially succeeds still exits 1. Native errors, invalid requests, and readiness timeouts also exit 1. diff --git a/docs/cli/materialize.md b/docs/cli/materialize.md index e75a72b5..0b7904bd 100644 --- a/docs/cli/materialize.md +++ b/docs/cli/materialize.md @@ -56,6 +56,13 @@ never touched, under any flag. already belong to a newer allocation), and confirm its recipes have stopped before cleaning results. Local containers may need separate termination through their runtime; see [execution limits](../user/cluster.md#execution-requirements-and-limits). +- **Honors recipe resources.** CPU, memory, and GPU requests must fit one worker + and are reserved through standard Dask scheduling. GPU recipes run one at a + time per worker and see only their requested devices. Recipes without `gpus` + see none. The whole selected graph is checked + before preparation or submission. Recipe `time_limit` is unsupported and + refused; allocation walltime remains supported. + See [recipe resource requirements](../user/cluster.md#recipe-resource-requirements). - **Fetches what it needs.** Declared inputs whose annexed content is not in this clone are fetched before anything hashes. - **Commits as it goes.** Each output lands in its own commit, written diff --git a/docs/cli/run.md b/docs/cli/run.md index 1e21c53e..deef5c44 100644 --- a/docs/cli/run.md +++ b/docs/cli/run.md @@ -2,9 +2,8 @@ Run an ad-hoc command in the project environment, under isolation. This is the probe verb: it executes exactly one command the way a -recipe would be executed — same environment, same sandbox — so "does -it work under `lc run`?" and "will it work as a recipe?" are the same -question. +recipe would be executed — same environment, same sandbox. Recipes must also +declare the resources they need, including their GPU count. ## Synopsis @@ -40,6 +39,11 @@ that variable to an existing cluster. Set command-specific values inside the command, for example `lc run "$CLUSTER" -- env NAME=value python script.py`. Containerized commands use the image's environment and the sandbox overlays. +The command reserves one worker's full CPU, memory, and GPU budgets for its duration. +Its CUDA mask exposes only the reserved devices; a CPU-only allocation exposes +none, even on a host with GPUs. A recipe instead declares its GPU count explicitly. +See [GPU allocations](../user/cluster.md#gpu-allocations) for container prerequisites. + Interrupting the CLI detaches its client; the remote command may still be running. Stop the allocation with `lc compute down` and its full ID (a name can already belong to a newer allocation) before working with files the interrupted command diff --git a/docs/user/cluster.md b/docs/user/cluster.md index 0250908d..a0b7dd90 100644 --- a/docs/user/cluster.md +++ b/docs/user/cluster.md @@ -8,10 +8,14 @@ present. `lc materialize --check` and `lc status` remain local project inspectio ## Start locally No configuration is needed on a fresh installation. When -`~/.lightcone/compute.yaml` is absent, Lightcone exposes one built-in `local` offer: +`~/.lightcone/compute.yaml` is absent, Lightcone exposes a built-in `local` CPU offer: one logical CPU, 1 GiB, one node, and fast startup. Its default lifetime is 30 minutes, with a maximum of two hours. This creates no catalog file and starts no processes until you launch a cluster. +On Linux, detected NVIDIA GPUs add a `local-gpu` offer, or `local-gpu-1`, +`local-gpu-2`, and so on for different models. Each groups the visible devices +of one model and keeps the same small CPU/memory defaults. Use a custom catalog +for larger CPU or RAM budgets. See [GPU allocations](#gpu-allocations). ```bash lc compute resources @@ -134,15 +138,20 @@ list: optional `context`, and optional provider `launch` settings. Namespaces must be unique, and so must each provider/`context` pair. - An offer has a unique `name`, the `connection` it uses, per-node `resources` - (`cpus` and `memory` in GiB), `max_nodes`, and `time` with a `default` no - longer than its `max`. `startup` is optional (`fast`, `batch`, or the default + (`cpus`, `memory`, and optional `accelerators`), `max_nodes`, and `time` with a + `default` no longer than its `max`. `startup` is optional (`fast`, `batch`, or the default `unknown`), written either as a bare class or as `{class: …, source: …}`. `config` holds provider-specific settings. Catalog errors identify the invalid field, for example `offers.0.resources.cpus`. Unknown common fields and duplicate YAML keys are rejected. CPU and node counts -must be positive integers; memory is in GiB and may be fractional if it is an -exact number of bytes, and durations use minutes or hours such as `30m` or `2h`. +must be positive integers. Compute memory follows SkyPilot's binary-unit +convention: `32`, `32GB`, and `32GiB` mean 32 GiB. Fractional quantities must +represent an exact number of bytes. An accelerator declaration names one type +and a positive whole count: `accelerators: A100:4` or `accelerators: {A100: 4}`; +`accelerators: A100` means one. Omit it for CPU-only offers. Durations use ordered +day/hour/minute/second units, such as +`30m`, `1h30m`, or `45s`. Selection takes the first offer in catalog order that matches the request. An offer this host cannot provide is skipped: a local offer with more nodes, CPUs or @@ -203,8 +212,11 @@ offers: ``` An offer's `config` accepts `submit` (`sbatch`, the default, or `salloc`), -`account`, `partition`, `qos`, `constraint`, and `reservation`. Slurm offers must -state memory as a whole number of MiB. +`account`, `partition`, `qos`, `constraint`, `reservation`, and `gpu_type`. +For a named accelerator offer, `gpu_type` maps the public type to the site's +native Slurm GRES name; it is required even when the spellings happen to match. +A generic `GPU` offer may omit it. Slurm offers must state memory as a whole +number of MiB. Every setting under a Slurm connection's `launch` mapping is optional. The defaults assume a home directory that the login and compute nodes share: @@ -303,11 +315,122 @@ Each worker's files go under `//attempt-/`. Before starting Dask, every rank checks that Slurm gave it what the plan requested: the node count, CPUs per task and its actual CPU affinity, and memory -per node. Ranks other than zero wait up to 120 seconds for the scheduler, which +per node. GPU allocations also validate the native GPU count and CUDA visibility. +Ranks other than zero wait up to 120 seconds for the scheduler, which has as long to start. A failed check or timeout logs `Slurm Dask startup failed: …` to the submission log and exits nonzero. Look there when a job is active but never becomes ready. +## GPU allocations + +GPU support currently covers whole NVIDIA CUDA devices on Linux. MIG devices, +fractional GPUs, and other accelerator vendors are not supported. CPU-only use +does not require CUDA. Discovery respects the allocation's native CUDA mask and +does not initialize CUDA in the reusable Dask worker. + +`lc compute resources` shows available accelerator types and counts. Requests +use SkyPilot-style `NAME[:COUNT]`: `--gpus A100` means one A100, +`--gpus A100:4` means exactly four, and `--gpus GPU:4` accepts any model with +exactly four. Names match case-insensitively; GPU counts do not accept `+`. +Omitting `--gpus`, or passing `0`, selects CPU-only offers. + +Names are catalog labels, without an accelerator alias registry. Local discovery +uses native model names with whitespace and punctuation normalized to hyphens, for example +`NVIDIA-A100-SXM4-80GB`. Copy the type shown by `resources`, or use `GPU:N`. +Local allocations do not reserve GPUs exclusively against other allocations or +programs on the host. + +For example, a Slurm site could expose this additional offer under `offers`. +The account, shape, constraint, and native GPU type must be adjusted to the site; +this is an illustrative configuration, not a tested hardware deployment: + +```yaml +- name: gpu-batch + connection: perlmutter + resources: {cpus: 32, memory: 128GB, accelerators: 'A100:4'} + max_nodes: 2 + time: {default: 1h, max: 4h} + startup: batch + config: + account: myproject + constraint: gpu + gpu_type: a100 +``` + +With such an offer configured, inspect its plan before launching: + +```bash +lc compute launch --cpus 32 --memory 128GB --gpus A100:4 --dry-run +``` + +GPU commands receive selected device UUIDs through `CUDA_VISIBLE_DEVICES`. +CPU recipes receive an empty mask. This is cooperative device selection; +native OS/cgroup permissions remain the access boundary. Container runtimes +also need their normal GPU integration: + +- **Docker:** install the NVIDIA Container Toolkit; Lightcone supplies explicit + UUIDs through `--gpus`. See [NVIDIA's Docker setup](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html#configuring-docker). +- **Podman:** configure NVIDIA CDI with UUID device names. Lightcone supplies + `--device=nvidia.com/gpu=UUID`; see [NVIDIA's CDI guide](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/cdi-support.html). +- **podman-hpc:** Lightcone adds `--gpu` and the selected CUDA mask; see + [NERSC's GPU container guidance](https://docs.nersc.gov/development/containers/podman-hpc/overview/#using-nvidia-gpus-in-podman-hpc). + +CPU containers also override `NVIDIA_VISIBLE_DEVICES` to `void`, so a CUDA image's +default cannot enable GPU injection. GPU discovery, reservations, masks, and +runtime arguments are tested with simulated inventories and real subprocesses; +physical GPU execution has not yet been validated. + +## Recipe resource requirements + +Declare each recipe's needs in `astra.yaml`: + +```yaml +recipe: + command: python src/fit.py {output} + resources: + cpus: 4 + memory: 8Gi + gpus: 1 +``` + +Each recipe runs on one worker. Its CPU, memory, and GPU request must fit that +worker, even when the cluster has several nodes. Dask reserves CPU and memory +while the task runs, so recipes can run together only when their combined +requests fit. `task_slots_per_node` also caps concurrent tasks; it does not +limit how many CPUs a single recipe may request. + +CPUs must be positive whole numbers and default to one. Memory needs units: +`512Mi` and `8Gi` are binary sizes; `8GB` is decimal, unlike compute memory. +Bare quantities are not accepted. Without a memory +declaration, a recipe reserves the worker's entire memory budget, so only +one such recipe runs per worker. + +Recipe `gpus` is a nonnegative whole count, defaulting to zero; accelerator type +selection belongs to cluster allocation. A GPU recipe reserves the worker's +entire GPU budget, so only one GPU recipe runs on that worker at a time, while +its command sees only the requested number of devices. CPU recipes may still +run alongside it when CPU, memory, and task slots permit; their CUDA mask is +empty. `lc run` reserves the worker's entire CPU, memory, and GPU budgets and +exposes only those reserved GPUs. + +Recipe `time_limit` is not supported and is refused before preparation or +execution. Set the allocation lifetime with `lc compute launch --time` instead. +Fractional CPU/GPU counts, GPU model requests inside a recipe, and disk requests +are also rejected rather than ignored. + +`lc materialize` validates the complete selected graph against the cluster +before fetching inputs, preparing the environment, or starting a recipe. This +also validates currently complete outputs, which workers may need to rebuild +after an upstream change. Use `lc materialize --check` to inspect currency +without allocation. Read-only `status` and `--check` accept valid ASTRA resource +declarations even when this executor cannot satisfy them. + +These are scheduling reservations, not per-recipe CPU or RAM enforcement. +Recipes must respect their declarations; a subprocess can otherwise exceed +its request. Slurm enforces the overall allocation, while local execution +uses cooperative budgets. Leave capacity for the scheduler, workers, and other +overhead when declaring recipe requirements. + ## Execution requirements and limits Driver and workers must see the same project, prepared environment, and inputs diff --git a/src/lightcone/cli/compute.py b/src/lightcone/cli/compute.py index 59fb2517..aa139375 100644 --- a/src/lightcone/cli/compute.py +++ b/src/lightcone/cli/compute.py @@ -49,6 +49,11 @@ def _table(headers: list[str], rows: list[list[str]]) -> None: Console(markup=False).print(table) +def _duration(seconds: int) -> str: + minutes, remainder = divmod(seconds, 60) + return (f"{minutes}m" if minutes else "") + (f"{remainder}s" if remainder else "") + + @click.group() def compute() -> None: """Allocate resources, inspect clusters, and end allocations. @@ -70,15 +75,19 @@ def resources(as_json: bool) -> None: click.echo(json.dumps(data)) return _table( - ["OFFER", "CPUS", "MEMORY", "MAX NODES", "DEFAULT", "MAX TIME", "STARTUP"], + ["OFFER", "CPUS", "MEMORY", "GPUS", "MAX NODES", "DEFAULT", "MAX TIME", "STARTUP"], [ [ offer["name"], str(offer["resources"]["cpus"]), f"{offer['resources']['memory']:g} GiB", + ", ".join( + f"{name}:{count}" + for name, count in (offer["resources"]["accelerators"] or {}).items() + ) or "-", str(offer["max_nodes"]), - f"{offer['time']['default_seconds'] // 60}m", - f"{offer['time']['max_seconds'] // 60}m", + _duration(offer["time"]["default_seconds"]), + _duration(offer["time"]["max_seconds"]), offer["startup"], ] for offer in data["offers"] @@ -89,10 +98,13 @@ def resources(as_json: bool) -> None: @compute.command() @click.option("--name", help="Cluster name; defaults to a generated short name.") @click.option("--cpus", required=True, help="Logical CPUs per node; suffix + requests a minimum.") -@click.option("--memory", required=True, help="GiB per node; suffix + requests a minimum.") +@click.option("--memory", required=True, + help="Memory per node, e.g. 16 or 16GB; suffix + requests a minimum.") +@click.option("--gpus", default="0", show_default=True, + help="Accelerator NAME[:COUNT] per node, e.g. A100:4 or GPU:1; 0 requests CPU only.") @click.option("--num-nodes", default=1, type=click.IntRange(min=1), show_default=True) @click.option( - "--time", "walltime", help="Requested walltime, e.g. 30m or 2h; defaults to the offer." + "--time", "walltime", help="Requested walltime, e.g. 30m or 1h30m; defaults to the offer." ) @click.option( "--startup", type=click.Choice(["fast"]), help="Require a fast startup service class." @@ -103,6 +115,7 @@ def launch( name: str | None, cpus: str, memory: str, + gpus: str, num_nodes: int, walltime: str | None, startup: str | None, @@ -119,6 +132,7 @@ def launch( Request.parse( cpus, memory, + gpus=gpus, num_nodes=num_nodes, time=walltime, startup=startup, diff --git a/src/lightcone/engine/compute/__init__.py b/src/lightcone/engine/compute/__init__.py index 0f3c40c6..2139ca9b 100644 --- a/src/lightcone/engine/compute/__init__.py +++ b/src/lightcone/engine/compute/__init__.py @@ -102,7 +102,10 @@ def resources(self) -> dict[str, Any]: """Describe configured policy, without inventing live free capacity.""" return { "schema_version": 1, - "units": {"cpus": "logical CPUs per node", "memory": "GiB per node"}, + "units": { + "cpus": "logical CPUs per node", "memory": "GiB per node", + "accelerators": "type and count per node", + }, "offers": [ { "name": offer.name, @@ -142,6 +145,12 @@ def plan(self, request: Request, *, name: str | None = None) -> LaunchPlan: else offer.resources.memory_bytes != request.memory_bytes ): continue + if request.accelerators is None: + matches_accelerators = offer.resources.accelerators is None + else: + matches_accelerators = request.accelerators.matches(offer.resources.accelerators) + if not matches_accelerators: + continue try: provider = self.provider(self.catalog.connections[offer.connection]) plan = provider.plan(offer, request) diff --git a/src/lightcone/engine/compute/catalog.py b/src/lightcone/engine/compute/catalog.py index a9b92aa8..6f6d1f28 100644 --- a/src/lightcone/engine/compute/catalog.py +++ b/src/lightcone/engine/compute/catalog.py @@ -10,6 +10,9 @@ import yaml from pydantic import Field, ValidationError, model_validator +from lightcone.engine import gpu +from lightcone.engine.project import ProjectError + from .model import ( ComputeError, ComputeModel, @@ -88,15 +91,29 @@ def load(cls, path: Path | None = None) -> Catalog: except FileNotFoundError as exc: if configured or path.is_symlink(): raise ComputeError(f"cannot read compute catalog {path}: {exc}") from exc + local = Offer( + name="local", connection="local", + resources=Resources(cpus=1, memory_gib=Decimal(1)), + max_nodes=1, time=TimeLimits(default="30m", max="2h"), + startup=Startup(class_="fast"), + ) + offers = [local] + try: + devices = gpu.inventory() + except ProjectError: + devices = () # Optional discovery must not disable CPU-only execution. + names = sorted({device.name for device in devices}) + for index, name in enumerate(names, 1): + offers.append(local.replace( + name="local-gpu" if len(names) == 1 else f"local-gpu-{index}", + resources=local.resources.replace(accelerators={ + name: sum(device.name == name for device in devices), + }), + )) return cls( version=1, connections={"local": Connection(namespace=_LOCAL_NAMESPACE, provider="local")}, - offers=[Offer( - name="local", connection="local", - resources=Resources(cpus=1, memory_gib=Decimal(1)), - max_nodes=1, time=TimeLimits(default="30m", max="2h"), - startup=Startup(class_="fast"), - )], + offers=offers, ) except (OSError, UnicodeError, yaml.YAMLError) as exc: raise ComputeError(f"cannot read compute catalog {path}: {exc}") from exc diff --git a/src/lightcone/engine/compute/local.py b/src/lightcone/engine/compute/local.py index 0474d146..a99ac4ed 100644 --- a/src/lightcone/engine/compute/local.py +++ b/src/lightcone/engine/compute/local.py @@ -19,6 +19,7 @@ import psutil +from lightcone.engine import gpu from lightcone.engine.compute.model import ( ComputeError, Connection, @@ -41,6 +42,7 @@ read_private_json, write_private_json, ) +from lightcone.engine.project import ProjectError _OWNER_MODULE = "lightcone.engine.compute.local_runtime" _STOP_GRACE = 3.0 @@ -104,6 +106,21 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: if offer.resources.cpus > CPU_COUNT or offer.resources.memory_bytes > MEMORY_LIMIT: raise UnavailableOfferError("the local offer exceeds this host's CPU or RAM capacity") + try: + inventory = gpu.inventory() if offer.resources.gpus else () + except ProjectError as exc: + raise UnavailableOfferError(str(exc)) from exc + accelerator = offer.resources.accelerator_name or "GPU" + available = tuple( + device for device in inventory + if accelerator.casefold() == "gpu" or device.name.casefold() == accelerator.casefold() + ) + if len(available) < offer.resources.gpus: + raise UnavailableOfferError( + f"the local offer exceeds this host's visible CUDA capacity for {accelerator}" + ) + selected = available[:offer.resources.gpus] + names = {device.name for device in selected} seconds = request.seconds if request.seconds is not None else offer.time.default_seconds if seconds <= 0 or seconds > offer.time.max_seconds: raise ComputeError("local allocations require a finite time within the offer's limit") @@ -123,7 +140,9 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: "connection_root": str(self.root), "scratch_root": str(scratch), "task_slots_per_node": slots, - "resource_enforcement": "cooperative; no exclusive CPU or RAM reservation", + "gpu_devices": tuple(device.uuid for device in selected), + "accelerator_name": next(iter(names)) if len(names) == 1 else "GPU", + "resource_enforcement": "cooperative; no exclusive CPU, RAM, or GPU reservation", "termination_grace_seconds": _STOP_GRACE, }, ) @@ -134,6 +153,11 @@ def launch(self, plan: LaunchPlan) -> Identity: raise ComputeError("local launch plan belongs to a different connection or node count") if plan.name is not None: validate_name(plan.name) + devices = tuple(plan.details["gpu_devices"]) + if len(devices) != plan.resources.gpus or ( + devices and not set(devices).issubset(gpu.visible_devices()) + ): + raise UnavailableOfferError("the local plan's selected CUDA GPUs are no longer visible") boot = _boot_identity() token = uuid4().hex directory = private_directory(self.root / token, create=True) @@ -160,6 +184,7 @@ def launch(self, plan: LaunchPlan) -> Identity: stderr=subprocess.DEVNULL, start_new_session=True, close_fds=True, + env={**os.environ, "CUDA_VISIBLE_DEVICES": ",".join(devices)}, ) identity = Identity( namespace=self.connection.namespace, native_id=str(process.pid), token=token, @@ -174,6 +199,8 @@ def launch(self, plan: LaunchPlan) -> Identity: "host": identity.host, "cpus": plan.resources.cpus, "memory": plan.resources.memory_bytes, + "gpus": plan.resources.gpus, + "accelerator_name": plan.details["accelerator_name"], } write_private_json(directory / "identity.json", record) # The child waits for this file before publishing its TLS connection. @@ -233,6 +260,10 @@ def _record(self, identity: Identity) -> tuple[Path, dict[str, Any]]: raise ComputeError("the private locator does not match this local allocation identity") positive_int(record.get("cpus"), "recorded local cpus") positive_int(record.get("memory"), "recorded local memory") + if type(record.get("gpus")) is not int or record["gpus"] < 0: + raise ComputeError("recorded local gpus must be a nonnegative integer") + if not isinstance(record.get("accelerator_name"), str) or not record["accelerator_name"]: + raise ComputeError("recorded local accelerator_name must be a nonempty string") return directory, record def _process( @@ -347,7 +378,7 @@ def inspect(self, identity: Identity) -> Snapshot: except psutil.NoSuchProcess: process = None native_state = "not-running" - reason = "CPU and RAM budgets are cooperative, not exclusive OS reservations" + reason = "CPU, RAM, and GPU budgets are cooperative, not exclusive OS reservations" if process is None and (directory / "error.json").exists(): reason = str(read_private_json(directory / "error.json").get("error", "")) return Snapshot( @@ -355,6 +386,8 @@ def inspect(self, identity: Identity) -> Snapshot: phase="active" if process is not None else "ended", resources=Resources.from_bytes( cpus=int(record["cpus"]), memory_bytes=int(record["memory"]), + gpus=record["gpus"], + accelerator_name=record["accelerator_name"], ), num_nodes=1, evidence="configured", diff --git a/src/lightcone/engine/compute/local_runtime.py b/src/lightcone/engine/compute/local_runtime.py index dbb360e6..e23d83b8 100644 --- a/src/lightcone/engine/compute/local_runtime.py +++ b/src/lightcone/engine/compute/local_runtime.py @@ -11,6 +11,7 @@ from pathlib import Path from types import FrameType +from lightcone.engine.compute.model import ComputeError from lightcone.engine.compute.runtime import ( SCHEDULER_CONFIG, create_security, @@ -54,6 +55,13 @@ def expire(_signum: int, _frame: FrameType | None) -> None: from distributed import LocalCluster security = create_security(directory) + allocation = read_private_json(directory / "identity.json") + gpus = int(allocation["gpus"]) + if gpus: + from lightcone.engine.gpu import visible_devices + + if len(visible_devices()) != gpus: + raise ComputeError("visible CUDA GPUs do not match the local allocation envelope") with dask.config.set(SCHEDULER_CONFIG), LocalCluster( # type: ignore[no-untyped-call] n_workers=1, threads_per_worker=int(launch["task_slots"]), @@ -73,6 +81,9 @@ def expire(_signum: int, _frame: FrameType | None) -> None: # Recipes use subprocesses: Dask's Python-process RSS cannot enforce # their RAM envelope. Local resource limits are explicitly cooperative. memory_limit=0, + resources={ + "CPU": int(allocation["cpus"]), "MEMORY": int(allocation["memory"]), "GPU": gpus, + }, silence_logs=50, ) as cluster: write_private_json( diff --git a/src/lightcone/engine/compute/model.py b/src/lightcone/engine/compute/model.py index ba7a25d1..b29cd2c2 100644 --- a/src/lightcone/engine/compute/model.py +++ b/src/lightcone/engine/compute/model.py @@ -7,7 +7,7 @@ import re from collections.abc import Callable, Sequence from contextlib import AbstractContextManager -from decimal import Decimal, InvalidOperation, localcontext +from decimal import Decimal, localcontext from typing import Annotated, Any, Literal, Protocol, Self from uuid import UUID @@ -19,10 +19,12 @@ Field, PlainSerializer, ValidationError, + model_serializer, model_validator, ) from lightcone.engine.project import ProjectError +from lightcone.engine.units import duration_seconds, whole_bytes GIB = 1024**3 @@ -43,25 +45,24 @@ class UnavailableOfferError(ComputeError): def duration(value: object) -> int: - """Parse an explicit positive whole-minute/hour duration into seconds.""" - match = re.fullmatch(r"([1-9][0-9]*)([mh])", str(value)) - if match is None: - raise ComputeError("duration must be a positive number of minutes or hours, e.g. 30m or 1h") - return int(match[1]) * (60 if match[2] == "m" else 3600) + """Parse an explicit positive duration into seconds.""" + try: + return duration_seconds(value) + except ValueError as exc: + raise ComputeError(str(exc)) from exc def memory_bytes(value: object) -> int: - """Convert positive decimal GiB to an exact integer number of bytes.""" - if isinstance(value, bool) or not re.fullmatch(r"[0-9]+(?:\.[0-9]+)?", str(value)): - raise ComputeError("memory must be a positive number of GiB") + """Parse SkyPilot compute memory: bare GiB or binary KB/MB/GB/TB/PB units.""" + match = re.fullmatch(r"([0-9]+(?:\.[0-9]+)?)([KMGTPE]I?B|B)?", str(value), re.IGNORECASE) + if isinstance(value, bool) or match is None: + raise ComputeError("memory must be positive GiB or a quantity with KB/MB/GB/TB/PB units") + unit = (match[2] or "GB").upper().replace("I", "") + units = {"B": 1, **{f"{prefix}B": 1024**index for index, prefix in enumerate("KMGTPE", 1)}} try: - numerator, denominator = Decimal(str(value)).as_integer_ratio() - except InvalidOperation as exc: - raise ComputeError("memory must be a positive number of GiB") from exc - amount, remainder = divmod(numerator * GIB, denominator) - if amount <= 0 or remainder: - raise ComputeError("memory must be positive GiB exactly representable in bytes") - return amount + return whole_bytes(match[1], units[unit]) + except ValueError as exc: + raise ComputeError("memory must be positive GiB exactly representable in bytes") from exc def gib_from_bytes(value: int) -> Decimal: @@ -124,10 +125,10 @@ def _gib(value: object) -> Decimal: # digits when applying the same quantity rules as the CLI. literal = format(value, "f") if isinstance(value, Decimal) else value try: - memory_bytes(literal) + size = memory_bytes(literal) except ComputeError as exc: raise ValueError(str(exc)) from exc - return Decimal(str(literal)) + return gib_from_bytes(size) def _duration(value: str) -> str: @@ -161,24 +162,76 @@ def replace(self, **changes: Any) -> Self: return type(self).model_validate({**self.model_dump(), **changes}) +class Accelerator(ComputeModel): + """One accelerator type and whole-device count, using SkyPilot's notation.""" + + name: Annotated[str, Field(pattern=r"^[A-Za-z0-9][A-Za-z0-9_.-]*$")] + count: PositiveInt = 1 + + @model_validator(mode="before") + @classmethod + def shorthand(cls, value: Any) -> Any: + if isinstance(value, str): + match = re.fullmatch(r"([A-Za-z0-9][A-Za-z0-9_.-]*)(?::([1-9][0-9]*))?", value) + if match is None: + raise ValueError("accelerators must be NAME[:COUNT] with a positive whole count") + return {"name": match[1], "count": int(match[2] or 1)} + if isinstance(value, dict) and not ("name" in value and set(value) <= {"name", "count"}): + if len(value) != 1: + raise ValueError("accelerators must specify exactly one type and whole count") + name, count = next(iter(value.items())) + return {"name": name, "count": count} + return value + + @model_serializer + def serialize(self) -> dict[str, int]: + """Use the same single-type mapping in native records and public output.""" + return {self.name: self.count} + + def matches(self, available: Accelerator | None) -> bool: + """Match exact counts and types; the generic GPU name accepts any type.""" + return available is not None and self.count == available.count and ( + self.name.casefold() == "gpu" or self.name.casefold() == available.name.casefold() + ) + + class Resources(ComputeModel): """A per-node resource envelope with explicitly named memory units.""" cpus: Count memory_gib: GiB = Field(validation_alias="memory", serialization_alias="memory") + accelerators: Accelerator | None = None + + @property + def gpus(self) -> int: + return self.accelerators.count if self.accelerators is not None else 0 + + @property + def accelerator_name(self) -> str | None: + return self.accelerators.name if self.accelerators is not None else None @property def memory_bytes(self) -> int: return memory_bytes(format(self.memory_gib, "f")) @classmethod - def from_bytes(cls, *, cpus: int, memory_bytes: int) -> Self: + def from_bytes( + cls, *, cpus: int, memory_bytes: int, gpus: int = 0, accelerator_name: str = "GPU", + ) -> Self: """Represent native byte counts exactly, independently of Decimal precision.""" - return cls(cpus=cpus, memory_gib=gib_from_bytes(memory_bytes)) + if type(gpus) is not int or gpus < 0: + raise ValueError("gpus must be a nonnegative whole count") + return cls( + cpus=cpus, memory_gib=gib_from_bytes(memory_bytes), + accelerators=Accelerator(name=accelerator_name, count=gpus) if gpus else None, + ) - def as_dict(self) -> dict[str, int | float]: + def as_dict(self) -> dict[str, Any]: """Render public memory in GiB.""" - return {"cpus": self.cpus, "memory": self.memory_bytes / GIB} + return { + "cpus": self.cpus, "memory": self.memory_bytes / GIB, + "accelerators": self.accelerators.model_dump() if self.accelerators else None, + } class Request(ComputeModel): @@ -186,18 +239,28 @@ class Request(ComputeModel): cpus: PositiveInt memory_bytes: PositiveInt + accelerators: Accelerator | None = None num_nodes: PositiveInt = 1 min_cpus: bool = False min_memory: bool = False seconds: PositiveInt | None = None startup: Literal["fast"] | None = None + @property + def gpus(self) -> int: + return self.accelerators.count if self.accelerators is not None else 0 + + @property + def accelerator_name(self) -> str | None: + return self.accelerators.name if self.accelerators is not None else None + @classmethod def parse( cls, cpus: str, memory: str, *, + gpus: str = "0", num_nodes: int = 1, time: str | None = None, startup: str | None = None, @@ -207,6 +270,7 @@ def parse( return cls.model_validate({ "cpus": positive_int(cpus.removesuffix("+"), "cpus"), "memory_bytes": memory_bytes(memory.removesuffix("+")), + "accelerators": None if gpus == "0" else gpus, "num_nodes": positive_int(num_nodes, "num_nodes"), "min_cpus": cpus.endswith("+"), "min_memory": memory.endswith("+"), @@ -223,6 +287,7 @@ def as_dict(self) -> dict[str, Any]: "resources": { "cpus": f"{self.cpus}{'+' if self.min_cpus else ''}", "memory": f"{gib_from_bytes(self.memory_bytes):f}{'+' if self.min_memory else ''}", + "accelerators": self.accelerators.model_dump() if self.accelerators else None, }, "time_seconds": self.seconds, "startup": self.startup, diff --git a/src/lightcone/engine/compute/slurm.py b/src/lightcone/engine/compute/slurm.py index 86d66729..cd43f510 100644 --- a/src/lightcone/engine/compute/slurm.py +++ b/src/lightcone/engine/compute/slurm.py @@ -100,6 +100,44 @@ def _value(value: object, name: str) -> str: return value +def _native_gpus(row: Mapping[str, str]) -> tuple[str, int] | None: + """Read a per-node GPU count, never divide an aggregate into invented grants.""" + for field, prefix in (("TresPerNode", "gres/"), ("Gres", "")): + value = row.get(field, "") + counts = [] + names = set() + for entry in value.split(","): + if not entry.startswith(f"{prefix}gpu"): + continue + match = re.fullmatch( + rf"{prefix}gpu(?::([A-Za-z0-9][A-Za-z0-9_.-]*))?[:=]([0-9]+)", entry, + ) + if match is None: + return None + names.add(match[1] or "GPU") + counts.append(int(match[2])) + if counts: + if "GPU" in names and len(names) > 1: + return None # A total plus typed subcounts must not be double-counted. + return next(iter(names)) if len(names) == 1 else "GPU", sum(counts) + if field == "Gres" and value in {"(null)", "N/A", "none"}: + return "GPU", 0 + # Complete native TRES with no GPU entry proves a CPU-only job. An + # aggregate GPU total does not prove a homogeneous per-node allocation. + for field in ("ReqTRES", "AllocTRES"): + value = row.get(field, "") + if value and value not in {"(null)", "N/A"}: + entries = value.split(",") + if any(entry.startswith("gres/gpu") for entry in entries): + return None + if all("=" in entry for entry in entries) and any( + entry.startswith("cpu=") for entry in entries + ): + return "GPU", 0 + return None + return None + + class SlurmProvider: """Submit, observe, and cancel allocations using the selected Slurm authority.""" @@ -185,11 +223,26 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: "qos", "constraint", "reservation", + "gpu_type", }: raise ComputeError(f"unknown Slurm offer settings: {', '.join(sorted(extra))}") submit = config.get("submit", "sbatch") if submit not in {"sbatch", "salloc"}: raise ComputeError("Slurm submit must be sbatch or salloc") + gpu_type = config.get("gpu_type") + if gpu_type is not None: + if not isinstance(gpu_type, str) or not re.fullmatch( + r"[A-Za-z0-9][A-Za-z0-9_.-]*", gpu_type, + ): + raise ComputeError("Slurm gpu_type must name one native GPU GRES type") + if not offer.resources.gpus: + raise ComputeError("Slurm gpu_type requires accelerator resources") + elif (offer.resources.accelerator_name or "GPU").casefold() != "gpu": + raise ComputeError("a named Slurm accelerator offer requires its native gpu_type") + gres = ( + f"gpu:{gpu_type + ':' if gpu_type else ''}{offer.resources.gpus}" + if offer.resources.gpus else "none" + ) cpus, memory = offer.resources.cpus, offer.resources.memory_bytes if memory % _MIB: raise ComputeError("Slurm offer memory must be an exact whole number of MiB") @@ -223,6 +276,8 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: f"--time={hours:02}:{minutes:02}:{seconds_part:02}", f"--chdir={paths['cwd']}", ] + if offer.resources.gpus: + args.append(f"--gres={gres}") return LaunchPlan( connection=self.connection, offer=offer, @@ -231,6 +286,7 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: details={ "submit": submit, "native_args": args, + "gres": gres, **paths, "task_slots_per_node": slots, "interface": interface, @@ -244,6 +300,7 @@ def _payload(self, plan: LaunchPlan, token: str) -> list[str]: f"--ntasks={plan.num_nodes}", "--ntasks-per-node=1", f"--cpus-per-task={plan.resources.cpus}", + f"--gres={details['gres']}", # One process per node holds the whole allocation, so binding to # exactly its allocated hardware threads is the only useful mask. "--cpu-bind=threads", @@ -264,6 +321,8 @@ def _payload(self, plan: LaunchPlan, token: str) -> list[str]: str(plan.resources.cpus), "--memory-bytes", str(plan.resources.memory_bytes), + "--gpus", + str(plan.resources.gpus), "--task-slots", str(details["task_slots_per_node"]), ] @@ -495,11 +554,16 @@ def _snapshot(self, identity: Identity, row: Mapping[str, str]) -> Snapshot: nodes = int(nodes_text) if nodes_text.isdigit() and int(nodes_text) > 0 else None cpus = row.get("CPUs/Task", "") memory = re.fullmatch(r"([0-9]+)([KMGT]?)", row.get("MinMemoryNode", "")) + accelerators = _native_gpus(row) resources = None - if cpus.isdigit() and int(cpus) > 0 and memory and int(memory[1]) > 0: + if ( + cpus.isdigit() and int(cpus) > 0 and memory and int(memory[1]) > 0 + and accelerators is not None + ): scale = {"": _MIB, "K": 1024, "M": _MIB, "G": 1024**3, "T": 1024**4} resources = Resources.from_bytes( - cpus=int(cpus), memory_bytes=int(memory[1]) * scale[memory[2]] + cpus=int(cpus), memory_bytes=int(memory[1]) * scale[memory[2]], + accelerator_name=accelerators[0], gpus=accelerators[1], ) return Snapshot( identity=identity, diff --git a/src/lightcone/engine/compute/slurm_bootstrap.py b/src/lightcone/engine/compute/slurm_bootstrap.py index 6e35cdf0..35a718ef 100644 --- a/src/lightcone/engine/compute/slurm_bootstrap.py +++ b/src/lightcone/engine/compute/slurm_bootstrap.py @@ -14,6 +14,7 @@ from typing import Any from uuid import UUID +from lightcone.engine import gpu from lightcone.engine.compute.model import ComputeError, Connection, Identity from lightcone.engine.compute.runtime import ( SCHEDULER_CONFIG, @@ -48,6 +49,7 @@ def _allocation(args: argparse.Namespace) -> tuple[Identity, int, int]: args.num_nodes < 1 or args.cpus < 1 or args.memory_bytes < 1 + or args.gpus < 0 or args.task_slots < 1 or values["SLURM_NTASKS"] != args.num_nodes or values["SLURM_JOB_NUM_NODES"] != args.num_nodes @@ -61,6 +63,14 @@ def _allocation(args: argparse.Namespace) -> tuple[Identity, int, int]: memory = os.environ.get("SLURM_MEM_PER_NODE", "") if not memory.isdigit() or int(memory) * 1024**2 < args.memory_bytes: raise ComputeError("native per-node memory does not match the allocation envelope") + if args.gpus: + native_gpus = os.environ.get("SLURM_GPUS_ON_NODE", "") + if not native_gpus.isdigit() or int(native_gpus) < args.gpus: + raise ComputeError("native per-node GPUs do not match the allocation envelope") + # Slurm/NVML numbers devices in PCI order; CUDA's default is different. + os.environ["CUDA_DEVICE_ORDER"] = "PCI_BUS_ID" + if len(gpu.visible_devices()) < args.gpus: + raise ComputeError("visible CUDA GPUs are fewer than the allocation envelope") restarts = os.environ.get("SLURM_RESTART_COUNT", "0") if not restarts.isdigit(): raise ComputeError("invalid native Slurm restart count") @@ -76,6 +86,8 @@ async def run(args: argparse.Namespace) -> None: from distributed import Scheduler, Worker identity, restarts, rank = _allocation(args) + if not args.gpus: + os.environ["CUDA_VISIBLE_DEVICES"] = "" connection = Connection( namespace=identity.namespace, provider="slurm", launch={"connection_root": args.connection_root}, @@ -94,6 +106,7 @@ async def run(args: argparse.Namespace) -> None: "num_nodes": args.num_nodes, "cpus": args.cpus, "memory_bytes": args.memory_bytes, + "gpus": args.gpus, "task_slots": args.task_slots, } address = {"interface": args.interface} if args.interface else {"host": socket.gethostname()} @@ -101,6 +114,7 @@ async def run(args: argparse.Namespace) -> None: **address, "nthreads": args.task_slots, "memory_limit": 0, + "resources": {"CPU": args.cpus, "MEMORY": args.memory_bytes, "GPU": args.gpus}, "local_directory": str(scratch), "dashboard_address": "127.0.0.1:0", "dashboard": False, @@ -157,7 +171,7 @@ def main() -> None: parser = argparse.ArgumentParser(description=__doc__) for name in ("submission", "namespace", "connection-root"): parser.add_argument(f"--{name}", required=True) - for name in ("num-nodes", "cpus", "memory-bytes", "task-slots"): + for name in ("num-nodes", "cpus", "memory-bytes", "gpus", "task-slots"): parser.add_argument(f"--{name}", required=True, type=int) parser.add_argument("--scratch-root") parser.add_argument("--interface") diff --git a/src/lightcone/engine/container.py b/src/lightcone/engine/container.py index d7f995de..c1d48f3c 100644 --- a/src/lightcone/engine/container.py +++ b/src/lightcone/engine/container.py @@ -29,6 +29,7 @@ import sys import tarfile import tempfile +from collections.abc import Sequence from dataclasses import dataclass from pathlib import Path from typing import Literal, cast @@ -408,7 +409,8 @@ def converge(runtime: Runtime) -> list[str]: def policy_for( - runtime: Runtime, read_paths: list[Path], *, write_dir: Path | None = None + runtime: Runtime, read_paths: list[Path], *, write_dir: Path | None = None, + gpu_devices: Sequence[str] = (), ) -> sandbox.Policy: """Build the exec policy for a resolved runtime. @@ -422,6 +424,7 @@ def policy_for( runtime: The resolved runtime. read_paths: Declared inputs, as :func:`sandbox.exec_policy` takes. write_dir: The directory a recipe's output lands in; absent for a probe. + gpu_devices: Allocated CUDA device UUIDs to expose to this command. Returns: The policy for this world. @@ -432,6 +435,7 @@ def policy_for( env_dir=runtime.env_dir, containerized=runtime.mode == "containerized", write_dir=write_dir, + gpu_devices=gpu_devices, ) @@ -683,4 +687,3 @@ def _machine_preflight(root: Path) -> None: f" podman machine set --volume {root}\n podman machine start" ) - diff --git a/src/lightcone/engine/execution_resources.py b/src/lightcone/engine/execution_resources.py new file mode 100644 index 00000000..ef6513f6 --- /dev/null +++ b/src/lightcone/engine/execution_resources.py @@ -0,0 +1,162 @@ +"""Recipe resource requests and admission to stock Dask workers.""" + +from __future__ import annotations + +import math +import re +from typing import Any, Self + +from pydantic import BaseModel, ConfigDict, Field, ValidationError, field_validator + +from lightcone.engine.project import ProjectError +from lightcone.engine.units import whole_bytes + + +class TaskResources(BaseModel): + """Reserve CPU, memory, and GPU capacity on stock Dask workers.""" + + model_config = ConfigDict(frozen=True, strict=True, extra="forbid") + + cpus: int = Field(default=1, gt=0) + gpus: int = Field(default=0, ge=0) + memory_bytes: int | None = Field(default=None, gt=0) + + @field_validator("cpus", mode="before") + @classmethod + def _whole_cpus(cls, value: object) -> object: + # ASTRA permits fractional CPUs. This executor reserves whole CPUs; + # accepting 4.0 is exact, whereas rounding 0.5 would hide a policy change. + if isinstance(value, float): + if not math.isfinite(value) or not value.is_integer(): + raise ValueError("fractional CPUs are not supported; request whole CPUs") + return int(value) + return value + + @classmethod + def parse(cls, value: object) -> Self: + """Parse ASTRA's ``recipe.resources`` into explicit execution units. + + Args: + value: The recipe resource mapping, or ``None`` when omitted. + + Returns: + A validated CPU, memory, and GPU request. + + Raises: + ProjectError: A requirement is invalid or cannot be honored. + """ + if value is None: + return cls() + if not isinstance(value, dict): + raise ProjectError("recipe.resources must be a mapping") + if "time_limit" in value: + raise ProjectError( + "recipe time_limit is not supported; use cluster allocation walltime" + ) + if extra := value.keys() - {"cpus", "memory", "gpus"}: + names = ", ".join(sorted(map(str, extra))) + raise ProjectError(f"unsupported recipe resource requirements: {names}") + parsed = {"cpus": value.get("cpus", 1), "gpus": value.get("gpus", 0)} + if "memory" in value: + parsed["memory_bytes"] = _memory(value["memory"]) + try: + return cls.model_validate(parsed) + except ValidationError as exc: + detail = "; ".join( + f"{'.'.join(map(str, item['loc']))}: {item['msg']}" + for item in exc.errors(include_url=False, include_input=False) + ) + raise ProjectError(f"invalid recipe resources: {detail}") from exc + + def requirements( + self, workers: dict[str, Any], *, whole_worker: bool = False + ) -> dict[str, float]: + """Choose Dask resource reservations that fit an individual worker. + + Args: + workers: The ``workers`` mapping from Dask's scheduler information. + whole_worker: Reserve a worker's entire CPU, memory, and GPU budget for + an arbitrary command without declared resource requirements. + + Returns: + Dask's numeric ``CPU``, ``MEMORY``, and optional ``GPU`` reservations. + GPU recipes reserve the worker's full GPU budget, so only one GPU + recipe uses that worker's device mask at a time. + + Raises: + ProjectError: Capacity is unknown, a request cannot fit, or an + unspecified budget is ambiguous across heterogeneous workers. + """ + capacities: set[tuple[float, float, float]] = set() + for info in workers.values(): + resources = info.get("resources", {}) if isinstance(info, dict) else {} + values = [] + for name in ("CPU", "MEMORY", "GPU"): + value = ( + resources.get(name, 0 if name == "GPU" else None) + if isinstance(resources, dict) else None + ) + if ( + isinstance(value, bool) + or not isinstance(value, (int, float)) + or not math.isfinite(value) + or (value < 0 if name == "GPU" else value <= 0) + or not float(value).is_integer() + ): + raise ProjectError( + "cluster workers must advertise positive whole CPU and MEMORY budgets " + "and a nonnegative whole GPU count; " + "relaunch the cluster with the current Lightcone installation" + ) + values.append(float(value)) + capacities.add((values[0], values[1], values[2])) + if not capacities: + raise ProjectError("cluster has no workers available for execution") + if ( + (whole_worker or self.memory_bytes is None) + and len({(cpus, memory) for cpus, memory, _ in capacities}) != 1 + ): + raise ProjectError( + "unspecified task resources require workers with identical CPU and memory budgets" + ) + available_cpus, available_memory, _ = next(iter(capacities)) + requested = { + "CPU": available_cpus if whole_worker else float(self.cpus), + "MEMORY": ( + available_memory + if whole_worker or self.memory_bytes is None + else float(self.memory_bytes) + ), + } + matches = { + gpus for cpus, memory, gpus in capacities + if cpus >= requested["CPU"] and memory >= requested["MEMORY"] and gpus >= self.gpus + } + if not matches: + raise ProjectError( + f"task needs {requested['CPU']:g} CPUs and " + f"{requested['MEMORY'] / 1024**3:g} GiB and {self.gpus} GPUs on one worker; " + "no worker in this cluster can satisfy that request" + ) + if self.gpus or whole_worker: + if len(matches) != 1: + raise ProjectError( + "GPU execution requires matching workers with identical GPU budgets" + ) + if gpus := next(iter(matches)): + requested["GPU"] = gpus + return requested + + +def _memory(value: object) -> int: + if not isinstance(value, str) or not ( + match := re.fullmatch(r"([0-9]+(?:\.[0-9]+)?)\s*([KMGTPE]i?B?|kB?|B)", value) + ): + raise ProjectError("recipe memory must include units, e.g. 512Mi, 16Gi, or 8GB") + unit = match[2].lower() + exponent = 0 if unit == "b" else "kmgtpe".index(unit[0]) + 1 + factor: int = (1024 if "i" in unit else 1000) ** exponent + try: + return whole_bytes(match[1], factor) + except ValueError as exc: + raise ProjectError(f"recipe {exc}") from exc diff --git a/src/lightcone/engine/gpu.py b/src/lightcone/engine/gpu.py new file mode 100644 index 00000000..ee280fda --- /dev/null +++ b/src/lightcone/engine/gpu.py @@ -0,0 +1,141 @@ +"""Discover CUDA-visible GPU identities without initializing CUDA in a worker. + +CUDA owns device ordering and native visibility masks. A short isolated process +asks the driver for UUIDs; interpreting numeric masks ourselves would confuse +Slurm's allocation-local indices with host device numbers. +""" + +from __future__ import annotations + +import json +import os +import re +import subprocess +import sys +from pathlib import Path +from typing import NamedTuple + +_UUID = re.compile(r"GPU-[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}") +_DEVICE_ROOT = Path("/dev") + + +class Device(NamedTuple): + """A native CUDA identity and its model name, with punctuation made CLI-safe.""" + + uuid: str + name: str + + +def inventory() -> tuple[Device, ...]: + """Return devices visible under this process's native CUDA mask. + + Missing CUDA drivers or an empty visible set return no devices. Broken + drivers and unsupported partitioned GPUs raise instead of inventing capacity. + + Returns: + Native UUIDs and model names, in CUDA visibility order. + + Raises: + ProjectError: CUDA discovery failed or returned an untrustworthy inventory. + """ + from lightcone.engine.project import ProjectError + + if os.environ.get("CUDA_VISIBLE_DEVICES") == "" or sys.platform != "linux": + return () + try: + result = subprocess.run( + [sys.executable, "-I", str(Path(__file__))], + stdin=subprocess.DEVNULL, capture_output=True, text=True, timeout=15, + ) + if result.returncode: + raise ValueError(result.stderr.strip()[:1024] or "CUDA probe exited unsuccessfully") + devices = json.loads(result.stdout) + if not isinstance(devices, list): + raise ValueError("CUDA probe returned invalid GPU identities") + result_devices = [] + for item in devices: + if ( + not isinstance(item, dict) or item.keys() != {"uuid", "name"} + or not isinstance(item["uuid"], str) or not _UUID.fullmatch(item["uuid"]) + or not isinstance(item["name"], str) + or not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]*", item["name"]) + ): + raise ValueError("CUDA probe returned invalid GPU identities") + result_devices.append(Device(**item)) + if len({device.uuid for device in result_devices}) != len(result_devices): + raise ValueError("CUDA probe returned duplicate GPU identities") + return tuple(result_devices) + except (OSError, subprocess.SubprocessError, ValueError) as exc: + raise ProjectError(f"cannot discover visible NVIDIA GPUs: {exc}") from exc + + +def visible_devices() -> tuple[str, ...]: + """Return native CUDA UUIDs suitable for a subprocess visibility mask.""" + return tuple(device.uuid for device in inventory()) + + +def device_paths() -> tuple[Path, ...]: + """Return existing NVIDIA device nodes needed by direct CUDA commands. + + These grants retain native OS/cgroup permissions. CUDA visibility controls + cooperative device selection; it is not an additional device-isolation layer. + """ + candidates = [ + *(_DEVICE_ROOT / name for name in ("nvidiactl", "nvidia-uvm", "nvidia-uvm-tools")), + *_DEVICE_ROOT.glob("nvidia[0-9]*"), + *(_DEVICE_ROOT / "nvidia-caps").glob("*"), + ] + return tuple(sorted(path for path in candidates if path.is_char_device())) + + +def _probe() -> list[dict[str, str]]: + # Keep this child stdlib-only: CUDA initialization never enters the CLI, + # allocation owner, or reusable Dask worker. No context or memory is allocated. + import ctypes + from uuid import UUID + + try: + driver = ctypes.CDLL("libcuda.so.1") + except OSError: + return [] + + def check(code: int) -> None: + if code: + raise RuntimeError(f"CUDA driver returned error {code}") + + driver.cuInit.argtypes = [ctypes.c_uint] + status = driver.cuInit(0) + if status == 100: # CUDA_ERROR_NO_DEVICE + return [] + check(status) + driver.cuDeviceGetCount.argtypes = [ctypes.POINTER(ctypes.c_int)] + driver.cuDeviceGet.argtypes = [ctypes.POINTER(ctypes.c_int), ctypes.c_int] + driver.cuDeviceGetUuid.argtypes = [ctypes.c_void_p, ctypes.c_int] + driver.cuDeviceGetUuid_v2.argtypes = [ctypes.c_void_p, ctypes.c_int] + driver.cuDeviceGetName.argtypes = [ctypes.c_void_p, ctypes.c_int, ctypes.c_int] + count = ctypes.c_int() + check(driver.cuDeviceGetCount(ctypes.byref(count))) + devices = [] + for ordinal in range(count.value): + device = ctypes.c_int() + check(driver.cuDeviceGet(ctypes.byref(device), ordinal)) + physical, instance = ctypes.create_string_buffer(16), ctypes.create_string_buffer(16) + check(driver.cuDeviceGetUuid(physical, device)) + check(driver.cuDeviceGetUuid_v2(instance, device)) + if physical.raw != instance.raw: + raise RuntimeError("partitioned (MIG) GPUs are not supported; request whole GPUs") + name = ctypes.create_string_buffer(256) + check(driver.cuDeviceGetName(name, len(name), device)) + devices.append({ + "uuid": f"GPU-{UUID(bytes=physical.raw)}", + "name": re.sub(r"[^A-Za-z0-9_.-]+", "-", name.value.decode("ascii")).strip("-"), + }) + return devices + + +if __name__ == "__main__": + try: + print(json.dumps(_probe())) + except Exception as error: + print(str(error), file=sys.stderr) + raise SystemExit(1) from error diff --git a/src/lightcone/engine/materialize.py b/src/lightcone/engine/materialize.py index e9791540..8b956eac 100644 --- a/src/lightcone/engine/materialize.py +++ b/src/lightcone/engine/materialize.py @@ -34,7 +34,7 @@ import functools import json import re -from collections.abc import Iterator, Sequence +from collections.abc import Iterable, Iterator, Sequence from contextlib import contextmanager from dataclasses import asdict, dataclass, field, replace from pathlib import Path @@ -42,6 +42,7 @@ from uuid import uuid4 from lightcone.engine import assets, container, dataset, identity, plan, project, worker +from lightcone.engine.execution_resources import TaskResources from lightcone.engine.plan import Graph, Key, Task from lightcone.engine.project import ProjectError @@ -516,6 +517,7 @@ def materialize( _converge_crate(root, report, full, dsid) return report with cluster_for_run(cluster_id) as scheduler: + requirements = scheduler.validate(graph.tasks.values()) _fetch_inputs(root, graph, report) # Materialize is one of the two verbs allowed to build the image (the # other is `lc build`); the probe and the rerun entry point only find @@ -576,6 +578,7 @@ def materialize( foreign[key], *[pending[dep] for dep in task.depends_on], key=_name(key), + resources=requirements[key], ) from lightcone.engine.compute import UNSTOPPED @@ -646,13 +649,18 @@ class Scheduler(Protocol): they land, keeping the commit logic independent of the provider. """ - def submit(self, fn: Any, *args: Any, key: str) -> Any: + def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: + """Validate all requests and return their Dask resource reservations.""" + ... + + def submit(self, fn: Any, *args: Any, key: str, resources: dict[str, float]) -> Any: """Schedule a call. Args: fn: The function to run. *args: Its arguments, upstream handles included. key: A display name for the task. + resources: Validated reservations for this task. Returns: A handle to pass to dependents. @@ -678,14 +686,25 @@ class _Dask: client: Any invocation: str output: Forwarder + workers: dict[str, Any] + + def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: + """Require each selected task to fit a worker before any task starts.""" + requests = {} + for task in tasks: + try: + requests[task.key] = TaskResources.parse(task.resources).requirements(self.workers) + except ProjectError as exc: + raise ProjectError(f"{_name(task.key)}: {exc}") from exc + return requests - def submit(self, fn: Any, *args: Any, key: str) -> Any: + def submit(self, fn: Any, *args: Any, key: str, resources: dict[str, float]) -> Any: """Submit an ordinary Dask task with a unique key and forwarded output.""" from lightcone.engine.compute.output import call return self.client.submit( call, fn, self.output.topic, key, *args, - key=f"lc-{self.invocation}-{key}", pure=False, + key=f"lc-{self.invocation}-{key}", pure=False, resources=resources, ) def completed(self, handles: list[Any]) -> Iterator[worker.TaskResult]: @@ -727,7 +746,7 @@ def cluster_for_run(cluster_id: str) -> Iterator[Scheduler]: with compute.connect(cluster_id) as client: invocation = uuid4().hex with forwarding(client, stdout="stderr") as output: - yield _Dask(client, invocation, output) + yield _Dask(client, invocation, output, client.scheduler_info()["workers"]) def _fetch_inputs(root: Path, graph: Graph, report: MaterializeReport) -> None: diff --git a/src/lightcone/engine/plan.py b/src/lightcone/engine/plan.py index 46067016..513f410c 100644 --- a/src/lightcone/engine/plan.py +++ b/src/lightcone/engine/plan.py @@ -23,9 +23,10 @@ from __future__ import annotations -from dataclasses import dataclass +from dataclasses import dataclass, field from graphlib import CycleError, TopologicalSorter from pathlib import Path +from typing import Any from lightcone.engine import assets, identity from lightcone.engine.project import SPEC_FILENAME, ProjectError @@ -51,6 +52,8 @@ class Task: produced_by: dict[str, Key] decisions: dict[str, str] definition_version: str + #: ASTRA's declaration; executor support is checked only when executing. + resources: dict[str, Any] = field(default_factory=dict) @property def manifest_path(self) -> Path: @@ -331,6 +334,7 @@ def file_of(out: object) -> Path: ) except ValueError as e: raise ProjectError(f"output `{out.id}`: {e}") from e + resources = dict((out.definition.get("recipe") or {}).get("resources") or {}) tasks.append( Task( @@ -344,6 +348,7 @@ def file_of(out: object) -> Path: definition_version=identity.definition_version( recipe=recipe, decisions=out.decisions, fmt=str(out.format) ), + resources=resources, ) ) return tasks diff --git a/src/lightcone/engine/run.py b/src/lightcone/engine/run.py index d54b7ed2..21f4b688 100644 --- a/src/lightcone/engine/run.py +++ b/src/lightcone/engine/run.py @@ -20,7 +20,8 @@ from typing import Any from uuid import uuid4 -from lightcone.engine import container, sandbox +from lightcone.engine import container, gpu, sandbox +from lightcone.engine.execution_resources import TaskResources from lightcone.engine.project import ( SPEC_FILENAME, ProjectError, @@ -51,13 +52,17 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. require_uv() paths = input_paths(project, read_spec(project)) with compute.connect(cluster_id) as client: + resources = TaskResources().requirements( + client.scheduler_info()["workers"], whole_worker=True, + ) runtime = container.runtime_for_run(project, build=False) notes = [f"uv: {warning}" for warning in container.converge(runtime)] invocation = uuid4().hex with forwarding(client) as output: future = client.submit( call, _probe, output.topic, "probe", runtime, paths, tuple(command), - key=f"lc-{invocation}-probe", pure=False, + int(resources.get("GPU", 0)), + key=f"lc-{invocation}-probe", pure=False, resources=resources, ) try: outcome: sandbox.Outcome = future.result() @@ -76,10 +81,17 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. def _probe( runtime: container.Runtime, paths: list[Path], command: tuple[str, ...], + gpu_count: int, *, output: Callable[[str, bytes], None], ) -> sandbox.Outcome: """Execute the prepared probe; the driver alone converges its environment.""" - built = container.policy_for(runtime, paths) + gpu_devices = gpu.visible_devices()[:gpu_count] if gpu_count else () + if len(gpu_devices) < gpu_count: + raise ProjectError( + f"probe reserved {gpu_count} GPUs but this worker can access " + f"only {len(gpu_devices)} CUDA devices" + ) + built = container.policy_for(runtime, paths, gpu_devices=gpu_devices) with sandbox.scope(built) as policy: outcome = sandbox.run( container.backend(runtime), policy, command, cwd=runtime.root, diff --git a/src/lightcone/engine/sandbox/oci.py b/src/lightcone/engine/sandbox/oci.py index 0acbf4e9..1debe632 100644 --- a/src/lightcone/engine/sandbox/oci.py +++ b/src/lightcone/engine/sandbox/oci.py @@ -81,7 +81,21 @@ def wrap(self, policy: Policy, argv: Sequence[str]) -> list[str]: # host-layout collision `_write_roots` documents for direct mode. mounts = [f"--volume={path.resolve()}:{path}:ro" for path in policy.read] mounts += [f"--volume={path.resolve()}:{path}:rw" for path in policy.write] - overlay = [f"--env={k}={v}" for k, v in sorted(policy.env.items())] + devices = policy.env.get("CUDA_VISIBLE_DEVICES", "") + environment = dict(policy.env) + if not devices: + # CUDA images may default to all devices under an NVIDIA runtime. + environment["NVIDIA_VISIBLE_DEVICES"] = "void" + overlay = [f"--env={k}={v}" for k, v in sorted(environment.items())] + gpu_flags = [] + if devices: + if self.runtime == "podman-hpc": + gpu_flags = ["--gpu"] + elif self.runtime == "docker": + # Docker parses this argument as CSV, including its quotes. + gpu_flags = ["--gpus", f'"device={devices}"'] + else: + gpu_flags = [f"--device=nvidia.com/gpu={device}" for device in devices.split(",")] return [ self.runtime, "run", "--rm", "--entrypoint", "", @@ -97,6 +111,7 @@ def wrap(self, policy: Policy, argv: Sequence[str]) -> list[str]: # would rewrite the user's own file contexts on disk. "--security-opt", "label=disable", *self.user_flags, + *gpu_flags, *mounts, "--tmpfs", "/tmp:rw,exec", "--shm-size", "1g", diff --git a/src/lightcone/engine/sandbox/policy.py b/src/lightcone/engine/sandbox/policy.py index 0c91ba68..71b5a94b 100644 --- a/src/lightcone/engine/sandbox/policy.py +++ b/src/lightcone/engine/sandbox/policy.py @@ -32,6 +32,7 @@ from fnmatch import fnmatch from pathlib import Path +from lightcone.engine import gpu from lightcone.engine.sandbox.model import Policy #: The utility tier of the exec allowlist. A maintained policy @@ -182,6 +183,7 @@ def exec_policy( env_dir: Path | None = None, containerized: bool = False, write_dir: Path | None = None, + gpu_devices: Sequence[str] = (), ) -> Policy: """Build what a sandboxed command may touch. @@ -211,6 +213,7 @@ def exec_policy( directory holding its output file, shared with the siblings declared beside it. Absent for a probe, which has no analysis node and gets the project's own ``results/`` whole. + gpu_devices: CUDA device UUIDs allocated to this command; empty hides GPUs. Returns: The policy. The in-tree write scope is granted only if it exists — @@ -234,6 +237,8 @@ def exec_policy( (tmp_home / sub).mkdir(parents=True, exist_ok=True) in_tree_write = write_dir if write_dir is not None else project / "results" + overlay = home_overlay(tmp_home, env_dir, containerized=containerized) + overlay["CUDA_VISIBLE_DEVICES"] = ",".join(gpu_devices) if containerized: # Declared spellings, not realpaths — the one shape that keeps # its paths unresolved. These become mount *destinations*, and a @@ -246,14 +251,15 @@ def exec_policy( write=_declared([tmp_home, in_tree_write]), execute=(), tmp_home=tmp_home, - env=home_overlay(tmp_home, env_dir, containerized=True), + env=overlay, ) python = _venv_python(env_dir) # EXECUTE on the interpreter *file*; READ on the install root beside # it, for the stdlib. See :func:`_venv_python` and :func:`_stdlib_root`. stdlib = _stdlib_root(python) - write = _existing([tmp_home, in_tree_write, *_write_roots(project)]) + devices = gpu.device_paths() if gpu_devices else () + write = _existing([tmp_home, in_tree_write, *_write_roots(project), *devices]) read = _existing([project, *read_paths, *stdlib, *(Path(p) for p in _OS_READ_BASELINE)]) return Policy( @@ -261,7 +267,7 @@ def exec_policy( write=write, execute=_existing(_exec_set(env_dir, python)), tmp_home=tmp_home, - env=home_overlay(tmp_home, env_dir), + env=overlay, ) diff --git a/src/lightcone/engine/units.py b/src/lightcone/engine/units.py new file mode 100644 index 00000000..2d47130a --- /dev/null +++ b/src/lightcone/engine/units.py @@ -0,0 +1,45 @@ +"""Exact byte quantities and explicit walltime units used by execution and compute.""" + +from __future__ import annotations + +import re +from decimal import Decimal, InvalidOperation + + +def whole_bytes(amount: str, unit_bytes: int) -> int: + """Scale a decimal memory amount into positive bytes without rounding. + + Args: + amount: A decimal quantity in the caller's units. + unit_bytes: The number of bytes represented by one unit. + + Raises: + ValueError: The quantity is not positive or exactly representable in bytes. + """ + try: + numerator, denominator = Decimal(amount).as_integer_ratio() + except (InvalidOperation, ValueError, OverflowError) as exc: + raise ValueError("memory must be positive and exactly representable in bytes") from exc + result, remainder = divmod(numerator * unit_bytes, denominator) + if result <= 0 or remainder: + raise ValueError("memory must be positive and exactly representable in bytes") + return result + + +def duration_seconds(value: object) -> int: + """Parse a positive duration such as ``30m``, ``1h30m``, or ``45s``. + + Args: + value: Ordered day, hour, minute, and second components. + + Raises: + ValueError: The duration is malformed, empty, or zero. + """ + if not isinstance(value, str) or not ( + match := re.fullmatch(r"(?:(\d+)d)?(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?", value) + ): + raise ValueError("duration must use explicit units, e.g. 30m, 1h30m, or 45s") + seconds = sum(int(part or 0) * unit for part, unit in zip(match.groups(), (86400, 3600, 60, 1))) + if seconds <= 0: + raise ValueError("duration must be positive") + return seconds diff --git a/src/lightcone/engine/worker.py b/src/lightcone/engine/worker.py index af1be88a..6189f95d 100644 --- a/src/lightcone/engine/worker.py +++ b/src/lightcone/engine/worker.py @@ -35,7 +35,8 @@ from pathlib import Path from typing import Literal -from lightcone.engine import assets, container, dataset, identity, plan, project, sandbox +from lightcone.engine import assets, container, dataset, gpu, identity, plan, project, sandbox +from lightcone.engine.execution_resources import TaskResources from lightcone.engine.plan import Key, Task from lightcone.engine.project import ( ProjectError, @@ -227,6 +228,13 @@ def execute( ``ok`` with the output's ``data_version``, or ``failed``. Commits nothing and never touches git beyond reading HEAD. """ + resources = TaskResources.parse(task.resources) + gpu_devices = gpu.visible_devices()[:resources.gpus] if resources.gpus else () + if len(gpu_devices) < resources.gpus: + raise ProjectError( + f"recipe requests {resources.gpus} GPUs but this worker can access " + f"only {len(gpu_devices)} CUDA devices" + ) if moved := _gate(root, context.env_version): return TaskResult(task.key, "failed", reason=moved) @@ -246,7 +254,7 @@ def execute( read_paths = [p for p in task.inputs.values() if p.exists()] policy = container.policy_for( - context.runtime, read_paths, write_dir=task.output_path.parent + context.runtime, read_paths, write_dir=task.output_path.parent, gpu_devices=gpu_devices, ) started_at = _now() with sandbox.scope(policy): diff --git a/tests/conftest.py b/tests/conftest.py index 2a4d164a..260e2cc8 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -5,7 +5,7 @@ import shutil import subprocess import textwrap -from collections.abc import Callable, Iterator +from collections.abc import Callable, Iterable, Iterator from contextlib import contextmanager from pathlib import Path from unittest.mock import MagicMock @@ -15,6 +15,7 @@ from lightcone.engine import dataset, project, templates from lightcone.engine.compute.model import Identity +from lightcone.engine.plan import Key, Task from lightcone.engine.project import _run as _real_run CLUSTER_ID = Identity( @@ -151,7 +152,13 @@ class _Inline: are the upstream results themselves, exactly what the worker expects. """ - def submit(self, fn: Callable[..., object], *args: object, key: str) -> object: + def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: + """Run fixture tasks without a finite cluster resource envelope.""" + return {task.key: {} for task in tasks} + + def submit( + self, fn: Callable[..., object], *args: object, key: str, resources: dict[str, float], + ) -> object: return fn(*args) def completed(self, handles: list[object]) -> Iterator[object]: @@ -180,7 +187,8 @@ def cluster_id(monkeypatch: pytest.MonkeyPatch) -> Iterator[str]: from lightcone.engine import compute with LocalCluster( - n_workers=1, threads_per_worker=2, processes=False, dashboard_address=None + n_workers=1, threads_per_worker=2, processes=False, dashboard_address=None, + resources={"CPU": 2, "MEMORY": 1024**3}, ) as cluster: @contextmanager def connect(value: str) -> Iterator[Client]: diff --git a/tests/test_compute.py b/tests/test_compute.py index 7e56ab99..351b9738 100644 --- a/tests/test_compute.py +++ b/tests/test_compute.py @@ -18,6 +18,7 @@ from lightcone.engine.compute.catalog import Catalog from lightcone.engine.compute.model import ( GIB, + Accelerator, ComputeError, Connection, Identity, @@ -29,9 +30,11 @@ Startup, TimeLimits, UnavailableOfferError, + duration, memory_bytes, validate_name, ) +from lightcone.engine.gpu import Device NAMESPACE = "5a9d058c-7c6e-4e2a-919b-786f1148536c" IDENTITY = Identity(namespace=NAMESPACE, native_id="1234", token="abc") @@ -39,6 +42,7 @@ @pytest.fixture def default_home(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + monkeypatch.setattr("lightcone.engine.gpu.inventory", lambda: ()) expanduser = Path.expanduser def expand(path: Path) -> Path: @@ -364,6 +368,176 @@ def test_memory_conversion_does_not_round_fractional_bytes() -> None: memory_bytes("0.000000000931322574615478515625000000000000000001") +@pytest.mark.parametrize("memory", ["8", "8GB", "8gb", "8192MB", "8GiB", "0.0078125TB"]) +def test_compute_memory_uses_skypilot_binary_units(memory: str) -> None: + assert Request.parse("1", memory + "+").memory_bytes == 8 * GIB + assert Request.parse("1", memory + "+").min_memory + assert Resources.model_validate({"cpus": 1, "memory": memory}).memory_bytes == 8 * GIB + + +def test_recipe_and_compute_memory_keep_their_specification_units() -> None: + from lightcone.engine.execution_resources import TaskResources + + assert Request.parse("1", "8GB").memory_bytes == 8 * GIB + assert TaskResources.parse({"memory": "8GB"}).memory_bytes == 8_000_000_000 + + +def test_compute_memory_rejects_unit_suffix_without_a_size_prefix() -> None: + with pytest.raises(ComputeError, match="memory"): + Request.parse("1", "1IB") + + +@pytest.mark.parametrize("value", ["A100:4", {"A100": 4}]) +def test_accelerator_sky_notations_share_one_model(value: object) -> None: + resource = Resources.model_validate({"cpus": 1, "memory": 1, "accelerators": value}) + assert resource.accelerators == Accelerator(name="A100", count=4) + assert resource.as_dict()["accelerators"] == {"A100": 4} + + +@pytest.mark.parametrize("value", [{"A100": 1, "H100": 1}, ["A100:1", "H100:1"], {"A100:1"}]) +def test_accelerator_alternatives_are_not_silently_treated_as_capacity(value: object) -> None: + with pytest.raises(ValidationError): + Resources.model_validate({"cpus": 1, "memory": 1, "accelerators": value}) + + +@pytest.mark.parametrize("count", [-1, 0, True, 0.5, "1", None]) +def test_accelerator_envelopes_require_positive_integer_counts(count: object) -> None: + with pytest.raises(ValidationError): + Resources.model_validate({"cpus": 1, "memory": 1, "accelerators": {"A100": count}}) + with pytest.raises(ValidationError): + Request.model_validate({"cpus": 1, "memory_bytes": GIB, "accelerators": {"A100": count}}) + + +@pytest.mark.parametrize("gpus", ["-1", "1++", "", "A100:0.5", "A100:2+"]) +def test_cli_gpu_requests_reject_invalid_accelerator_specifications(gpus: str) -> None: + with pytest.raises(ComputeError, match="accelerators"): + Request.parse("1", "1", gpus=gpus) + + +def test_accelerator_types_and_counts_survive_native_and_request_roundtrips() -> None: + resource = Resources.from_bytes(cpus=8, memory_bytes=16 * GIB, gpus=4, accelerator_name="A100") + assert resource.as_dict() == {"cpus": 8, "memory": 16, "accelerators": {"A100": 4}} + assert Resources.model_validate_json(resource.model_dump_json(by_alias=True)) == resource + assert resource.replace(cpus=4).accelerators == Accelerator(name="A100", count=4) + request = Request.parse("8", "16", gpus="A100:2") + assert request.gpus == 2 and request.accelerator_name == "A100" + assert request.as_dict()["resources"]["accelerators"] == {"A100": 2} + assert Request.parse("8", "16", gpus="A100").gpus == 1 + assert Request.parse("8", "16").gpus == 0 + + +def test_numeric_native_accelerator_names_remain_types() -> None: + resource = Resources.from_bytes(cpus=1, memory_bytes=GIB, gpus=2, accelerator_name="4090") + request = Request.parse("1", "1", gpus="4090:2") + assert request.accelerators is not None + assert request.accelerators.matches(resource.accelerators) + assert Request.parse("1", "1", gpus="4090").accelerators == Accelerator(name="4090", count=1) + + +def test_accelerator_selection_honors_type_and_exact_count( + catalog: Path, provider: MagicMock, +) -> None: + data = yaml.safe_load(catalog.read_text()) + cpu = data["offers"][0] + data["offers"].insert(0, { + **cpu, "name": "gpu", "resources": {**cpu["resources"], "accelerators": "A100:4"}, + }) + catalog.write_text(yaml.safe_dump(data)) + service = compute.Compute() + assert service.plan(Request.parse("4", "8")).offer.name == "quick" + assert service.plan(Request.parse("4", "8", gpus="a100:4")).offer.name == "gpu" + assert service.plan(Request.parse("4", "8", gpus="GPU:4")).offer.name == "gpu" + for gpus in ("A100", "H100:4"): + with pytest.raises(ComputeError, match="no configured offer"): + service.plan(Request.parse("4", "8", gpus=gpus)) + data["offers"][0]["resources"]["accelerators"] = "GPU:4" + catalog.write_text(yaml.safe_dump(data)) + with pytest.raises(ComputeError, match="no configured offer"): + compute.Compute().plan(Request.parse("4", "8", gpus="A100:4")) + + +def test_builtin_gpu_offer_is_optional_and_does_not_replace_cpu_offer( + default_home: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr("lightcone.engine.gpu.inventory", lambda: ( + Device("GPU-one", "A100"), Device("GPU-two", "A100"), + )) + loaded = Catalog.load() + assert [(offer.name, offer.resources.gpus) for offer in loaded.offers] == [ + ("local", 0), ("local-gpu", 2), + ] + assert loaded.offers[1].connection == loaded.offers[0].connection + assert loaded.offers[1].resources.accelerator_name == "A100" + assert not list(default_home.iterdir()) + + +def test_failed_optional_gpu_discovery_keeps_builtin_cpu_offer( + default_home: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + def failed() -> tuple[Device, ...]: + raise ComputeError("CUDA driver could not enumerate visible devices") + + monkeypatch.setattr("lightcone.engine.gpu.inventory", failed) + assert [(offer.name, offer.resources.gpus) for offer in Catalog.load().offers] == [("local", 0)] + + +def test_builtin_mixed_gpu_inventory_exposes_each_model_separately( + default_home: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr("lightcone.engine.gpu.inventory", lambda: ( + Device("GPU-one", "H100"), Device("GPU-two", "A100"), Device("GPU-three", "H100"), + )) + loaded = Catalog.load() + assert [(offer.name, offer.resources.as_dict()["accelerators"]) for offer in loaded.offers] == [ + ("local", None), ("local-gpu-1", {"A100": 1}), ("local-gpu-2", {"H100": 2}), + ] + + +def test_configured_catalogs_do_not_probe_local_gpus( + catalog: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + def unexpected() -> tuple[Device, ...]: + pytest.fail("configured catalogs must not discover ambient local GPUs") + + monkeypatch.setattr("lightcone.engine.gpu.inventory", unexpected) + assert Catalog.load().offers + + +def test_cli_gpu_request_and_resource_output(catalog: Path, provider: MagicMock) -> None: + data = yaml.safe_load(catalog.read_text()) + data["offers"][0]["resources"]["accelerators"] = {"A100": 2} + catalog.write_text(yaml.safe_dump(data)) + runner = CliRunner() + result = runner.invoke(main, [ + "compute", "launch", "--cpus", "4", "--memory", "8", "--gpus", "A100:2", + "--dry-run", "--json", + ]) + assert result.exit_code == 0, result.output + plan = json.loads(result.output)["plan"] + assert plan["request"]["resources"]["accelerators"] == {"A100": 2} + assert plan["resources"]["accelerators"] == {"A100": 2} + resources = runner.invoke(main, ["compute", "resources", "--json"]) + assert json.loads(resources.output)["units"]["accelerators"] == "type and count per node" + rendered = runner.invoke(main, ["compute", "resources"]).output + assert "GPUS" in rendered and "A100:2" in rendered + + +@pytest.mark.parametrize( + ("value", "seconds"), + [("1h30m", 5400), ("45s", 45), ("2d3h4m5s", 183845)], +) +def test_allocation_durations_accept_compound_units(value: str, seconds: int) -> None: + assert duration(value) == seconds + assert TimeLimits(default=value, max="3d").default_seconds == seconds + assert Request.parse("1", "1", time=value).seconds == seconds + + +@pytest.mark.parametrize("value", ["", "0s", "1.5h", "30m1h", "1h30", 60, True]) +def test_allocation_duration_refuses_ambiguous_or_zero_values(value: object) -> None: + with pytest.raises(ComputeError): + duration(value) + + @pytest.mark.parametrize("size", [1, GIB // 2, 8 * GIB, 2**80 + 1]) def test_resource_units_survive_construction_serialization_and_updates(size: int) -> None: whole, fraction = divmod(size, GIB) @@ -410,7 +584,9 @@ def test_catalog_uses_the_public_models_and_roundtrips_without_an_adapter(catalo assert type(instance) is model assert "name" not in loaded.connections["test"].model_dump() dumped = loaded.model_dump(by_alias=True) - assert dumped["offers"][0]["resources"] == {"cpus": 4, "memory": Decimal(8)} + assert dumped["offers"][0]["resources"] == { + "cpus": 4, "memory": Decimal(8), "accelerators": None, + } assert dumped["offers"][0]["time"] == {"default": "30m", "max": "2h"} assert Catalog.model_validate(dumped) == loaded assert Catalog.model_validate_json(loaded.model_dump_json(by_alias=True)) == loaded @@ -717,14 +893,14 @@ def test_cli_resources_dry_run_launch_down(catalog: Path, provider: MagicMock) - assert result.exit_code == 0, result.output resources = json.loads(result.output) assert [item["name"] for item in resources["offers"]] == ["quick", "large"] - assert resources["offers"][0]["resources"] == {"cpus": 4, "memory": 8} + assert resources["offers"][0]["resources"] == {"cpus": 4, "memory": 8, "accelerators": None} args = ["compute", "launch", "--cpus", "4", "--memory", "8", "--json"] result = runner.invoke(main, [*args, "--dry-run"]) assert result.exit_code == 0, result.output plan = json.loads(result.output)["plan"] assert plan["offer"] == "quick" assert plan["connection"] == "test" - assert plan["resources"] == {"cpus": 4, "memory": 8} + assert plan["resources"] == {"cpus": 4, "memory": 8, "accelerators": None} assert plan["time_seconds"] == 1800 assert plan["startup"] == "fast" assert "memory_gib" not in result.output @@ -737,6 +913,16 @@ def test_cli_resources_dry_run_launch_down(catalog: Path, provider: MagicMock) - provider.terminate.assert_called_once_with(IDENTITY) +def test_cli_resources_preserves_seconds(catalog: Path) -> None: + data = yaml.safe_load(catalog.read_text()) + data["offers"][0]["time"] = {"default": "45s", "max": "1m30s"} + catalog.write_text(yaml.safe_dump(data)) + result = CliRunner().invoke(main, ["compute", "resources"]) + assert result.exit_code == 0, result.output + assert "45s" in result.output + assert "1m30s" in result.output + + def test_cli_launch_name_output_can_be_captured_without_json( catalog: Path, provider: MagicMock, ) -> None: diff --git a/tests/test_compute_local.py b/tests/test_compute_local.py index 0da9d92a..2dea6470 100644 --- a/tests/test_compute_local.py +++ b/tests/test_compute_local.py @@ -15,11 +15,15 @@ from pathlib import Path from types import SimpleNamespace from typing import Any +from unittest.mock import MagicMock from uuid import uuid4 import psutil import pytest +from lightcone.engine import gpu +from lightcone.engine.compute import Compute, local, local_runtime +from lightcone.engine.compute.catalog import Catalog from lightcone.engine.compute.local import LocalProvider from lightcone.engine.compute.model import ( ComputeError, @@ -37,6 +41,7 @@ read_private_json, write_private_json, ) +from lightcone.engine.project import ProjectError @pytest.fixture @@ -807,3 +812,151 @@ def test_local_plan_does_not_infer_policy_from_login_hostname_or_slurm_environme plan = provider.plan(offer, Request(cpus=1, memory_bytes=512 * 1024**2)) assert plan.resources == offer.resources assert not provider.root.exists() + + +@pytest.mark.parametrize("gpus", [0, 1]) +def test_local_gpu_plan_freezes_devices_and_publishes_the_selected_envelope( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, gpus: int, +) -> None: + devices = (f"GPU-{uuid4()}", f"GPU-{uuid4()}") + visible = MagicMock(return_value=tuple(gpu.Device(item, "NVIDIA-A100") for item in devices)) + monkeypatch.setattr(gpu, "inventory", visible) + offer = Offer( + name="gpu", connection="workstation", + resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=gpus), + max_nodes=1, time=TimeLimits(default="1m", max="1m"), + ) + plan = provider.plan(offer, Request.parse("1", "0.5", gpus=f"GPU:{gpus}" if gpus else "0")) + assert plan.details["gpu_devices"] == devices[:gpus] + assert not provider.root.exists() + monkeypatch.setattr(local, "_boot_identity", lambda: str(uuid4())) + popen = MagicMock(return_value=SimpleNamespace(pid=12345)) + monkeypatch.setattr(local.subprocess, "Popen", popen) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "an-ambient-mask") + + identity = provider.launch(plan) + + assert popen.call_args.kwargs["env"]["CUDA_VISIBLE_DEVICES"] == ",".join(devices[:gpus]) + assert os.environ["CUDA_VISIBLE_DEVICES"] == "an-ambient-mask" + record = read_private_json(provider.root / identity.token / "identity.json") + assert record["gpus"] == gpus + monkeypatch.setattr(provider, "_process", lambda *_: None) + snapshot = provider.inspect(identity) + assert snapshot.resources is not None and snapshot.resources.gpus == gpus + assert snapshot.resources.accelerator_name == ("NVIDIA-A100" if gpus else None) + assert visible.call_count == (2 if gpus else 0) + + +def test_local_gpu_capacity_is_checked_at_plan_and_again_before_spawn( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, +) -> None: + device = f"GPU-{uuid4()}" + offer = Offer( + name="gpu", connection="workstation", + resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=1), + max_nodes=1, time=TimeLimits(default="1m", max="1m"), + ) + request = Request.parse("1", "0.5", gpus="GPU:1") + monkeypatch.setattr(gpu, "inventory", lambda: ()) + with pytest.raises(ComputeError, match="visible CUDA capacity for GPU"): + provider.plan(offer, request) + monkeypatch.setattr(gpu, "inventory", lambda: (gpu.Device(device, "NVIDIA-A100"),)) + plan = provider.plan(offer, request) + monkeypatch.setattr(gpu, "inventory", lambda: (gpu.Device(f"GPU-{uuid4()}", "NVIDIA-A100"),)) + with pytest.raises(ComputeError, match="no longer visible"): + provider.launch(plan) + assert not provider.root.exists() + + +def test_unavailable_local_gpu_driver_does_not_hide_a_later_slurm_offer( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(gpu, "inventory", MagicMock(side_effect=ProjectError("CUDA driver failed"))) + local_offer = Offer( + name="local-gpu", connection="workstation", + resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=1), + max_nodes=1, time=TimeLimits(default="1m", max="1m"), + ) + service = Compute.__new__(Compute) + service.catalog = Catalog( + version=1, + connections={ + "workstation": provider.connection, + "hpc": Connection(namespace=str(uuid4()), provider="slurm"), + }, + offers=[local_offer, local_offer.replace(name="batch-gpu", connection="hpc")], + ) + plan = service.plan(Request.parse("1", "0.5", gpus="GPU:1")) + assert plan.offer.name == "batch-gpu" + assert not provider.root.exists() + + +@pytest.mark.parametrize("requested,count,expected", [ + ("nvidia-a100", 1, (1,)), ("GPU", 2, (0, 1)), ("A100", 1, ()), +]) +def test_local_named_accelerators_match_native_models_without_guessing_aliases( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, + requested: str, count: int, expected: tuple[int, ...], +) -> None: + devices = ( + gpu.Device(f"GPU-{uuid4()}", "NVIDIA-H100"), + gpu.Device(f"GPU-{uuid4()}", "NVIDIA-A100"), + gpu.Device(f"GPU-{uuid4()}", "NVIDIA-A100"), + ) + monkeypatch.setattr(gpu, "inventory", lambda: devices) + offer = Offer( + name="gpu", connection="workstation", + resources=Resources.from_bytes( + cpus=1, memory_bytes=512 * 1024**2, gpus=count, accelerator_name=requested, + ), + max_nodes=1, time=TimeLimits(default="1m", max="1m"), + ) + request = Request.parse("1", "0.5", gpus=f"{requested}:{count}") + if not expected: + with pytest.raises(ComputeError, match="visible CUDA capacity for A100"): + provider.plan(offer, request) + return + plan = provider.plan(offer, request) + assert plan.details["gpu_devices"] == tuple(devices[index].uuid for index in expected) + assert plan.details["accelerator_name"] == ("NVIDIA-A100" if count == 1 else "GPU") + + +@pytest.mark.parametrize("visible_count", [0, 1, 2]) +def test_local_runtime_verifies_gpu_visibility_before_advertising_resources( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, visible_count: int, +) -> None: + import distributed + + directory = private_directory(tmp_path / "allocation", create=True) + scratch = private_directory(tmp_path / "scratch", create=True) + write_private_json(directory / "launch.json", { + "identity": "allocation", "deadline": time.monotonic() + 60, + "task_slots": 1, "scratch": str(scratch), + }) + write_private_json(directory / "identity.json", {"cpus": 1, "memory": 1024**3, "gpus": 1}) + monkeypatch.setattr(sys, "argv", ["local_runtime", str(directory)]) + monkeypatch.setattr(os, "getsid", lambda _: os.getpid()) + monkeypatch.setattr(os, "getpgrp", os.getpid) + monkeypatch.setattr(os, "umask", lambda _: 0) + monkeypatch.setattr(local_runtime.atexit, "register", lambda *_: None) + monkeypatch.setattr(signal, "setitimer", lambda *_: None) + monkeypatch.setattr( + signal, "signal", lambda signum, handler: handler(signum, None) + if signum == signal.SIGTERM else None, + ) + monkeypatch.setattr(local_runtime, "create_security", lambda _: None) + monkeypatch.setattr( + gpu, "visible_devices", lambda: tuple(f"GPU-{i}" for i in range(visible_count)), + ) + cluster = MagicMock() + cluster.return_value.__enter__.return_value.scheduler.id = "Scheduler-gpu" + monkeypatch.setattr(distributed, "LocalCluster", cluster) + + if visible_count != 1: + with pytest.raises(ComputeError, match="CUDA GPUs do not match"): + local_runtime.main() + cluster.assert_not_called() + assert not (directory / "connection.json").exists() + else: + local_runtime.main() + assert cluster.call_args.kwargs["resources"] == {"CPU": 1, "MEMORY": 1024**3, "GPU": 1} diff --git a/tests/test_compute_slurm.py b/tests/test_compute_slurm.py index 7d4b0fa9..41402360 100644 --- a/tests/test_compute_slurm.py +++ b/tests/test_compute_slurm.py @@ -13,11 +13,12 @@ from pathlib import Path from types import SimpleNamespace from typing import Any -from unittest.mock import MagicMock +from unittest.mock import AsyncMock, MagicMock import psutil import pytest +from lightcone.engine import gpu from lightcone.engine.compute import Compute, slurm, slurm_bootstrap from lightcone.engine.compute.catalog import Catalog from lightcone.engine.compute.model import ( @@ -96,6 +97,7 @@ def _control( f"JobId=123 JobName=lc-v1-{name} UserId=alice({owner}) JobState={state}\n" f" Comment={comment} \n" f" NumNodes=2 NumCPUs=512 CPUs/Task=256 MinMemoryNode=480G Restarts={restarts} " + "ReqTRES=cpu=512,mem=960G,node=2 " "SubmitTime=2026-09-27T10:00:00 StartTime=2026-09-27T10:00:05 Reason=None\n" ) @@ -169,6 +171,59 @@ def test_plan_preserves_native_envelope_without_native_queries( assert calls == [] +@pytest.mark.parametrize("submit", ["sbatch", "salloc"]) +def test_gpu_plan_requests_per_node_devices_for_the_allocation_and_step( + provider: slurm.SlurmProvider, offer: Offer, submit: str, +) -> None: + offer = offer.replace( + resources=offer.resources.replace(accelerators={"GPU": 4}), + config={**offer.config, "submit": submit, "constraint": "gpu"}, + ) + plan = provider.plan(offer, Request.parse("256", "480", gpus="GPU:4", num_nodes=2)) + assert "--gres=gpu:4" in plan.details["native_args"] + assert "--constraint=gpu" in plan.details["native_args"] + payload = provider._payload(plan, TOKEN) + assert "--gres=gpu:4" in payload + assert "--ntasks-per-node=1" in payload + assert payload[payload.index("--gpus") + 1] == "4" + + +def test_named_accelerator_uses_explicit_native_gres_mapping( + provider: slurm.SlurmProvider, offer: Offer, +) -> None: + offer = offer.replace(resources=offer.resources.replace(accelerators={"A100": 4})) + request = Request.parse("256", "480", gpus="A100:4", num_nodes=2) + with pytest.raises(ComputeError, match="requires its native gpu_type"): + provider.plan(offer, request) + offer = offer.replace(config={**offer.config, "gpu_type": "a100_80gb"}) + plan = provider.plan(offer, request) + assert "--gres=gpu:a100_80gb:4" in plan.details["native_args"] + assert "--gres=gpu:a100_80gb:4" in provider._payload(plan, TOKEN) + assert plan.resources.accelerator_name == "A100" + + +@pytest.mark.parametrize("gpu_type", ["", "a100:4", "a100,v100", "two types", 4]) +def test_slurm_gpu_type_must_be_one_native_type( + provider: slurm.SlurmProvider, offer: Offer, gpu_type: object, +) -> None: + offer = offer.replace( + resources=offer.resources.replace(accelerators={"A100": 4}), + config={**offer.config, "gpu_type": gpu_type}, + ) + with pytest.raises(ComputeError, match="one native GPU GRES type"): + provider.plan(offer, Request.parse("256", "480", gpus="A100:4")) + + +def test_cpu_offer_cannot_request_a_gpu_type( + provider: slurm.SlurmProvider, offer: Offer, +) -> None: + with pytest.raises(ComputeError, match="requires accelerator resources"): + provider.plan( + offer.replace(config={**offer.config, "gpu_type": "a100"}), + Request.parse("256", "480"), + ) + + def test_default_launch_assumes_a_shared_home_and_node_local_scratch( offer: Offer, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -696,6 +751,37 @@ def test_discovery_preserves_allocations_with_unknown_native_resource_evidence( assert snapshot.evidence == "unknown" +@pytest.mark.parametrize("native,expected,name", [ + ("TresPerNode=gres/gpu:4", 4, "GPU"), + ("TresPerNode=gres/gpu:a100:4", 4, "a100"), + ("TresPerNode=gres/gpu:4090:2", 2, "4090"), + ("TresPerNode=gres/gpu:a100:2,gres/gpu:v100:2", 4, "GPU"), + ("Gres=gpu:a100:2", 2, "a100"), + ("Gres=(null)", 0, None), + ("ReqTRES=cpu=512,mem=960G,node=2", 0, None), + ("ReqTRES=cpu=512,mem=960G,node=2,gres/gpu=8", None, None), + ("TresPerNode=gres/gpu:unknown", None, None), + ("TresPerNode=gres/gpu:a100/80gb:2", None, None), + ("TresPerNode=gres/gpu:1.5", None, None), + ("TresPerNode=gres/gpu:4,gres/gpu:a100:4", None, None), + ("", None, None), +]) +def test_gpu_discovery_reports_only_native_per_node_evidence( + provider: slurm.SlurmProvider, monkeypatch: pytest.MonkeyPatch, + native: str, expected: int | None, name: str | None, +) -> None: + control = _control().replace("ReqTRES=cpu=512,mem=960G,node=2", native) + _native(monkeypatch, {"squeue": _live(), "scontrol": control}) + snapshot, = provider.discover() + if expected is None: + assert snapshot.resources is None + assert snapshot.evidence == "unknown" + else: + assert snapshot.resources is not None and snapshot.resources.gpus == expected + assert snapshot.resources.accelerator_name == name + assert snapshot.evidence == "requested" + + def test_native_query_failure_is_not_an_empty_list( provider: slurm.SlurmProvider, monkeypatch: pytest.MonkeyPatch ) -> None: @@ -909,6 +995,7 @@ def _bootstrap_args(tmp_path: Path) -> argparse.Namespace: num_nodes=2, cpus=1, memory_bytes=64 * 1024**2, + gpus=0, task_slots=1, interface=None, ) @@ -916,6 +1003,7 @@ def _bootstrap_args(tmp_path: Path) -> argparse.Namespace: def _bootstrap_env(rank: int) -> dict[str, str]: return { + "CUDA_VISIBLE_DEVICES": "", "SLURM_JOB_ID": "123", "SLURM_PROCID": str(rank), "SLURM_NTASKS": "2", @@ -936,14 +1024,83 @@ def test_bootstrap_refuses_mismatched_native_envelope( slurm_bootstrap._allocation(_bootstrap_args(tmp_path)) +@pytest.mark.parametrize("native,visible_count,valid", [ + ("", 2, False), ("unknown", 2, False), ("1", 2, False), + ("2", 0, False), ("2", 1, False), ("2", 2, True), ("4", 4, True), +]) +def test_gpu_bootstrap_requires_native_and_cuda_capacity_without_rewriting_the_mask( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, + native: str, visible_count: int, valid: bool, +) -> None: + for key, value in _bootstrap_env(0).items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("SLURM_GPUS_ON_NODE", native) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "1,3") + monkeypatch.setenv("CUDA_DEVICE_ORDER", "FASTEST_FIRST") + + def devices() -> tuple[str, ...]: + assert os.environ["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" + assert os.environ["CUDA_VISIBLE_DEVICES"] == "1,3" + return tuple(f"GPU-{i}" for i in range(visible_count)) + + monkeypatch.setattr(gpu, "visible_devices", devices) + args = _bootstrap_args(tmp_path) + args.gpus = 2 + if valid: + _, _, rank = slurm_bootstrap._allocation(args) + assert rank == 0 + else: + with pytest.raises(ComputeError, match="GPUs"): + slurm_bootstrap._allocation(args) + assert os.environ["CUDA_VISIBLE_DEVICES"] == "1,3" + + +def test_gpu_worker_advertises_verified_capacity_with_the_native_mask( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + import distributed + + args = _bootstrap_args(tmp_path) + args.gpus = 2 + for key, value in _bootstrap_env(1).items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("SLURM_GPUS_ON_NODE", "2") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "1,3") + monkeypatch.setattr(gpu, "visible_devices", lambda: ("GPU-first", "GPU-second")) + connection = Connection( + namespace=NAMESPACE, provider="slurm", launch={"connection_root": args.connection_root}, + ) + directory = private_directory(slurm.attempt_directory(connection, IDENTITY, 0), create=True) + write_private_json(directory / "identity.json", { + "namespace": NAMESPACE, "native_id": "123", "token": TOKEN, "uid": os.getuid(), + "restarts": 0, "num_nodes": 2, "cpus": 1, "memory_bytes": args.memory_bytes, + "gpus": 2, "task_slots": 1, + }) + monkeypatch.setattr(slurm_bootstrap, "load_security", lambda _: None) + worker = MagicMock() + worker.__aenter__ = AsyncMock(return_value=worker) + worker.finished = AsyncMock() + factory = MagicMock(return_value=worker) + monkeypatch.setattr(distributed, "Worker", factory) + + asyncio.run(slurm_bootstrap.run(args)) + + assert factory.call_args.kwargs["resources"] == { + "CPU": 1, "MEMORY": args.memory_bytes, "GPU": 2, + } + assert os.environ["CUDA_VISIBLE_DEVICES"] == "1,3" + + def test_worker_rendezvous_has_a_finite_deadline( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: for key, value in _bootstrap_env(1).items(): monkeypatch.setenv(key, value) monkeypatch.setattr(slurm_bootstrap, "_STARTUP_TIMEOUT", 0) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "an-ambient-GPU") with pytest.raises(ComputeError, match="timed out"): asyncio.run(slurm_bootstrap.run(_bootstrap_args(tmp_path))) + assert os.environ["CUDA_VISIBLE_DEVICES"] == "" def test_bootstrap_defaults_scratch_to_the_node_temporary_directory( @@ -1032,6 +1189,7 @@ def test_standard_bootstrap_starts_scheduler_and_worker_on_rank_zero_and_worker_ workers = client.scheduler_info()["workers"] assert {worker["name"] for worker in workers.values()} == {"lightcone-0", "lightcone-1"} assert all(worker["nthreads"] == 1 for worker in workers.values()) + assert all(worker["resources"]["GPU"] == 0 for worker in workers.values()) assert client.submit(sum, [2, 3]).result(timeout=5) == 5 assert client.scheduler_info()["address"].startswith("tls://127.0.0.1:") assert all(address.startswith("tls://127.0.0.1:") for address in workers) diff --git a/tests/test_execution_resources.py b/tests/test_execution_resources.py new file mode 100644 index 00000000..c81ec0b4 --- /dev/null +++ b/tests/test_execution_resources.py @@ -0,0 +1,195 @@ +"""Parse recipe requests and refuse impossible execution before submitting work.""" + +from __future__ import annotations + +from typing import Any + +import pytest +from pydantic import ValidationError + +from lightcone.engine.execution_resources import TaskResources +from lightcone.engine.project import ProjectError + +GIB = 1024**3 + + +def _workers(*capacities: tuple[int, int]) -> dict[str, Any]: + return { + f"worker-{index}": {"resources": {"CPU": cpus, "MEMORY": memory}} + for index, (cpus, memory) in enumerate(capacities) + } + + +@pytest.mark.parametrize( + ("memory", "expected"), + [("512Mi", 512 * 1024**2), ("1.5GiB", 3 * GIB // 2), ("8GB", 8_000_000_000), + ("1B", 1), ("2 Ti", 2 * 1024**4), ("1000kB", 1_000_000)], +) +def test_memory_units_have_explicit_decimal_or_binary_meaning(memory: str, expected: int) -> None: + assert TaskResources.parse({"memory": memory}).memory_bytes == expected + + +@pytest.mark.parametrize("duration", ["1h30m", "30m", "45s", None]) +def test_recipe_time_limit_is_explicitly_refused(duration: object) -> None: + with pytest.raises(ProjectError, match="recipe time_limit is not supported"): + TaskResources.parse({"time_limit": duration}) + + +def test_integral_astra_float_cpu_count_is_accepted_without_rounding() -> None: + assert TaskResources.parse({"cpus": 4.0}).cpus == 4 + with pytest.raises(ProjectError, match="fractional CPUs"): + TaskResources.parse({"cpus": 0.5}) + + +@pytest.mark.parametrize( + "declaration", + [ + {"cpus": 0}, {"cpus": True}, {"cpus": "4"}, {"cpus": None}, + {"memory": "0Gi"}, {"memory": "0.1B"}, {"memory": "16"}, {"memory": 16}, + {"memory": "400m"}, {"memory": None}, + {"time_limit": "0m"}, {"time_limit": ""}, {"time_limit": "5m2h"}, + {"time_limit": "unlimited"}, {"disk": "10Gi"}, {"ram": "1Gi"}, + {"gpus": -1}, {"gpus": True}, {"gpus": 0.5}, {"gpus": "1"}, {"gpus": None}, + ], +) +def test_invalid_or_unhonored_declarations_are_not_silently_ignored( + declaration: dict[str, Any], +) -> None: + with pytest.raises(ProjectError): + TaskResources.parse(declaration) + + +def test_internal_resource_models_remain_validated() -> None: + with pytest.raises(ValidationError): + TaskResources(memory_bytes=-1) + with pytest.raises(ValidationError): + TaskResources(gpus=-1) + + +def test_recipe_memory_uses_exact_bytes_without_decimal_context_rounding() -> None: + one_byte = "0.000000000931322574615478515625" + assert TaskResources.parse({"memory": f"{one_byte}Gi"}).memory_bytes == 1 + with pytest.raises(ProjectError, match="exactly representable"): + TaskResources.parse({"memory": f"{one_byte}00000000000000001Gi"}) + + +def test_declared_requests_reserve_exact_cpu_and_memory_budgets() -> None: + task = TaskResources.parse({"cpus": 4, "memory": "6Gi"}) + assert task.requirements(_workers((8, 16 * GIB))) == {"CPU": 4, "MEMORY": 6 * GIB} + + +def test_missing_memory_reserves_entire_worker_instead_of_guessing() -> None: + assert TaskResources().requirements(_workers((8, 16 * GIB), (8, 16 * GIB))) == { + "CPU": 1, "MEMORY": 16 * GIB, + } + + +def test_probe_reserves_an_entire_worker() -> None: + assert TaskResources().requirements(_workers((8, 16 * GIB)), whole_worker=True) == { + "CPU": 8, "MEMORY": 16 * GIB, + } + + +def test_cpu_count_is_independent_of_dask_execution_threads() -> None: + workers = _workers((8, 16 * GIB)) + workers["worker-0"]["nthreads"] = 1 + assert TaskResources(cpus=8).requirements(workers)["CPU"] == 8 + + +def test_a_task_must_fit_one_worker_not_the_sum_of_the_cluster() -> None: + with pytest.raises(ProjectError, match="on one worker"): + TaskResources(cpus=8, memory_bytes=20 * GIB).requirements( + _workers((4, 16 * GIB), (4, 16 * GIB)) + ) + + +def test_cpu_and_memory_must_fit_on_the_same_worker() -> None: + with pytest.raises(ProjectError, match="no worker"): + TaskResources(cpus=8, memory_bytes=16 * GIB).requirements( + _workers((8, 4 * GIB), (4, 16 * GIB)) + ) + + +def test_explicit_requests_can_select_a_fitting_worker() -> None: + assert TaskResources(cpus=8, memory_bytes=8 * GIB).requirements( + _workers((4, 4 * GIB), (8, 16 * GIB)) + ) == {"CPU": 8, "MEMORY": 8 * GIB} + + +@pytest.mark.parametrize("whole_worker", [False, True]) +def test_missing_budgets_are_not_guessed_for_heterogeneous_workers(whole_worker: bool) -> None: + with pytest.raises(ProjectError, match="identical"): + TaskResources().requirements( + _workers((4, 4 * GIB), (8, 16 * GIB)), whole_worker=whole_worker + ) + + +@pytest.mark.parametrize( + "workers", + [{}, {"worker": {}}, {"worker": {"resources": {"CPU": 1}}}, + {"worker": {"resources": {"CPU": True, "MEMORY": GIB}}}, + {"worker": {"resources": {"CPU": 1, "MEMORY": float("nan")}}}], +) +def test_absent_or_unknown_worker_capacity_refuses_execution(workers: dict[str, Any]) -> None: + with pytest.raises(ProjectError): + TaskResources().requirements(workers) + + +def test_gpu_recipe_reserves_the_whole_worker_gpu_budget() -> None: + workers = _workers((8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 4 + task = TaskResources.parse({"cpus": 2, "memory": "4Gi", "gpus": 1}) + assert task.requirements(workers) == {"CPU": 2, "MEMORY": 4 * GIB, "GPU": 4} + + +def test_gpu_request_must_fit_one_worker() -> None: + workers = _workers((8, 16 * GIB), (8, 16 * GIB)) + for worker in workers.values(): + worker["resources"]["GPU"] = 1 + with pytest.raises(ProjectError, match="2 GPUs on one worker"): + TaskResources(gpus=2).requirements(workers) + + +def test_cpu_workers_cannot_satisfy_gpu_requests() -> None: + with pytest.raises(ProjectError, match="1 GPUs on one worker"): + TaskResources(gpus=1).requirements(_workers((8, 16 * GIB))) + + +def test_cpu_recipe_does_not_reserve_gpu_capacity() -> None: + workers = _workers((8, 16 * GIB), (8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 4 + assert TaskResources().requirements(workers) == {"CPU": 1, "MEMORY": 16 * GIB} + + +def test_probe_reserves_cpu_memory_and_all_gpus() -> None: + workers = _workers((8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 4 + assert TaskResources().requirements(workers, whole_worker=True) == { + "CPU": 8, "MEMORY": 16 * GIB, "GPU": 4, + } + + +def test_gpu_requirements_ignore_workers_that_cannot_fit_the_recipe() -> None: + workers = _workers((2, 2 * GIB), (8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 1 + workers["worker-1"]["resources"]["GPU"] = 4 + assert TaskResources(cpus=4, memory_bytes=8 * GIB, gpus=1).requirements(workers) == { + "CPU": 4, "MEMORY": 8 * GIB, "GPU": 4, + } + + +@pytest.mark.parametrize("whole_worker", [False, True]) +def test_whole_gpu_reservations_refuse_ambiguous_worker_budgets(whole_worker: bool) -> None: + workers = _workers((8, 16 * GIB), (8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 1 + workers["worker-1"]["resources"]["GPU"] = 4 + with pytest.raises(ProjectError, match="identical GPU budgets"): + TaskResources(gpus=1).requirements(workers, whole_worker=whole_worker) + + +@pytest.mark.parametrize("capacity", [-1, 0.5, True, "1", None, float("nan"), float("inf")]) +def test_malformed_gpu_capacity_is_never_ignored(capacity: object) -> None: + workers = _workers((8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = capacity + with pytest.raises(ProjectError, match="GPU count"): + TaskResources().requirements(workers) diff --git a/tests/test_gpu.py b/tests/test_gpu.py new file mode 100644 index 00000000..b5568397 --- /dev/null +++ b/tests/test_gpu.py @@ -0,0 +1,144 @@ +"""CUDA owns native device enumeration; only verified identities become capacity.""" + +from __future__ import annotations + +import ctypes +import json +import subprocess +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import Mock +from uuid import UUID + +import pytest + +from lightcone.engine import gpu +from lightcone.engine.project import ProjectError + +GPU = "GPU-01234567-89ab-cdef-0123-456789abcdef" +DEVICE = {"uuid": GPU, "name": "NVIDIA-A100-SXM4-80GB"} + + +def test_empty_mask_never_loads_or_queries_cuda(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "") + run = Mock(side_effect=AssertionError("must not query hidden devices")) + monkeypatch.setattr(gpu.subprocess, "run", run) + assert gpu.visible_devices() == () + + +def test_uuid_probe_preserves_the_native_mask_and_uses_isolated_python( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(gpu.sys, "platform", "linux") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + run = Mock(return_value=subprocess.CompletedProcess([], 0, json.dumps([DEVICE]), "")) + monkeypatch.setattr(gpu.subprocess, "run", run) + assert gpu.visible_devices() == (GPU,) + argv = run.call_args.args[0] + assert argv[1] == "-I" + assert Path(argv[2]) == Path(gpu.__file__) + assert "env" not in run.call_args.kwargs # Native visibility is inherited unchanged. + assert run.call_args.kwargs["timeout"] > 0 + + +@pytest.mark.parametrize("payload", [ + {}, [42], [GPU], [{"uuid": "GPU-bad", "name": "A100"}], + [{"uuid": GPU, "name": "A100:4"}], [{"uuid": GPU, "name": ""}], +]) +def test_invalid_inventory_is_not_capacity( + monkeypatch: pytest.MonkeyPatch, payload: object, +) -> None: + monkeypatch.setattr(gpu.sys, "platform", "linux") + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + monkeypatch.setattr( + gpu.subprocess, "run", + Mock(return_value=subprocess.CompletedProcess([], 0, json.dumps(payload), "")), + ) + with pytest.raises(ProjectError, match="invalid GPU identities"): + gpu.visible_devices() + + +def test_inventory_preserves_models_and_rejects_duplicate_identities( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(gpu.sys, "platform", "linux") + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + run = Mock(return_value=subprocess.CompletedProcess([], 0, json.dumps([DEVICE]), "")) + monkeypatch.setattr(gpu.subprocess, "run", run) + assert gpu.inventory() == (gpu.Device(GPU, DEVICE["name"]),) + run.return_value.stdout = json.dumps([DEVICE, DEVICE]) + with pytest.raises(ProjectError, match="duplicate GPU identities"): + gpu.inventory() + + +def test_probe_failure_is_not_reported_as_an_empty_machine(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(gpu.sys, "platform", "linux") + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + monkeypatch.setattr( + gpu.subprocess, "run", + Mock(return_value=subprocess.CompletedProcess([], 1, "", "CUDA driver returned error 999")), + ) + with pytest.raises(ProjectError, match="error 999"): + gpu.visible_devices() + + +def test_missing_cuda_is_a_cpu_only_machine(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(ctypes, "CDLL", Mock(side_effect=OSError("no CUDA driver"))) + assert gpu._probe() == [] + + +@pytest.mark.parametrize("partitioned", [False, True]) +@pytest.mark.parametrize("model_name", ["NVIDIA A100-SXM4-80GB", "TITAN X (Pascal)"]) +def test_probe_uses_cuda_device_handles_and_rejects_mig( + monkeypatch: pytest.MonkeyPatch, partitioned: bool, model_name: str, +) -> None: + # The visible ordinal maps to a driver handle, not a host GPU index. The + # real ctypes buffers and signatures exercise the probe's FFI boundary. + raw = UUID(GPU.removeprefix("GPU-")).bytes + + def count(pointer: object) -> int: + ctypes.cast(pointer, ctypes.POINTER(ctypes.c_int))[0] = 1 + return 0 + + def device(pointer: object, ordinal: int) -> int: + assert ordinal == 0 + ctypes.cast(pointer, ctypes.POINTER(ctypes.c_int))[0] = 7 + return 0 + + def identity(buffer: object, handle: ctypes.c_int, *, instance: bool = False) -> int: + assert handle.value == 7 + ctypes.memmove(buffer, bytes(16) if partitioned and instance else raw, 16) + return 0 + + def name(buffer: object, size: int, handle: ctypes.c_int) -> int: + assert handle.value == 7 + value = model_name.encode("ascii") + b"\x00" + assert size >= len(value) + ctypes.memmove(buffer, value, len(value)) + return 0 + + driver = SimpleNamespace( + cuInit=Mock(return_value=0), + cuDeviceGetCount=Mock(side_effect=count), + cuDeviceGet=Mock(side_effect=device), + cuDeviceGetUuid=Mock(side_effect=identity), + cuDeviceGetUuid_v2=Mock(side_effect=lambda b, d: identity(b, d, instance=True)), + cuDeviceGetName=Mock(side_effect=name), + ) + monkeypatch.setattr(ctypes, "CDLL", Mock(return_value=driver)) + if partitioned: + with pytest.raises(RuntimeError, match="MIG"): + gpu._probe() + else: + expected = "NVIDIA-A100-SXM4-80GB" if model_name.startswith("NVIDIA") else "TITAN-X-Pascal" + assert gpu._probe() == [{"uuid": GPU, "name": expected}] + + +def test_only_nvidia_character_nodes_are_granted( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + for name in ("nvidia0", "nvidiactl", "nvidia-user-file", "unrelated"): + (tmp_path / name).touch() + monkeypatch.setattr(gpu, "_DEVICE_ROOT", tmp_path) + monkeypatch.setattr(Path, "is_char_device", lambda p: p.name in {"nvidia0", "nvidiactl"}) + assert gpu.device_paths() == (tmp_path / "nvidia0", tmp_path / "nvidiactl") diff --git a/tests/test_gpu_execution.py b/tests/test_gpu_execution.py new file mode 100644 index 00000000..06e45b83 --- /dev/null +++ b/tests/test_gpu_execution.py @@ -0,0 +1,106 @@ +"""GPU requests reach each command without changing the shared worker environment.""" + +from __future__ import annotations + +import os +from collections.abc import Callable +from dataclasses import replace +from pathlib import Path +from unittest.mock import Mock + +import pytest + +from lightcone.engine import assets, container, gpu, identity, plan, worker +from lightcone.engine import run as engine_run +from lightcone.engine.project import ProjectError + +_SPEC = """ +version: "0.0.13" +name: analysis +inputs: [] +outputs: + - id: visibility + type: metric + format: txt + recipe: + command: printf '%s' "$CUDA_VISIBLE_DEVICES" > {output} +""" + + +@pytest.fixture +def project(analysis: Callable[..., Path]) -> tuple[Path, plan.Task, worker.RunContext]: + root = analysis(_SPEC) + task = plan.build(root).tasks[("baseline", "visibility")] + context = worker.RunContext( + env_version=identity.env_version(root), + head=("0123456789abcdef", "https://example/analysis.git"), + versions=assets.Versions(), + runtime=container.runtime_for_run(root, build=False), + uv_version="0.0.0-test", + ) + return root, task, context + + +@pytest.mark.parametrize("requested", [0, 1, 2]) +def test_recipe_gets_requested_uuid_mask_in_its_actual_subprocess( + project: tuple[Path, plan.Task, worker.RunContext], + monkeypatch: pytest.MonkeyPatch, requested: int, +) -> None: + root, task, context = project + visible = ("GPU-second", "GPU-first") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "native-allocation") + monkeypatch.setattr(gpu, "visible_devices", lambda: visible) + monkeypatch.setattr(gpu, "device_paths", lambda: ()) + result = worker.execute(root, replace(task, resources={"gpus": requested}), {}, context) + assert result.status == "ok", result.reason + assert task.output_path.read_text() == ",".join(visible[:requested]) + assert os.environ["CUDA_VISIBLE_DEVICES"] == "native-allocation" + + +def test_worker_gpu_mismatch_preserves_existing_output_and_manifest( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, +) -> None: + root, task, context = project + task.output_path.parent.mkdir(parents=True, exist_ok=True) + task.output_path.write_text("previous output") + task.manifest_path.write_text("previous manifest") + monkeypatch.setattr(gpu, "visible_devices", lambda: ("GPU-only",)) + with pytest.raises(ProjectError, match="requests 2 GPUs.*only 1 CUDA devices"): + worker.execute(root, replace(task, resources={"gpus": 2}), {}, context) + assert task.output_path.read_text() == "previous output" + assert task.manifest_path.read_text() == "previous manifest" + + +@pytest.mark.parametrize("reserved", [0, 1, 2]) +def test_probe_exposes_only_reserved_devices_from_the_native_allocation( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + reserved: int, +) -> None: + _, _, context = project + visible = ("GPU-first", "GPU-second", "GPU-unreserved") + discover = Mock(return_value=visible) + monkeypatch.setattr(gpu, "visible_devices", discover) + monkeypatch.setattr(gpu, "device_paths", lambda: ()) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "native-allocation") + received: list[bytes] = [] + outcome = engine_run._probe( + context.runtime, [], ("sh", "-c", "printf '%s' \"$CUDA_VISIBLE_DEVICES\""), + reserved, + output=lambda stream, data: received.append(data) if stream == "stdout" else None, + ) + assert outcome.returncode == 0 + assert b"".join(received).decode() == ",".join(visible[:reserved]) + assert discover.call_count == bool(reserved) + assert os.environ["CUDA_VISIBLE_DEVICES"] == "native-allocation" + + +def test_probe_gpu_shortage_refuses_before_running_the_command( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, +) -> None: + _, _, context = project + monkeypatch.setattr(gpu, "visible_devices", lambda: ("GPU-only",)) + execute = Mock() + monkeypatch.setattr(engine_run.sandbox, "run", execute) + with pytest.raises(ProjectError, match="reserved 2 GPUs.*only 1 CUDA devices"): + engine_run._probe(context.runtime, [], ("true",), 2, output=lambda *_: None) + execute.assert_not_called() diff --git a/tests/test_materialize.py b/tests/test_materialize.py index c9b9a29b..ddaf7c86 100644 --- a/tests/test_materialize.py +++ b/tests/test_materialize.py @@ -471,7 +471,9 @@ def digest(path: Path) -> str: return real(path) class _Copied(_Inline): - def submit(self, fn: Callable[..., object], *args: object, key: str) -> object: + def submit( + self, fn: Callable[..., object], *args: object, key: str, resources: dict[str, float], + ) -> object: return fn(*pickle.loads(pickle.dumps(args))) monkeypatch.setattr(assets, "data_version", digest) @@ -1060,6 +1062,142 @@ def test_the_recorded_command_holds_on_a_fresh_clone( # ---- the scheduler seam ---------------------------------------------------- +def _resource_cluster( + monkeypatch: pytest.MonkeyPatch, *, workers: int = 1, gpus: int = 0, +) -> None: + from distributed import Client, LocalCluster + + from lightcone.engine import compute + + @contextmanager + def connect(cluster_id: str) -> Iterator[Any]: + with LocalCluster( + n_workers=workers, threads_per_worker=4, processes=False, + dashboard_address=None, resources={"CPU": 4, "MEMORY": 2 * 1024**3, "GPU": gpus}, + ) as cluster, Client(cluster, set_as_default=False) as client: + yield client + + monkeypatch.setattr(compute, "connect", connect) + + +@pytest.mark.parametrize("resource_spec", ["cpus: 5", "memory: 3Gi"]) +def test_resource_refusal_precedes_preparation_and_all_recipes( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, resource_spec: str, +) -> None: + spec = _SPEC.replace( + "command: cat", f"resources: {{{resource_spec}}}\n command: cat" + ) + root = analysis(spec, universes={"baseline": _UNIVERSE}) + before = dataset.head(root) + _resource_cluster(monkeypatch) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("preparation began before all resource requests were validated") + + monkeypatch.setattr(engine, "_fetch_inputs", unexpected) + monkeypatch.setattr(engine.container, "runtime_for_run", unexpected) + with pytest.raises(ProjectError, match="baseline/second:.*no worker"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert dataset.head(root) == before + assert not dataset.status(root) + assert not (root / "results/baseline/first.txt").exists() + + +@pytest.mark.parametrize("resource_spec", [ + "gpus: 1", "disk: 1Gi", "cpus: 0.5", "cpus: 0", "memory: null", "time_limit: 1s", +]) +def test_execution_only_resources_do_not_block_read_only_commands( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, resource_spec: str, +) -> None: + spec = _SPEC.replace( + "command: cat", f"resources: {{{resource_spec}}}\n command: cat", + ) + root = analysis(spec, universes={"baseline": _UNIVERSE}) + assert len(engine.status(root).outputs) == 2 + assert set(engine.check(root, []).planned) == {"baseline/first", "baseline/second"} + + _resource_cluster(monkeypatch) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("preparation began before execution requirements were validated") + + monkeypatch.setattr(engine, "_fetch_inputs", unexpected) + with pytest.raises(ProjectError, match="baseline/second"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert not dataset.status(root) + + +def test_empty_cluster_refuses_before_project_preparation( + root: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + _resource_cluster(monkeypatch, workers=0) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("preparation began without an available worker") + + monkeypatch.setattr(engine, "_fetch_inputs", unexpected) + with pytest.raises(ProjectError, match="no workers"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert not dataset.status(root) + + +@pytest.mark.parametrize( + ("resource_spec", "gpus", "expected_parallelism"), + [ + ("cpus: 3, memory: 256Mi", 0, 1), + ("cpus: 1, memory: 1Gi", 0, 2), + ("cpus: 1, memory: 256Mi, gpus: 1", 2, 1), + ], +) +def test_real_dask_respects_recipe_resource_reservations( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, + resource_spec: str, gpus: int, expected_parallelism: int, +) -> None: + # Four Dask threads would run all four subprocesses together without + # resource reservations. Each independent output records its live interval. + spec = 'version: "0.0.13"\nname: analysis\ninputs: []\noutputs:\n' + "".join( + f" - id: task{index}\n" + " type: metric\n" + " format: json\n" + " recipe:\n" + f" resources: {{{resource_spec}}}\n" + " command: python src/work.py {output}\n" + for index in range(4) + ) + root = analysis(spec, files={"src/work.py": """ + import json + import os + import sys + import time + from pathlib import Path + start = time.monotonic() + time.sleep(0.5) + Path(sys.argv[1]).write_text(json.dumps([ + start, time.monotonic(), os.environ.get("CUDA_VISIBLE_DEVICES"), + ])) + """}) + from lightcone.engine import gpu + + monkeypatch.setattr(gpu, "visible_devices", lambda: ("GPU-first", "GPU-second")) + monkeypatch.setattr(gpu, "device_paths", lambda: ()) + _resource_cluster(monkeypatch, gpus=gpus) + + report = engine.materialize(root, [], cluster_id=CLUSTER_ID) + + assert report.ok and len(report.made) == 4 + events = [] + for path in (root / "results/baseline").glob("task*.json"): + start, finish, visible = json.loads(path.read_text()) + assert visible == ("GPU-first" if gpus else "") + events.extend([(start, 1), (finish, -1)]) + live = peak = 0 + for _, change in sorted(events): + live += change + peak = max(peak, live) + assert peak == expected_parallelism + assert not dataset.status(root) + + def test_a_real_cluster_still_fits_through_the_seam(root: Path, cluster_id: str) -> None: """The one test that starts Dask. The seam is only worth having if the thing it abstracts still goes through it.""" @@ -1084,7 +1222,8 @@ def test_a_processes_cluster_fits_through_the_seam( @contextmanager def processes(cluster_id: str) -> Iterator[Any]: with LocalCluster( - n_workers=2, threads_per_worker=1, processes=True, dashboard_address=None + n_workers=2, threads_per_worker=1, processes=True, dashboard_address=None, + resources={"CPU": 1, "MEMORY": 1024**3}, ) as cluster: with Client(cluster, set_as_default=False) as client: yield client @@ -1320,3 +1459,4 @@ def test_an_output_the_spec_dropped_is_excluded_and_named(root: Path, inline: No assert any(".second.manifest.json" in w for w in report.warnings) document = (root / "ro-crate-metadata.json").read_text() assert "results/baseline/second.txt" not in document + diff --git a/tests/test_plan.py b/tests/test_plan.py index 48a994b0..9f069d38 100644 --- a/tests/test_plan.py +++ b/tests/test_plan.py @@ -122,6 +122,39 @@ def test_an_output_addresses_its_own_file(tmp_path: Path) -> None: assert "results/baseline/fit.json" in task.recipe +def test_recipe_resources_survive_graph_resolution(tmp_path: Path) -> None: + spec = _SPEC.replace( + "command: python src/fit.py", + "resources: {cpus: 4, memory: 6Gi, time_limit: 1h30m}\n" + " command: python src/fit.py", + ) + graph = _build(_project(tmp_path, spec)) + task = graph.tasks[("baseline", "fit")] + assert task.resources == {"cpus": 4, "memory": "6Gi", "time_limit": "1h30m"} + assert graph.tasks[("baseline", "report")].resources == {} + + +@pytest.mark.parametrize( + ("declaration", "expected"), + [ + ("gpus: 1", {"gpus": 1}), + ("disk: 10Gi", {"disk": "10Gi"}), + ("cpus: 0.5", {"cpus": 0.5}), + ("cpus: 0", {"cpus": 0}), + ("memory: null", {"memory": None}), + ], +) +def test_execution_support_does_not_limit_graph_construction( + tmp_path: Path, declaration: str, expected: dict[str, object], +) -> None: + spec = _SPEC.replace( + "command: python src/fit.py", + f"resources: {{{declaration}}}\n command: python src/fit.py", + ) + task = _build(_project(tmp_path, spec)).tasks[("baseline", "fit")] + assert task.resources == expected + + def test_a_declared_input_resolves_to_its_source(tmp_path: Path) -> None: task = _build(_project(tmp_path)).tasks[("baseline", "fit")] assert task.inputs == {"catalog": tmp_path / "data" / "catalog.fits"} @@ -280,5 +313,3 @@ def test_an_output_without_a_format_is_refused_by_name(tmp_path: Path) -> None: - - diff --git a/tests/test_sandbox_oci.py b/tests/test_sandbox_oci.py index cf8cdccd..d346fd5b 100644 --- a/tests/test_sandbox_oci.py +++ b/tests/test_sandbox_oci.py @@ -7,7 +7,9 @@ from __future__ import annotations +import csv import subprocess +from dataclasses import replace from pathlib import Path from typing import Any @@ -185,6 +187,40 @@ def test_runtimes_differ_only_in_their_spellings(root: Path, policy: Policy) -> assert p == d == h +@pytest.mark.parametrize("runtime", ["podman", "docker", "podman-hpc"]) +def test_gpu_flags_name_only_the_requested_devices( + root: Path, policy: Policy, runtime: str, +) -> None: + devices = "GPU-first,GPU-second" + selected = replace(policy, env={**policy.env, "CUDA_VISIBLE_DEVICES": devices}) + backend = _backend(root, runtime) + argv = backend.wrap(selected, ["true"]) + assert argv == backend.wrap(selected, ["true"]) + assert f"--env=CUDA_VISIBLE_DEVICES={devices}" in argv + if runtime == "podman-hpc": + assert "--gpu" in argv + elif runtime == "docker": + assert next(csv.reader([argv[argv.index("--gpus") + 1]])) == [f"device={devices}"] + else: + assert [arg for arg in argv if arg.startswith("--device=")] == [ + "--device=nvidia.com/gpu=GPU-first", "--device=nvidia.com/gpu=GPU-second", + ] + assert "all" not in argv + assert "--device=nvidia.com/gpu=all" not in argv + + +@pytest.mark.parametrize("runtime", ["podman", "docker", "podman-hpc"]) +def test_cpu_containers_do_not_request_gpu_access(root: Path, policy: Policy, runtime: str) -> None: + policy = replace(policy, env={**policy.env, "NVIDIA_VISIBLE_DEVICES": "all"}) + argv = _backend(root, runtime).wrap(policy, ["true"]) + assert "--env=CUDA_VISIBLE_DEVICES=" in argv + assert "--env=NVIDIA_VISIBLE_DEVICES=void" in argv + assert "--env=NVIDIA_VISIBLE_DEVICES=all" not in argv + assert policy.env["NVIDIA_VISIBLE_DEVICES"] == "all" + assert "--gpu" not in argv and "--gpus" not in argv + assert not any(arg.startswith("--device=") for arg in argv) + + def test_the_environment_is_an_allowlist_never_ambient( root: Path, policy: Policy, monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/tests/test_sandbox_policy.py b/tests/test_sandbox_policy.py index 0cfc2605..dbaedcb2 100644 --- a/tests/test_sandbox_policy.py +++ b/tests/test_sandbox_policy.py @@ -238,6 +238,29 @@ def test_the_entropy_sources_stay_read_only(built: policy_module.Policy) -> None assert device not in built.write, node +@pytest.mark.parametrize("containerized", [False, True]) +@pytest.mark.parametrize("devices", [(), ("GPU-first", "GPU-second")]) +def test_gpu_policy_sets_command_visibility_without_changing_the_host( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, + containerized: bool, devices: tuple[str, ...], +) -> None: + from lightcone.engine import gpu + + project = tmp_path / "project" + project.mkdir() + device = tmp_path / "nvidia0" + device.touch() + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "native-allocation") + monkeypatch.setattr(gpu, "device_paths", lambda: (device,)) + with scope(policy_module.exec_policy( + project, containerized=containerized, gpu_devices=devices, + )) as built: + assert built.env["CUDA_VISIBLE_DEVICES"] == ",".join(devices) + assert (device in built.write) == (bool(devices) and not containerized) + assert Path("/dev") not in built.write + assert os.environ["CUDA_VISIBLE_DEVICES"] == "native-allocation" + + def test_proc_and_sys_are_not_restricted(built: policy_module.Policy) -> None: """Real tools write them — /proc/self/oom_score_adj, coredump_filter, MPI and CUDA runtimes poking /sys — and none of it is a channel From fd9b2d3e94635cca953574ba289486173942de50 Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Tue, 29 Sep 2026 04:24:40 -0700 Subject: [PATCH 2/3] Use native GPU masks and Dask resource reservations Signed-off-by: Francois Lanusse --- CLAUDE.md | 47 +++--- docs/api/compute.md | 56 +++---- docs/api/index.md | 1 - docs/api/sandbox.md | 21 ++- docs/api/worker.md | 16 +- docs/architecture.md | 21 +-- docs/cli/compute.md | 9 +- docs/cli/materialize.md | 8 +- docs/cli/run.md | 4 +- docs/user/cluster.md | 94 ++++++------ src/lightcone/engine/compute/catalog.py | 29 +--- src/lightcone/engine/compute/local.py | 44 +++--- src/lightcone/engine/compute/local_runtime.py | 8 +- .../engine/compute/slurm_bootstrap.py | 5 +- src/lightcone/engine/container.py | 11 +- src/lightcone/engine/execution_resources.py | 5 +- src/lightcone/engine/gpu.py | 141 ----------------- src/lightcone/engine/run.py | 14 +- src/lightcone/engine/sandbox/oci.py | 24 +-- src/lightcone/engine/sandbox/policy.py | 27 +++- src/lightcone/engine/worker.py | 23 ++- tests/test_compute.py | 45 ++---- tests/test_compute_local.py | 110 +++++++------ tests/test_compute_slurm.py | 40 +++-- tests/test_gpu.py | 144 ------------------ tests/test_gpu_execution.py | 76 +++++---- tests/test_materialize.py | 8 +- tests/test_sandbox_oci.py | 37 ++--- tests/test_sandbox_policy.py | 42 +++-- 29 files changed, 412 insertions(+), 698 deletions(-) delete mode 100644 src/lightcone/engine/gpu.py delete mode 100644 tests/test_gpu.py diff --git a/CLAUDE.md b/CLAUDE.md index e8927406..6263249c 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1727,10 +1727,10 @@ and use it for every native ownership check and filter. **Local compute needs no setup.** An absent implicit `~/.lightcone/compute.yaml` selects a built-in local catalog: one CPU, 1 GiB, one node, fast startup, 30-minute default and two-hour maximum lifetime. It writes no catalog and starts no cluster. -Linux CUDA discovery adds GPU offers grouped by native model (`local-gpu`, or -`local-gpu-1`, etc. for mixed models), retaining those CPU/RAM defaults. Discovery -failure must not disable the CPU offer. Explicit local allocations remain -cooperative; they do not reserve a device against other host programs/allocations. +GPU offers require an explicit catalog and, for local launches, a nonempty +`CUDA_VISIBLE_DEVICES` mask on Linux. No GPU auto-discovery. Local GPU capacity and +model labels are configured, not hardware-verified; allocations do not reserve +devices exclusively against other host programs or allocations. Configured catalogs replace it completely; missing explicit paths and invalid files are errors. Execution still requires an explicitly launched cluster's name or ID. @@ -1763,10 +1763,10 @@ uses binary units: bare `32`, `32GB`, and `32GiB` agree. Catalog resources use one `accelerators: NAME[:COUNT]` or a one-entry mapping; CLI `--gpus A100:4`, `A100`, or generic `GPU:4` selects an exact positive whole count, while `0` means CPU only. Type matching is case-insensitive; no GPU `+`, fractions, or -global model alias registry. Local discovery reports native names with whitespace -and punctuation normalized to hyphens (e.g. `NVIDIA-A100-SXM4-80GB`). Named Slurm offers must map -their public label to the site's GRES type through `config.gpu_type`; generic -`GPU` offers may omit that setting. Preserve native evidence in observations. +global model alias registry. Local accelerator labels are trusted configuration. +Named Slurm offers must map their public label to the site's GRES type through +`config.gpu_type`; generic `GPU` offers may omit that setting. Preserve native +evidence in observations. **Configured compute roots may be filesystem aliases.** Resolve connection and scratch roots before appending managed namespace, submission, or attempt paths. @@ -1802,28 +1802,25 @@ or submission, then pass reservations explicitly to submission. Recipe `time_lim is unsupported and must fail explicitly; allocation walltime remains supported. Workers advertise CPU/MEMORY/GPU; tasks reserve their declarations, with omitted RAM reserving a whole worker's memory and probes reserving all whole-worker budgets. -GPU recipes reserve the worker's full GPU budget, one GPU recipe at a time, while -their commands see only their requested count. Recipe `gpus` defaults to zero and -does not select a model. Recipe memory retains ASTRA units (`8Gi` binary, +GPU recipes reserve the worker's full GPU budget, one GPU recipe at a time, and +inherit its whole allocation mask. Recipe `gpus` is a minimum capacity requirement, +not a visibility limit; it defaults to zero and does not select a model. Recipe memory retains ASTRA units (`8Gi` binary, `8GB` decimal, no bare quantities), independently of compute's SkyPilot units. Thread slots remain a separate concurrency cap. Reservations are cooperative, not per-command OS CPU/RAM limits; unsupported disk/model requests and fractional CPU/GPU counts fail explicitly. -**GPU discovery and visibility stay native and per-command.** `engine.gpu` probes -the CUDA Driver API in an isolated stdlib process, respecting native masks and -returning UUIDs/model names without initializing CUDA in a reusable worker. No -new Python GPU dependency or custom Dask worker. Whole NVIDIA GPUs on Linux only; -MIG/fractional GPUs and other vendors are unsupported. Verify requested devices -before deleting outputs. Set the command policy's CUDA mask; never mutate the -shared worker environment. CPU commands get an empty mask, and probes expose only -their reserved count, even if the native mask is larger. Direct policies grant -known NVIDIA character nodes; native permissions/cgroups still govern access. -OCI uses Docker's explicit UUID `--gpus` (CSV quoting for multiple devices), Podman -CDI UUID device names, or podman-hpc `--gpu` plus the CUDA mask. CPU containers -override `NVIDIA_VISIBLE_DEVICES=void` to defeat image defaults. The CUDA mask is -cooperative visibility, not a security claim. Physical GPU execution remains -unvalidated; tests simulate inventory and check real subprocess masks/native argv. +**GPU visibility comes from the allocation, not device discovery.** No CUDA probe, +UUID inventory, MIG detection, model verification, or custom Dask worker. Local +launch freezes the externally supplied nonempty CUDA mask and optional device +order. Slurm validates native GPU counts, preserves its mask, and sets +`CUDA_DEVICE_ORDER=PCI_BUS_ID`. `exec_policy(use_gpus=True)` inherits that whole +mask; CPU commands get an empty one. Never mutate the reusable worker's environment. +Direct GPU policies grant native NVIDIA character nodes; OS permissions and cgroups +remain authoritative. Container GPU execution supports podman-hpc `--gpu` only; +ordinary Docker/Podman GPU requests fail explicitly. CPU containers remain supported +on all runtimes and set `NVIDIA_VISIBLE_DEVICES=void`. Physical GPU execution remains +unvalidated; tests check real subprocess masks and native argv without GPU hardware. **One catalog selector, `LC_COMPUTE_CONFIG` (2026-09).** `lc compute --config` was removed: `run` and `materialize` resolve clusters through the catalog too, diff --git a/docs/api/compute.md b/docs/api/compute.md index 7c97b833..3758504c 100644 --- a/docs/api/compute.md +++ b/docs/api/compute.md @@ -18,10 +18,9 @@ It owns no service, registry, or saved current-cluster selection. `Catalog.load()` defaults to `~/.lightcone/compute.yaml`. When that implicit file is absent, the built-in catalog exposes a `local` offer: one CPU, 1 GiB, one node, -fast startup, 30-minute default and two-hour maximum lifetime. Successful Linux -CUDA discovery adds one GPU offer per native model: `local-gpu` for one model, -or `local-gpu-1`, `local-gpu-2`, etc. GPU discovery failure preserves the CPU offer. -It creates no configuration file or allocation. Configured catalogs replace it completely. +fast startup, 30-minute default and two-hour maximum lifetime. GPU offers require +an explicit catalog. It creates no configuration file or allocation. Configured +catalogs replace it completely. Missing paths selected through an argument or `LC_COMPUTE_CONFIG`, unreadable files, and invalid catalogs remain errors. Stable connection namespaces let separate invocations discover and attach to the same local allocations. @@ -133,10 +132,10 @@ checks that one worker can satisfy it and returns the resource dictionary used by `Client.submit`. An omitted memory request reserves the full homogeneous worker budget; `whole_worker=True` reserves CPU, memory, and GPUs for a probe. Recipe GPU counts -default to zero; a GPU recipe reserves the full GPU budget of a fitting worker, -while exposing only the requested device count. This serializes GPU recipes per -worker without a device-assignment service. Unsupported disk/type requests and -fractional CPU/GPU counts fail before execution. +default to zero; a GPU recipe reserves the full GPU budget of a fitting worker +and inherits its whole allocation mask. The requested count is a minimum, not a +visibility limit. This serializes GPU recipes per worker without device assignment. +Unsupported disk/type requests and fractional CPU/GPU counts fail before execution. The materialize scheduler validates every selected task before preparation or submission, preventing earlier tasks from starting before a later impossible @@ -153,29 +152,24 @@ Recipe memory remains ASTRA-style: `8Gi` is binary, `8GB` is decimal, and units are required. Allocation memory follows the compute convention above; keep the two parsers' contracts explicit even though they share exact byte arithmetic. -## GPU discovery and visibility - -`engine.gpu.inventory()` uses a short isolated stdlib process to query the CUDA -Driver API for native UUIDs and model names, respecting `CUDA_VISIBLE_DEVICES`. -`visible_devices()` returns those UUIDs; `device_paths()` lists NVIDIA character -device nodes for direct sandbox grants. CUDA is never initialized in the reusable -worker, and no additional GPU Python dependency or custom Dask worker is needed. -Missing drivers produce an empty inventory. Discovery rejects invalid identities, -driver failures, and partitioned MIG devices rather than inventing capacity. - -Local launch selects devices matching the offer, freezes their UUID mask in the -allocation environment, and checks the visible count before starting Dask. Slurm -validates native GPU capacity and CUDA visibility before advertising the offer's -GPU budget. Workers check recipe visibility before resetting output files. -Slurm bootstrap sets `CUDA_DEVICE_ORDER=PCI_BUS_ID` before discovery so CUDA -interprets native numeric masks in Slurm/NVML's device order. -Probes receive the reserved GPU count explicitly and never expose excess native -devices. CPU commands receive an empty CUDA mask without GPU discovery. - -The sandbox applies masks per command, preserving the worker's shared environment. -OCI backends request native device injection by UUID; direct policies grant known -NVIDIA device nodes. Visibility is cooperative, with native permissions and cgroups -still authoritative. See [GPU deployment requirements](../user/cluster.md#gpu-allocations). +## GPU allocation and visibility + +Local GPU offers require Linux and an explicit nonempty `CUDA_VISIBLE_DEVICES`. +Planning freezes that mask and optional `CUDA_DEVICE_ORDER`; launch passes them +to the worker unchanged. Count and model are catalog declarations, not hardware +observations. The built-in local offer remains CPU-only. + +Slurm requests native GPU GRES and validates `SLURM_GPUS_ON_NODE` before +advertising the worker's GPU budget. Bootstrap preserves Slurm's CUDA mask and +sets `CUDA_DEVICE_ORDER=PCI_BUS_ID`. There is no CUDA probe, device inventory, or +custom Dask worker. + +The sandbox's `use_gpus` policy option inherits the worker's mask for GPU commands +and supplies an empty mask for CPU commands, without modifying the reusable +worker's environment. Direct GPU policies grant native NVIDIA character devices. +Container GPU execution uses podman-hpc's `--gpu`; ordinary Docker and Podman GPU +requests are refused. Native permissions and cgroups remain authoritative. +See [GPU deployment requirements](../user/cluster.md#gpu-allocations). ## Execution output and teardown diff --git a/docs/api/index.md b/docs/api/index.md index a202a896..d79ee66b 100644 --- a/docs/api/index.md +++ b/docs/api/index.md @@ -19,7 +19,6 @@ is responsibility and contract, not every signature. | [`materialize`](materialize.md) | The driver: gates, scheduling, the save/restore loop, status | impure | | [`compute`](compute.md) | Resource requests, native allocation lifecycle, borrowed Dask clients | impure | | [`execution_resources`](compute.md) | Task resource admission | pure | -| [`gpu`](compute.md#gpu-discovery-and-visibility) | Native CUDA inventory and NVIDIA device paths | impure | | [`sandbox`](sandbox.md) | The exec boundary: policy, backends, attestation, denials | mixed | | [`image` & `container`](container.md) | The container hatch: declaration → image → archive → runtime | pure / impure | | [`crate`](crate.md) | The publication view: the repo as an RO-Crate | pure | diff --git a/docs/api/sandbox.md b/docs/api/sandbox.md index 3911b242..6a61e0e0 100644 --- a/docs/api/sandbox.md +++ b/docs/api/sandbox.md @@ -18,7 +18,7 @@ plus `lightcone/_sandbox_exec.py`, the Landlock shim. | `Capability` | What this host can do — `detect()`'s answer, the only `sys.platform` branch. | | `Attestation` | What was actually enforced, derived from the flags applied — never from what the matrix says should have happened. | | `Backend.wrap(policy, argv)` | The pure rewrite. `contains_prefix` declares whether the uv hop rides inside (a container is a world; a host mechanism trusts host plumbing). | -| `exec_policy(...)` | Shared policy builder with caller-supplied write scope and GPU UUIDs. Building it creates a private `$HOME`; `scope()` owns its cleanup. | +| `exec_policy(...)` | Shared policy builder with caller-supplied write scope and `use_gpus` setting. Building it creates a private `$HOME`; `scope()` owns its cleanup. | | `Unavailable` | A real backend that wraps to the same argv and attests `fs: open`. Saying so is the caller's job; pretending is nobody's. | | `denial.explain()` / `denial.trailer()` | Best-guess remedies (allowed to return nothing) and the unconditional trailer on every nonzero sandboxed exit. | @@ -26,17 +26,16 @@ An optional output receiver gets stdout/stderr byte chunks. Capturing output nev decodes or normalizes stdout; only the retained stderr tail is decoded for denial classification. Without a receiver, stdout remains inherited. -`exec_policy(..., gpu_devices=...)` carries allocated CUDA UUIDs into the command's -`CUDA_VISIBLE_DEVICES`, with an empty mask for CPU-only execution. Direct policies -grant existing NVIDIA character device nodes only for GPU commands. Native -permissions still apply; the CUDA mask controls cooperative visibility, not -hostile-code device isolation. No worker-wide environment mutation is involved. +`exec_policy(..., use_gpus=True)` passes the allocation's `CUDA_VISIBLE_DEVICES` +and optional `CUDA_DEVICE_ORDER` through unchanged. CPU commands receive an empty +CUDA mask. Direct GPU policies grant existing NVIDIA character device nodes. +Native permissions still apply; visibility is cooperative, and the reusable +worker's environment is never modified. -The pure OCI rewrite requests UUIDs through Docker's `--gpus device=...`, Podman's -`--device=nvidia.com/gpu=UUID`, or podman-hpc's `--gpu` plus the CUDA mask. CPU -containers also set `NVIDIA_VISIBLE_DEVICES=void`, overriding GPU-enabled image -defaults. Runtime prerequisites are documented under -[GPU allocations](../user/cluster.md#gpu-allocations). +The pure OCI rewrite adds podman-hpc's `--gpu` for GPU commands. Ordinary Docker +and Podman GPU execution is explicitly refused. CPU containers remain supported +on all runtimes and set `NVIDIA_VISIBLE_DEVICES=void` to override image defaults. +See [GPU allocations](../user/cluster.md#gpu-allocations). ## What must stay true diff --git a/docs/api/worker.md b/docs/api/worker.md index acd2e628..01d39541 100644 --- a/docs/api/worker.md +++ b/docs/api/worker.md @@ -20,16 +20,16 @@ Cluster execution supplies an output receiver to `materialize`/`execute`, which passes byte chunks from the sandbox back to the invocation. Standalone reruns retain direct terminal output. The driver submits each cluster task with its CPU, memory, and GPU reservations. Before resetting outputs, `execute` validates -resource syntax, checks native GPU visibility, and selects the requested number -of UUIDs. Standalone reruns apply the same checks but do not perform Dask resource -admission. Recipe `time_limit` is explicitly refused. +resource syntax and builds the command policy with GPU access enabled only when +the recipe requests it. Standalone reruns apply the same checks but do not perform +Dask resource admission. Recipe `time_limit` is explicitly refused. ## Key symbols | Symbol | Role | |---|---| | `materialize(root, task, context, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns a `TaskResult`, always. | -| `execute(root, task, input_versions, context)` | Validate resource syntax and device visibility, run a recipe unconditionally, then record its payload and manifest. | +| `execute(root, task, input_versions, context)` | Validate resource syntax and GPU policy, run a recipe unconditionally, then record its payload and manifest. | | `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, reason, and diagnostic notes. `.usable` is what dependents check. | | `main(argv)` | The rerun entry point: guards, converges the project environment from the commit's own lock, resolves its own HEAD and runtime, executes. | | `lc_version()` | The engine version every manifest records. | @@ -41,10 +41,10 @@ admission. Recipe `time_limit` is explicitly refused. make Dask abort every task in flight; reporting all independent failures in one run is most of what owning the loop buys. - **Device visibility belongs to each command.** CPU recipes receive an empty - `CUDA_VISIBLE_DEVICES`; GPU recipes receive only their selected UUIDs through + `CUDA_VISIBLE_DEVICES`; GPU recipes inherit the allocation's whole mask through the sandbox policy. Never modify the reusable worker's shared environment. Admission reserves the worker's full GPU budget for one GPU recipe at a time, - even when its requested visible count is smaller. + even when its minimum requested count is smaller. - **`data_version` is computed here, before anything is staged** — the dependent's argument *is* this return value, so the digest must exist while the files are still unannexed. Deriving it from @@ -77,5 +77,5 @@ admission. Recipe `time_limit` is explicitly refused. `tests/test_worker.py` — real recipes through the real boundary against a real repository (the `analysis` fixture): whether gates hold and bytes land are not questions a stub can answer. -`tests/test_gpu_execution.py` checks real subprocess masks and refusal before -output deletion using a simulated GPU inventory; it requires no physical GPU. +`tests/test_gpu_execution.py` checks real subprocess masks and unsupported runtime +refusal before output deletion; it requires no physical GPU. diff --git a/docs/architecture.md b/docs/architecture.md index 9e02f404..5eff3927 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -69,9 +69,9 @@ The division of labor is strict and load-bearing: budgets; submissions reserve the recipe's requirements. CPU and memory reservations coordinate scheduling rather than imposing per-recipe OS limits. Recipe `time_limit` is explicitly refused; allocation walltime remains supported. - A GPU recipe reserves the worker's full GPU budget and exposes only its - requested devices through a per-command CUDA mask. CPU recipes expose none; - probes reserve and expose the whole worker budget. + A GPU recipe reserves the worker's full GPU budget and inherits its allocation + mask, even when it requests fewer GPUs. CPU recipes expose none; probes reserve + the whole worker budget and inherit the allocation mask. - **Values are resolved once and handed down.** HEAD, the container runtime, and the foreign-write facts are read by the driver and passed to workers as values — a worker that asked git itself could @@ -145,7 +145,8 @@ records what was *actually* enforced — never what should have been. There is one policy builder, `exec_policy`: probes and recipes share environment and filesystem rules, with write scope and GPU visibility supplied by the caller. -Probes expose their reserved worker's GPUs; recipes expose their declared count. +GPU recipes and probes inherit the worker's allocation mask; CPU commands get an +empty mask. ## The container hatch @@ -165,9 +166,9 @@ config-blob id, never a tag. `engine.compute` owns allocation lifecycle through a small provider protocol. A YAML catalog supplies ordered resource offers and stable native service namespaces. When the implicit default file is absent, a built-in local catalog -provides one CPU and 1 GiB without setup, plus GPU offers grouped by model when -Linux CUDA discovery succeeds. An explicit catalog replaces those defaults; -missing explicit paths and invalid files remain errors. No catalog is written and +provides one CPU and 1 GiB without setup. GPU offers need an explicit catalog, +which replaces the built-in defaults. Missing explicit paths and invalid files +remain errors. No catalog is written and no allocation starts until `compute launch` resolves resources and submits once. Slurm queries and validated local OS identities are authoritative for allocations; Dask is the authority for connected workers. Private scheduler/TLS files are connection @@ -176,9 +177,9 @@ material, not a registry. Allocation requests use SkyPilot-style CPU/memory exact or minimum quantities and one accelerator type/count. Providers translate those requests into native allocations; Lightcone does not depend on SkyPilot or carry its GPU alias registry. -An isolated stdlib CUDA probe discovers native device UUIDs and model names; -stock Dask workers remain unchanged. GPU container access uses each runtime's -native mechanism. See [GPU setup](user/cluster.md#gpu-allocations). +Stock Dask workers inherit the allocation's native CUDA mask; Lightcone does not +probe GPU hardware. Container GPU access uses podman-hpc's native `--gpu` option. +See [GPU setup](user/cluster.md#gpu-allocations). `compute.connect(CLUSTER_ID)` borrows a standard Dask client and closes only that client on exit. Both execution commands require a cluster ID. The materialization diff --git a/docs/cli/compute.md b/docs/cli/compute.md index 23e05b81..b082434b 100644 --- a/docs/cli/compute.md +++ b/docs/cli/compute.md @@ -15,10 +15,9 @@ Without configuration, `resources` exposes a built-in `local` offer: one CPU, 1 GiB, one node, fast startup, and a 30-minute default lifetime (two-hour maximum). Launch it with `lc compute launch --cpus 1 --memory 1`; execution still requires the returned cluster name or its full immutable ID. -On Linux, visible NVIDIA GPUs also produce local GPU offers, grouped by model. -Use the accelerator names and counts shown by `resources`, or `GPU:N` to request -any model with exactly N GPUs per node. GPU discovery failure leaves the CPU -offer available. +GPU offers require an explicit catalog. On Linux, local GPU launches also require +an externally configured `CUDA_VISIBLE_DEVICES` mask; Lightcone does not discover +GPU hardware. See [GPU allocations](../user/cluster.md#gpu-allocations). `~/.lightcone/compute.yaml`, when present, replaces this built-in catalog. `LC_COMPUTE_CONFIG` selects another file for both compute and execution commands, @@ -66,7 +65,7 @@ all mean 16 GiB; `16GB+` permits a larger offer. `--gpus A100` means one, `--gpus GPU:4` accepts any GPU model, and the default `--gpus 0` selects CPU-only offers. Names match case-insensitively. GPU counts are positive whole numbers, with no `+` or fractional form. Lightcone does not -maintain SkyPilot's accelerator alias registry: copy local model names from +maintain SkyPilot's accelerator alias registry: use the labels configured in `resources` or use `GPU:N`. Catalog shapes use `accelerators: A100:4` or `accelerators: {A100: 4}`. See [GPU allocations](../user/cluster.md#gpu-allocations). diff --git a/docs/cli/materialize.md b/docs/cli/materialize.md index 0b7904bd..0f5fad9d 100644 --- a/docs/cli/materialize.md +++ b/docs/cli/materialize.md @@ -58,10 +58,10 @@ never touched, under any flag. termination through their runtime; see [execution limits](../user/cluster.md#execution-requirements-and-limits). - **Honors recipe resources.** CPU, memory, and GPU requests must fit one worker and are reserved through standard Dask scheduling. GPU recipes run one at a - time per worker and see only their requested devices. Recipes without `gpus` - see none. The whole selected graph is checked - before preparation or submission. Recipe `time_limit` is unsupported and - refused; allocation walltime remains supported. + time per worker and inherit the whole allocation's CUDA mask, which may expose + more GPUs than requested. Recipes without `gpus` see none. The whole selected + graph is checked before preparation or submission. Recipe `time_limit` is + unsupported and refused; allocation walltime remains supported. See [recipe resource requirements](../user/cluster.md#recipe-resource-requirements). - **Fetches what it needs.** Declared inputs whose annexed content is not in this clone are fetched before anything hashes. diff --git a/docs/cli/run.md b/docs/cli/run.md index deef5c44..07a66d7e 100644 --- a/docs/cli/run.md +++ b/docs/cli/run.md @@ -40,8 +40,8 @@ command, for example `lc run "$CLUSTER" -- env NAME=value python script.py`. Containerized commands use the image's environment and the sandbox overlays. The command reserves one worker's full CPU, memory, and GPU budgets for its duration. -Its CUDA mask exposes only the reserved devices; a CPU-only allocation exposes -none, even on a host with GPUs. A recipe instead declares its GPU count explicitly. +It inherits the allocation's whole CUDA mask; a CPU-only allocation exposes none, +even on a host with GPUs. A recipe declares its minimum GPU count explicitly. See [GPU allocations](../user/cluster.md#gpu-allocations) for container prerequisites. Interrupting the CLI detaches its client; the remote command may still be running. diff --git a/docs/user/cluster.md b/docs/user/cluster.md index a0b7dd90..98931195 100644 --- a/docs/user/cluster.md +++ b/docs/user/cluster.md @@ -12,10 +12,8 @@ No configuration is needed on a fresh installation. When one logical CPU, 1 GiB, one node, and fast startup. Its default lifetime is 30 minutes, with a maximum of two hours. This creates no catalog file and starts no processes until you launch a cluster. -On Linux, detected NVIDIA GPUs add a `local-gpu` offer, or `local-gpu-1`, -`local-gpu-2`, and so on for different models. Each groups the visible devices -of one model and keeps the same small CPU/memory defaults. Use a custom catalog -for larger CPU or RAM budgets. See [GPU allocations](#gpu-allocations). +Use a custom catalog for larger CPU or RAM budgets and for GPU offers. +See [GPU allocations](#gpu-allocations). ```bash lc compute resources @@ -315,7 +313,7 @@ Each worker's files go under `//attempt-/`. Before starting Dask, every rank checks that Slurm gave it what the plan requested: the node count, CPUs per task and its actual CPU affinity, and memory -per node. GPU allocations also validate the native GPU count and CUDA visibility. +per node. GPU allocations also validate Slurm's native GPU count. Ranks other than zero wait up to 120 seconds for the scheduler, which has as long to start. A failed check or timeout logs `Slurm Dask startup failed: …` to the submission log and exits nonzero. Look @@ -323,26 +321,39 @@ there when a job is active but never becomes ready. ## GPU allocations -GPU support currently covers whole NVIDIA CUDA devices on Linux. MIG devices, -fractional GPUs, and other accelerator vendors are not supported. CPU-only use -does not require CUDA. Discovery respects the allocation's native CUDA mask and -does not initialize CUDA in the reusable Dask worker. +GPU support uses NVIDIA CUDA devices on Linux. `lc compute resources` shows +configured accelerator types and counts; Lightcone does not probe CUDA or discover +local hardware. There is no fractional GPU or MIG management. -`lc compute resources` shows available accelerator types and counts. Requests -use SkyPilot-style `NAME[:COUNT]`: `--gpus A100` means one A100, -`--gpus A100:4` means exactly four, and `--gpus GPU:4` accepts any model with -exactly four. Names match case-insensitively; GPU counts do not accept `+`. -Omitting `--gpus`, or passing `0`, selects CPU-only offers. +Requests use SkyPilot-style `NAME[:COUNT]`: `--gpus A100` means one A100, +`--gpus A100:4` means exactly four, and `--gpus GPU:4` accepts any configured model +with exactly four. Names are case-insensitive catalog labels; GPU counts do not +accept `+`. Omitting `--gpus`, or passing `0`, selects CPU-only offers. -Names are catalog labels, without an accelerator alias registry. Local discovery -uses native model names with whitespace and punctuation normalized to hyphens, for example -`NVIDIA-A100-SXM4-80GB`. Copy the type shown by `resources`, or use `GPU:N`. -Local allocations do not reserve GPUs exclusively against other allocations or -programs on the host. +For local GPUs, add an offer to the [workstation catalog above](#customize-resource-offers): -For example, a Slurm site could expose this additional offer under `offers`. -The account, shape, constraint, and native GPU type must be adjusted to the site; -this is an illustrative configuration, not a tested hardware deployment: +```yaml + - name: workstation-gpu + connection: workstation + resources: {cpus: 4, memory: 8GB, accelerators: 'GPU:1'} + max_nodes: 1 + time: {default: 30m, max: 2h} + startup: fast +``` + +Set the devices available to that allocation when launching it: + +```bash +CUDA_VISIBLE_DEVICES=0 lc compute launch --cpus 4 --memory 8GB --gpus GPU:1 +``` + +Lightcone freezes the nonempty mask and `CUDA_DEVICE_ORDER`, if set, at launch. +You are responsible for matching the catalog's count and model to those devices; +Lightcone does not verify them. Local allocations do not reserve GPUs exclusively +against other allocations or programs on the host. + +On Slurm, the offer's `config.gpu_type` maps its catalog label to a native GRES +type. For example, add this offer with settings adjusted to your site: ```yaml - name: gpu-batch @@ -357,28 +368,24 @@ this is an illustrative configuration, not a tested hardware deployment: gpu_type: a100 ``` -With such an offer configured, inspect its plan before launching: - ```bash lc compute launch --cpus 32 --memory 128GB --gpus A100:4 --dry-run ``` -GPU commands receive selected device UUIDs through `CUDA_VISIBLE_DEVICES`. -CPU recipes receive an empty mask. This is cooperative device selection; -native OS/cgroup permissions remain the access boundary. Container runtimes -also need their normal GPU integration: +This is an illustrative shape, not a tested site configuration. Slurm allocates +GPUs through GRES; Lightcone checks the native count and passes Slurm's CUDA mask +through unchanged, using `CUDA_DEVICE_ORDER=PCI_BUS_ID`. -- **Docker:** install the NVIDIA Container Toolkit; Lightcone supplies explicit - UUIDs through `--gpus`. See [NVIDIA's Docker setup](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html#configuring-docker). -- **Podman:** configure NVIDIA CDI with UUID device names. Lightcone supplies - `--device=nvidia.com/gpu=UUID`; see [NVIDIA's CDI guide](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/cdi-support.html). -- **podman-hpc:** Lightcone adds `--gpu` and the selected CUDA mask; see - [NERSC's GPU container guidance](https://docs.nersc.gov/development/containers/podman-hpc/overview/#using-nvidia-gpus-in-podman-hpc). +GPU commands inherit the allocation's whole CUDA mask. CPU commands receive an +empty mask. These are cooperative visibility settings; native OS and cgroup +permissions remain authoritative. -CPU containers also override `NVIDIA_VISIBLE_DEVICES` to `void`, so a CUDA image's -default cannot enable GPU injection. GPU discovery, reservations, masks, and -runtime arguments are tested with simulated inventories and real subprocesses; -physical GPU execution has not yet been validated. +Containerized GPU execution currently supports **podman-hpc** through its native +`--gpu` option. See [NERSC's GPU container guidance](https://docs.nersc.gov/development/containers/podman-hpc/overview/#using-nvidia-gpus-in-podman-hpc). +GPU requests with ordinary Docker or Podman are explicitly refused. CPU execution +continues to support all three runtimes and sets `NVIDIA_VISIBLE_DEVICES=void` to +override GPU-enabled image defaults. Physical GPU execution remains a deployment +validation step. ## Recipe resource requirements @@ -407,11 +414,12 @@ one such recipe runs per worker. Recipe `gpus` is a nonnegative whole count, defaulting to zero; accelerator type selection belongs to cluster allocation. A GPU recipe reserves the worker's -entire GPU budget, so only one GPU recipe runs on that worker at a time, while -its command sees only the requested number of devices. CPU recipes may still -run alongside it when CPU, memory, and task slots permit; their CUDA mask is -empty. `lc run` reserves the worker's entire CPU, memory, and GPU budgets and -exposes only those reserved GPUs. +entire GPU budget, so only one GPU recipe runs on that worker at a time. The +requested count is a minimum capacity requirement: the command inherits the +worker's whole allocated CUDA mask and may see more GPUs than requested. CPU +recipes may still run alongside it when CPU, memory, and task slots permit; their +CUDA mask is empty. `lc run` reserves the worker's entire CPU, memory, and GPU +budgets and inherits that same allocation mask. Recipe `time_limit` is not supported and is refused before preparation or execution. Set the allocation lifetime with `lc compute launch --time` instead. diff --git a/src/lightcone/engine/compute/catalog.py b/src/lightcone/engine/compute/catalog.py index 6f6d1f28..a9b92aa8 100644 --- a/src/lightcone/engine/compute/catalog.py +++ b/src/lightcone/engine/compute/catalog.py @@ -10,9 +10,6 @@ import yaml from pydantic import Field, ValidationError, model_validator -from lightcone.engine import gpu -from lightcone.engine.project import ProjectError - from .model import ( ComputeError, ComputeModel, @@ -91,29 +88,15 @@ def load(cls, path: Path | None = None) -> Catalog: except FileNotFoundError as exc: if configured or path.is_symlink(): raise ComputeError(f"cannot read compute catalog {path}: {exc}") from exc - local = Offer( - name="local", connection="local", - resources=Resources(cpus=1, memory_gib=Decimal(1)), - max_nodes=1, time=TimeLimits(default="30m", max="2h"), - startup=Startup(class_="fast"), - ) - offers = [local] - try: - devices = gpu.inventory() - except ProjectError: - devices = () # Optional discovery must not disable CPU-only execution. - names = sorted({device.name for device in devices}) - for index, name in enumerate(names, 1): - offers.append(local.replace( - name="local-gpu" if len(names) == 1 else f"local-gpu-{index}", - resources=local.resources.replace(accelerators={ - name: sum(device.name == name for device in devices), - }), - )) return cls( version=1, connections={"local": Connection(namespace=_LOCAL_NAMESPACE, provider="local")}, - offers=offers, + offers=[Offer( + name="local", connection="local", + resources=Resources(cpus=1, memory_gib=Decimal(1)), + max_nodes=1, time=TimeLimits(default="30m", max="2h"), + startup=Startup(class_="fast"), + )], ) except (OSError, UnicodeError, yaml.YAMLError) as exc: raise ComputeError(f"cannot read compute catalog {path}: {exc}") from exc diff --git a/src/lightcone/engine/compute/local.py b/src/lightcone/engine/compute/local.py index a99ac4ed..9d83cc38 100644 --- a/src/lightcone/engine/compute/local.py +++ b/src/lightcone/engine/compute/local.py @@ -19,7 +19,6 @@ import psutil -from lightcone.engine import gpu from lightcone.engine.compute.model import ( ComputeError, Connection, @@ -42,7 +41,6 @@ read_private_json, write_private_json, ) -from lightcone.engine.project import ProjectError _OWNER_MODULE = "lightcone.engine.compute.local_runtime" _STOP_GRACE = 3.0 @@ -106,21 +104,15 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: if offer.resources.cpus > CPU_COUNT or offer.resources.memory_bytes > MEMORY_LIMIT: raise UnavailableOfferError("the local offer exceeds this host's CPU or RAM capacity") - try: - inventory = gpu.inventory() if offer.resources.gpus else () - except ProjectError as exc: - raise UnavailableOfferError(str(exc)) from exc - accelerator = offer.resources.accelerator_name or "GPU" - available = tuple( - device for device in inventory - if accelerator.casefold() == "gpu" or device.name.casefold() == accelerator.casefold() - ) - if len(available) < offer.resources.gpus: - raise UnavailableOfferError( - f"the local offer exceeds this host's visible CUDA capacity for {accelerator}" - ) - selected = available[:offer.resources.gpus] - names = {device.name for device in selected} + mask = "" + if offer.resources.gpus: + if sys.platform != "linux": + raise UnavailableOfferError("local GPU allocations require Linux") + mask = os.environ.get("CUDA_VISIBLE_DEVICES", "") + if not mask: + raise UnavailableOfferError( + "local GPU offers require an explicit nonempty CUDA_VISIBLE_DEVICES mask" + ) seconds = request.seconds if request.seconds is not None else offer.time.default_seconds if seconds <= 0 or seconds > offer.time.max_seconds: raise ComputeError("local allocations require a finite time within the offer's limit") @@ -140,8 +132,8 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: "connection_root": str(self.root), "scratch_root": str(scratch), "task_slots_per_node": slots, - "gpu_devices": tuple(device.uuid for device in selected), - "accelerator_name": next(iter(names)) if len(names) == 1 else "GPU", + "cuda_visible_devices": mask, + "cuda_device_order": os.environ.get("CUDA_DEVICE_ORDER"), "resource_enforcement": "cooperative; no exclusive CPU, RAM, or GPU reservation", "termination_grace_seconds": _STOP_GRACE, }, @@ -153,11 +145,6 @@ def launch(self, plan: LaunchPlan) -> Identity: raise ComputeError("local launch plan belongs to a different connection or node count") if plan.name is not None: validate_name(plan.name) - devices = tuple(plan.details["gpu_devices"]) - if len(devices) != plan.resources.gpus or ( - devices and not set(devices).issubset(gpu.visible_devices()) - ): - raise UnavailableOfferError("the local plan's selected CUDA GPUs are no longer visible") boot = _boot_identity() token = uuid4().hex directory = private_directory(self.root / token, create=True) @@ -174,6 +161,11 @@ def launch(self, plan: LaunchPlan) -> Identity: ) process: subprocess.Popen[bytes] | None = None identity: Identity | None = None + environment = {**os.environ, "CUDA_VISIBLE_DEVICES": plan.details["cuda_visible_devices"]} + if plan.details["cuda_device_order"] is None: + environment.pop("CUDA_DEVICE_ORDER", None) + else: + environment["CUDA_DEVICE_ORDER"] = plan.details["cuda_device_order"] try: # This allocation outlives a command; the ordinary run-to-completion # subprocess seam cannot own it. Logs are discarded rather than grow. @@ -184,7 +176,7 @@ def launch(self, plan: LaunchPlan) -> Identity: stderr=subprocess.DEVNULL, start_new_session=True, close_fds=True, - env={**os.environ, "CUDA_VISIBLE_DEVICES": ",".join(devices)}, + env=environment, ) identity = Identity( namespace=self.connection.namespace, native_id=str(process.pid), token=token, @@ -200,7 +192,7 @@ def launch(self, plan: LaunchPlan) -> Identity: "cpus": plan.resources.cpus, "memory": plan.resources.memory_bytes, "gpus": plan.resources.gpus, - "accelerator_name": plan.details["accelerator_name"], + "accelerator_name": plan.resources.accelerator_name or "GPU", } write_private_json(directory / "identity.json", record) # The child waits for this file before publishing its TLS connection. diff --git a/src/lightcone/engine/compute/local_runtime.py b/src/lightcone/engine/compute/local_runtime.py index e23d83b8..222a810f 100644 --- a/src/lightcone/engine/compute/local_runtime.py +++ b/src/lightcone/engine/compute/local_runtime.py @@ -11,7 +11,6 @@ from pathlib import Path from types import FrameType -from lightcone.engine.compute.model import ComputeError from lightcone.engine.compute.runtime import ( SCHEDULER_CONFIG, create_security, @@ -57,11 +56,8 @@ def expire(_signum: int, _frame: FrameType | None) -> None: security = create_security(directory) allocation = read_private_json(directory / "identity.json") gpus = int(allocation["gpus"]) - if gpus: - from lightcone.engine.gpu import visible_devices - - if len(visible_devices()) != gpus: - raise ComputeError("visible CUDA GPUs do not match the local allocation envelope") + if not gpus: + os.environ["CUDA_VISIBLE_DEVICES"] = "" with dask.config.set(SCHEDULER_CONFIG), LocalCluster( # type: ignore[no-untyped-call] n_workers=1, threads_per_worker=int(launch["task_slots"]), diff --git a/src/lightcone/engine/compute/slurm_bootstrap.py b/src/lightcone/engine/compute/slurm_bootstrap.py index 35a718ef..3db01229 100644 --- a/src/lightcone/engine/compute/slurm_bootstrap.py +++ b/src/lightcone/engine/compute/slurm_bootstrap.py @@ -14,7 +14,6 @@ from typing import Any from uuid import UUID -from lightcone.engine import gpu from lightcone.engine.compute.model import ComputeError, Connection, Identity from lightcone.engine.compute.runtime import ( SCHEDULER_CONFIG, @@ -67,10 +66,10 @@ def _allocation(args: argparse.Namespace) -> tuple[Identity, int, int]: native_gpus = os.environ.get("SLURM_GPUS_ON_NODE", "") if not native_gpus.isdigit() or int(native_gpus) < args.gpus: raise ComputeError("native per-node GPUs do not match the allocation envelope") + if not os.environ.get("CUDA_VISIBLE_DEVICES"): + raise ComputeError("GPU allocations require a native CUDA_VISIBLE_DEVICES mask") # Slurm/NVML numbers devices in PCI order; CUDA's default is different. os.environ["CUDA_DEVICE_ORDER"] = "PCI_BUS_ID" - if len(gpu.visible_devices()) < args.gpus: - raise ComputeError("visible CUDA GPUs are fewer than the allocation envelope") restarts = os.environ.get("SLURM_RESTART_COUNT", "0") if not restarts.isdigit(): raise ComputeError("invalid native Slurm restart count") diff --git a/src/lightcone/engine/container.py b/src/lightcone/engine/container.py index c1d48f3c..8f3f01d8 100644 --- a/src/lightcone/engine/container.py +++ b/src/lightcone/engine/container.py @@ -29,7 +29,6 @@ import sys import tarfile import tempfile -from collections.abc import Sequence from dataclasses import dataclass from pathlib import Path from typing import Literal, cast @@ -410,7 +409,7 @@ def converge(runtime: Runtime) -> list[str]: def policy_for( runtime: Runtime, read_paths: list[Path], *, write_dir: Path | None = None, - gpu_devices: Sequence[str] = (), + use_gpus: bool = False, ) -> sandbox.Policy: """Build the exec policy for a resolved runtime. @@ -424,18 +423,22 @@ def policy_for( runtime: The resolved runtime. read_paths: Declared inputs, as :func:`sandbox.exec_policy` takes. write_dir: The directory a recipe's output lands in; absent for a probe. - gpu_devices: Allocated CUDA device UUIDs to expose to this command. + use_gpus: Inherit the allocation's CUDA mask; otherwise hide GPUs. Returns: The policy for this world. """ + if use_gpus and runtime.mode == "containerized": + from lightcone.engine.sandbox.oci import require_gpu_runtime + + require_gpu_runtime(runtime.runtime) return sandbox.exec_policy( runtime.root, read_paths=read_paths, env_dir=runtime.env_dir, containerized=runtime.mode == "containerized", write_dir=write_dir, - gpu_devices=gpu_devices, + use_gpus=use_gpus, ) diff --git a/src/lightcone/engine/execution_resources.py b/src/lightcone/engine/execution_resources.py index ef6513f6..793e95ec 100644 --- a/src/lightcone/engine/execution_resources.py +++ b/src/lightcone/engine/execution_resources.py @@ -13,7 +13,7 @@ class TaskResources(BaseModel): - """Reserve CPU, memory, and GPU capacity on stock Dask workers.""" + """Reserve CPU, memory, and minimum GPU capacity on stock Dask workers.""" model_config = ConfigDict(frozen=True, strict=True, extra="forbid") @@ -81,7 +81,8 @@ def requirements( Returns: Dask's numeric ``CPU``, ``MEMORY``, and optional ``GPU`` reservations. GPU recipes reserve the worker's full GPU budget, so only one GPU - recipe uses that worker's device mask at a time. + recipe uses that worker's native device mask at a time. The requested + count is a minimum capacity, not a per-command visibility limit. Raises: ProjectError: Capacity is unknown, a request cannot fit, or an diff --git a/src/lightcone/engine/gpu.py b/src/lightcone/engine/gpu.py deleted file mode 100644 index ee280fda..00000000 --- a/src/lightcone/engine/gpu.py +++ /dev/null @@ -1,141 +0,0 @@ -"""Discover CUDA-visible GPU identities without initializing CUDA in a worker. - -CUDA owns device ordering and native visibility masks. A short isolated process -asks the driver for UUIDs; interpreting numeric masks ourselves would confuse -Slurm's allocation-local indices with host device numbers. -""" - -from __future__ import annotations - -import json -import os -import re -import subprocess -import sys -from pathlib import Path -from typing import NamedTuple - -_UUID = re.compile(r"GPU-[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}") -_DEVICE_ROOT = Path("/dev") - - -class Device(NamedTuple): - """A native CUDA identity and its model name, with punctuation made CLI-safe.""" - - uuid: str - name: str - - -def inventory() -> tuple[Device, ...]: - """Return devices visible under this process's native CUDA mask. - - Missing CUDA drivers or an empty visible set return no devices. Broken - drivers and unsupported partitioned GPUs raise instead of inventing capacity. - - Returns: - Native UUIDs and model names, in CUDA visibility order. - - Raises: - ProjectError: CUDA discovery failed or returned an untrustworthy inventory. - """ - from lightcone.engine.project import ProjectError - - if os.environ.get("CUDA_VISIBLE_DEVICES") == "" or sys.platform != "linux": - return () - try: - result = subprocess.run( - [sys.executable, "-I", str(Path(__file__))], - stdin=subprocess.DEVNULL, capture_output=True, text=True, timeout=15, - ) - if result.returncode: - raise ValueError(result.stderr.strip()[:1024] or "CUDA probe exited unsuccessfully") - devices = json.loads(result.stdout) - if not isinstance(devices, list): - raise ValueError("CUDA probe returned invalid GPU identities") - result_devices = [] - for item in devices: - if ( - not isinstance(item, dict) or item.keys() != {"uuid", "name"} - or not isinstance(item["uuid"], str) or not _UUID.fullmatch(item["uuid"]) - or not isinstance(item["name"], str) - or not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]*", item["name"]) - ): - raise ValueError("CUDA probe returned invalid GPU identities") - result_devices.append(Device(**item)) - if len({device.uuid for device in result_devices}) != len(result_devices): - raise ValueError("CUDA probe returned duplicate GPU identities") - return tuple(result_devices) - except (OSError, subprocess.SubprocessError, ValueError) as exc: - raise ProjectError(f"cannot discover visible NVIDIA GPUs: {exc}") from exc - - -def visible_devices() -> tuple[str, ...]: - """Return native CUDA UUIDs suitable for a subprocess visibility mask.""" - return tuple(device.uuid for device in inventory()) - - -def device_paths() -> tuple[Path, ...]: - """Return existing NVIDIA device nodes needed by direct CUDA commands. - - These grants retain native OS/cgroup permissions. CUDA visibility controls - cooperative device selection; it is not an additional device-isolation layer. - """ - candidates = [ - *(_DEVICE_ROOT / name for name in ("nvidiactl", "nvidia-uvm", "nvidia-uvm-tools")), - *_DEVICE_ROOT.glob("nvidia[0-9]*"), - *(_DEVICE_ROOT / "nvidia-caps").glob("*"), - ] - return tuple(sorted(path for path in candidates if path.is_char_device())) - - -def _probe() -> list[dict[str, str]]: - # Keep this child stdlib-only: CUDA initialization never enters the CLI, - # allocation owner, or reusable Dask worker. No context or memory is allocated. - import ctypes - from uuid import UUID - - try: - driver = ctypes.CDLL("libcuda.so.1") - except OSError: - return [] - - def check(code: int) -> None: - if code: - raise RuntimeError(f"CUDA driver returned error {code}") - - driver.cuInit.argtypes = [ctypes.c_uint] - status = driver.cuInit(0) - if status == 100: # CUDA_ERROR_NO_DEVICE - return [] - check(status) - driver.cuDeviceGetCount.argtypes = [ctypes.POINTER(ctypes.c_int)] - driver.cuDeviceGet.argtypes = [ctypes.POINTER(ctypes.c_int), ctypes.c_int] - driver.cuDeviceGetUuid.argtypes = [ctypes.c_void_p, ctypes.c_int] - driver.cuDeviceGetUuid_v2.argtypes = [ctypes.c_void_p, ctypes.c_int] - driver.cuDeviceGetName.argtypes = [ctypes.c_void_p, ctypes.c_int, ctypes.c_int] - count = ctypes.c_int() - check(driver.cuDeviceGetCount(ctypes.byref(count))) - devices = [] - for ordinal in range(count.value): - device = ctypes.c_int() - check(driver.cuDeviceGet(ctypes.byref(device), ordinal)) - physical, instance = ctypes.create_string_buffer(16), ctypes.create_string_buffer(16) - check(driver.cuDeviceGetUuid(physical, device)) - check(driver.cuDeviceGetUuid_v2(instance, device)) - if physical.raw != instance.raw: - raise RuntimeError("partitioned (MIG) GPUs are not supported; request whole GPUs") - name = ctypes.create_string_buffer(256) - check(driver.cuDeviceGetName(name, len(name), device)) - devices.append({ - "uuid": f"GPU-{UUID(bytes=physical.raw)}", - "name": re.sub(r"[^A-Za-z0-9_.-]+", "-", name.value.decode("ascii")).strip("-"), - }) - return devices - - -if __name__ == "__main__": - try: - print(json.dumps(_probe())) - except Exception as error: - print(str(error), file=sys.stderr) - raise SystemExit(1) from error diff --git a/src/lightcone/engine/run.py b/src/lightcone/engine/run.py index 21f4b688..2110cc20 100644 --- a/src/lightcone/engine/run.py +++ b/src/lightcone/engine/run.py @@ -20,7 +20,7 @@ from typing import Any from uuid import uuid4 -from lightcone.engine import container, gpu, sandbox +from lightcone.engine import container, sandbox from lightcone.engine.execution_resources import TaskResources from lightcone.engine.project import ( SPEC_FILENAME, @@ -61,7 +61,7 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. with forwarding(client) as output: future = client.submit( call, _probe, output.topic, "probe", runtime, paths, tuple(command), - int(resources.get("GPU", 0)), + resources.get("GPU", 0) > 0, key=f"lc-{invocation}-probe", pure=False, resources=resources, ) try: @@ -81,17 +81,11 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. def _probe( runtime: container.Runtime, paths: list[Path], command: tuple[str, ...], - gpu_count: int, + use_gpus: bool, *, output: Callable[[str, bytes], None], ) -> sandbox.Outcome: """Execute the prepared probe; the driver alone converges its environment.""" - gpu_devices = gpu.visible_devices()[:gpu_count] if gpu_count else () - if len(gpu_devices) < gpu_count: - raise ProjectError( - f"probe reserved {gpu_count} GPUs but this worker can access " - f"only {len(gpu_devices)} CUDA devices" - ) - built = container.policy_for(runtime, paths, gpu_devices=gpu_devices) + built = container.policy_for(runtime, paths, use_gpus=use_gpus) with sandbox.scope(built) as policy: outcome = sandbox.run( container.backend(runtime), policy, command, cwd=runtime.root, diff --git a/src/lightcone/engine/sandbox/oci.py b/src/lightcone/engine/sandbox/oci.py index 1debe632..dd206f69 100644 --- a/src/lightcone/engine/sandbox/oci.py +++ b/src/lightcone/engine/sandbox/oci.py @@ -22,6 +22,7 @@ from pathlib import Path from typing import Literal +from lightcone.engine.project import ProjectError from lightcone.engine.sandbox.boundary import SANDBOX_ENV from lightcone.engine.sandbox.model import Attestation, Capability, Policy @@ -30,6 +31,14 @@ OCIRuntime = Literal["podman", "docker", "podman-hpc"] +def require_gpu_runtime(runtime: str) -> None: + """Refuse runtimes that need device translation outside the native allocation.""" + if runtime != "podman-hpc": + raise ProjectError( + "GPU containers require podman-hpc; Docker/Podman GPUs are not supported" + ) + + @dataclass(frozen=True) class OCIBackend: """A container runtime, expressed as an argv rewrite.""" @@ -81,21 +90,16 @@ def wrap(self, policy: Policy, argv: Sequence[str]) -> list[str]: # host-layout collision `_write_roots` documents for direct mode. mounts = [f"--volume={path.resolve()}:{path}:ro" for path in policy.read] mounts += [f"--volume={path.resolve()}:{path}:rw" for path in policy.write] - devices = policy.env.get("CUDA_VISIBLE_DEVICES", "") + gpu_mask = policy.env.get("CUDA_VISIBLE_DEVICES", "") environment = dict(policy.env) - if not devices: + if not gpu_mask: # CUDA images may default to all devices under an NVIDIA runtime. environment["NVIDIA_VISIBLE_DEVICES"] = "void" overlay = [f"--env={k}={v}" for k, v in sorted(environment.items())] gpu_flags = [] - if devices: - if self.runtime == "podman-hpc": - gpu_flags = ["--gpu"] - elif self.runtime == "docker": - # Docker parses this argument as CSV, including its quotes. - gpu_flags = ["--gpus", f'"device={devices}"'] - else: - gpu_flags = [f"--device=nvidia.com/gpu={device}" for device in devices.split(",")] + if gpu_mask: + require_gpu_runtime(self.runtime) + gpu_flags = ["--gpu"] return [ self.runtime, "run", "--rm", "--entrypoint", "", diff --git a/src/lightcone/engine/sandbox/policy.py b/src/lightcone/engine/sandbox/policy.py index 71b5a94b..a6627e3b 100644 --- a/src/lightcone/engine/sandbox/policy.py +++ b/src/lightcone/engine/sandbox/policy.py @@ -32,7 +32,7 @@ from fnmatch import fnmatch from pathlib import Path -from lightcone.engine import gpu +from lightcone.engine.project import ProjectError from lightcone.engine.sandbox.model import Policy #: The utility tier of the exec allowlist. A maintained policy @@ -175,6 +175,18 @@ def utility(name: str) -> Path | None: "TMPDIR": ".tmp", } +_DEVICE_ROOT = Path("/dev") + + +def _gpu_device_paths() -> tuple[Path, ...]: + """Grant existing NVIDIA character devices, retaining native OS/cgroup limits.""" + candidates = [ + *(_DEVICE_ROOT / name for name in ("nvidiactl", "nvidia-uvm", "nvidia-uvm-tools")), + *_DEVICE_ROOT.glob("nvidia[0-9]*"), + *(_DEVICE_ROOT / "nvidia-caps").glob("*"), + ] + return tuple(sorted(path for path in candidates if path.is_char_device())) + def exec_policy( project: Path, @@ -183,7 +195,7 @@ def exec_policy( env_dir: Path | None = None, containerized: bool = False, write_dir: Path | None = None, - gpu_devices: Sequence[str] = (), + use_gpus: bool = False, ) -> Policy: """Build what a sandboxed command may touch. @@ -213,7 +225,7 @@ def exec_policy( directory holding its output file, shared with the siblings declared beside it. Absent for a probe, which has no analysis node and gets the project's own ``results/`` whole. - gpu_devices: CUDA device UUIDs allocated to this command; empty hides GPUs. + use_gpus: Inherit the allocation's CUDA mask; otherwise hide GPUs. Returns: The policy. The in-tree write scope is granted only if it exists — @@ -222,6 +234,9 @@ def exec_policy( disk; the caller owns removing it (see :func:`~lightcone.engine.sandbox.boundary.scope`). """ + gpu_mask = os.environ.get("CUDA_VISIBLE_DEVICES", "") if use_gpus else "" + if use_gpus and not gpu_mask: + raise ProjectError("GPU execution requires a nonempty allocation CUDA_VISIBLE_DEVICES mask") env_dir = env_dir if env_dir is not None else project / ".venv" # The containerized HOME lives under the project's own (gitignored) # `.lightcone/`, not the system temp dir: it is a mount source, and @@ -238,7 +253,9 @@ def exec_policy( in_tree_write = write_dir if write_dir is not None else project / "results" overlay = home_overlay(tmp_home, env_dir, containerized=containerized) - overlay["CUDA_VISIBLE_DEVICES"] = ",".join(gpu_devices) + overlay["CUDA_VISIBLE_DEVICES"] = gpu_mask + if use_gpus and "CUDA_DEVICE_ORDER" in os.environ: + overlay["CUDA_DEVICE_ORDER"] = os.environ["CUDA_DEVICE_ORDER"] if containerized: # Declared spellings, not realpaths — the one shape that keeps # its paths unresolved. These become mount *destinations*, and a @@ -258,7 +275,7 @@ def exec_policy( # EXECUTE on the interpreter *file*; READ on the install root beside # it, for the stdlib. See :func:`_venv_python` and :func:`_stdlib_root`. stdlib = _stdlib_root(python) - devices = gpu.device_paths() if gpu_devices else () + devices = _gpu_device_paths() if use_gpus else () write = _existing([tmp_home, in_tree_write, *_write_roots(project), *devices]) read = _existing([project, *read_paths, *stdlib, *(Path(p) for p in _OS_READ_BASELINE)]) diff --git a/src/lightcone/engine/worker.py b/src/lightcone/engine/worker.py index 6189f95d..e6662b8b 100644 --- a/src/lightcone/engine/worker.py +++ b/src/lightcone/engine/worker.py @@ -35,7 +35,7 @@ from pathlib import Path from typing import Literal -from lightcone.engine import assets, container, dataset, gpu, identity, plan, project, sandbox +from lightcone.engine import assets, container, dataset, identity, plan, project, sandbox from lightcone.engine.execution_resources import TaskResources from lightcone.engine.plan import Key, Task from lightcone.engine.project import ( @@ -229,12 +229,6 @@ def execute( nothing and never touches git beyond reading HEAD. """ resources = TaskResources.parse(task.resources) - gpu_devices = gpu.visible_devices()[:resources.gpus] if resources.gpus else () - if len(gpu_devices) < resources.gpus: - raise ProjectError( - f"recipe requests {resources.gpus} GPUs but this worker can access " - f"only {len(gpu_devices)} CUDA devices" - ) if moved := _gate(root, context.env_version): return TaskResult(task.key, "failed", reason=moved) @@ -247,17 +241,18 @@ def execute( # cannot reach a sibling, a longer id, another output's sidecar, or a # scope directory of the same name. task.output_path.parent.mkdir(parents=True, exist_ok=True) - task.manifest_path.unlink(missing_ok=True) - for stale in task.output_path.parent.glob(f"{task.output_id}.*"): - if stale.is_file() or stale.is_symlink(): - stale.unlink() - read_paths = [p for p in task.inputs.values() if p.exists()] policy = container.policy_for( - context.runtime, read_paths, write_dir=task.output_path.parent, gpu_devices=gpu_devices, + context.runtime, read_paths, write_dir=task.output_path.parent, + use_gpus=resources.gpus > 0, ) - started_at = _now() with sandbox.scope(policy): + # Validate device visibility and container support before removing outputs. + task.manifest_path.unlink(missing_ok=True) + for stale in task.output_path.parent.glob(f"{task.output_id}.*"): + if stale.is_file() or stale.is_symlink(): + stale.unlink() + started_at = _now() outcome = sandbox.run( container.backend(context.runtime), policy, diff --git a/tests/test_compute.py b/tests/test_compute.py index 351b9738..da51a264 100644 --- a/tests/test_compute.py +++ b/tests/test_compute.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import subprocess from decimal import Decimal, localcontext from pathlib import Path from unittest.mock import MagicMock @@ -34,7 +35,6 @@ memory_bytes, validate_name, ) -from lightcone.engine.gpu import Device NAMESPACE = "5a9d058c-7c6e-4e2a-919b-786f1148536c" IDENTITY = Identity(namespace=NAMESPACE, native_id="1234", token="abc") @@ -42,7 +42,6 @@ @pytest.fixture def default_home(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: - monkeypatch.setattr("lightcone.engine.gpu.inventory", lambda: ()) expanduser = Path.expanduser def expand(path: Path) -> Path: @@ -456,51 +455,27 @@ def test_accelerator_selection_honors_type_and_exact_count( compute.Compute().plan(Request.parse("4", "8", gpus="A100:4")) -def test_builtin_gpu_offer_is_optional_and_does_not_replace_cpu_offer( +def test_builtin_catalog_stays_cpu_only_without_probing_native_gpus( default_home: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: - monkeypatch.setattr("lightcone.engine.gpu.inventory", lambda: ( - Device("GPU-one", "A100"), Device("GPU-two", "A100"), - )) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1") + probe = MagicMock(side_effect=AssertionError("catalog loading must not probe GPU hardware")) + monkeypatch.setattr(subprocess, "run", probe) loaded = Catalog.load() assert [(offer.name, offer.resources.gpus) for offer in loaded.offers] == [ - ("local", 0), ("local-gpu", 2), + ("local", 0), ] - assert loaded.offers[1].connection == loaded.offers[0].connection - assert loaded.offers[1].resources.accelerator_name == "A100" + probe.assert_not_called() assert not list(default_home.iterdir()) -def test_failed_optional_gpu_discovery_keeps_builtin_cpu_offer( - default_home: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - def failed() -> tuple[Device, ...]: - raise ComputeError("CUDA driver could not enumerate visible devices") - - monkeypatch.setattr("lightcone.engine.gpu.inventory", failed) - assert [(offer.name, offer.resources.gpus) for offer in Catalog.load().offers] == [("local", 0)] - - -def test_builtin_mixed_gpu_inventory_exposes_each_model_separately( - default_home: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - monkeypatch.setattr("lightcone.engine.gpu.inventory", lambda: ( - Device("GPU-one", "H100"), Device("GPU-two", "A100"), Device("GPU-three", "H100"), - )) - loaded = Catalog.load() - assert [(offer.name, offer.resources.as_dict()["accelerators"]) for offer in loaded.offers] == [ - ("local", None), ("local-gpu-1", {"A100": 1}), ("local-gpu-2", {"H100": 2}), - ] - - def test_configured_catalogs_do_not_probe_local_gpus( catalog: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: - def unexpected() -> tuple[Device, ...]: - pytest.fail("configured catalogs must not discover ambient local GPUs") - - monkeypatch.setattr("lightcone.engine.gpu.inventory", unexpected) + probe = MagicMock(side_effect=AssertionError("catalog loading must not probe GPU hardware")) + monkeypatch.setattr(subprocess, "run", probe) assert Catalog.load().offers + probe.assert_not_called() def test_cli_gpu_request_and_resource_output(catalog: Path, provider: MagicMock) -> None: diff --git a/tests/test_compute_local.py b/tests/test_compute_local.py index 2dea6470..4aee5a15 100644 --- a/tests/test_compute_local.py +++ b/tests/test_compute_local.py @@ -21,7 +21,6 @@ import psutil import pytest -from lightcone.engine import gpu from lightcone.engine.compute import Compute, local, local_runtime from lightcone.engine.compute.catalog import Catalog from lightcone.engine.compute.local import LocalProvider @@ -41,7 +40,6 @@ read_private_json, write_private_json, ) -from lightcone.engine.project import ProjectError @pytest.fixture @@ -815,63 +813,78 @@ def test_local_plan_does_not_infer_policy_from_login_hostname_or_slurm_environme @pytest.mark.parametrize("gpus", [0, 1]) -def test_local_gpu_plan_freezes_devices_and_publishes_the_selected_envelope( +@pytest.mark.parametrize("order", [None, "PCI_BUS_ID", "FASTEST_FIRST"]) +def test_local_gpu_plan_freezes_native_mask_and_publishes_the_configured_envelope( provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, gpus: int, + order: str | None, ) -> None: - devices = (f"GPU-{uuid4()}", f"GPU-{uuid4()}") - visible = MagicMock(return_value=tuple(gpu.Device(item, "NVIDIA-A100") for item in devices)) - monkeypatch.setattr(gpu, "inventory", visible) + monkeypatch.setattr(sys, "platform", "linux") + # Mask interpretation and agreement with the catalog belong to the operator. + mask = "GPU-opaque,3" + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) + if order is None: + monkeypatch.delenv("CUDA_DEVICE_ORDER", raising=False) + else: + monkeypatch.setenv("CUDA_DEVICE_ORDER", order) + probe = MagicMock(side_effect=AssertionError("local GPU planning must not probe hardware")) + monkeypatch.setattr(local.subprocess, "run", probe) offer = Offer( name="gpu", connection="workstation", - resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=gpus), + resources=Resources.from_bytes( + cpus=1, memory_bytes=512 * 1024**2, gpus=gpus, accelerator_name="A100", + ), max_nodes=1, time=TimeLimits(default="1m", max="1m"), ) plan = provider.plan(offer, Request.parse("1", "0.5", gpus=f"GPU:{gpus}" if gpus else "0")) - assert plan.details["gpu_devices"] == devices[:gpus] + assert plan.details["cuda_visible_devices"] == (mask if gpus else "") + assert plan.details["cuda_device_order"] == order assert not provider.root.exists() monkeypatch.setattr(local, "_boot_identity", lambda: str(uuid4())) popen = MagicMock(return_value=SimpleNamespace(pid=12345)) monkeypatch.setattr(local.subprocess, "Popen", popen) monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "an-ambient-mask") + monkeypatch.setenv("CUDA_DEVICE_ORDER", "changed-after-planning") identity = provider.launch(plan) - assert popen.call_args.kwargs["env"]["CUDA_VISIBLE_DEVICES"] == ",".join(devices[:gpus]) + assert popen.call_args.kwargs["env"]["CUDA_VISIBLE_DEVICES"] == (mask if gpus else "") + assert popen.call_args.kwargs["env"].get("CUDA_DEVICE_ORDER") == order assert os.environ["CUDA_VISIBLE_DEVICES"] == "an-ambient-mask" + assert os.environ["CUDA_DEVICE_ORDER"] == "changed-after-planning" record = read_private_json(provider.root / identity.token / "identity.json") assert record["gpus"] == gpus monkeypatch.setattr(provider, "_process", lambda *_: None) snapshot = provider.inspect(identity) assert snapshot.resources is not None and snapshot.resources.gpus == gpus - assert snapshot.resources.accelerator_name == ("NVIDIA-A100" if gpus else None) - assert visible.call_count == (2 if gpus else 0) + assert snapshot.resources.accelerator_name == ("A100" if gpus else None) + probe.assert_not_called() -def test_local_gpu_capacity_is_checked_at_plan_and_again_before_spawn( +@pytest.mark.parametrize("mask", [None, ""]) +def test_local_gpu_offers_require_an_explicit_nonempty_native_mask( provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, + mask: str | None, ) -> None: - device = f"GPU-{uuid4()}" + monkeypatch.setattr(sys, "platform", "linux") + if mask is None: + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + else: + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) offer = Offer( name="gpu", connection="workstation", resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=1), max_nodes=1, time=TimeLimits(default="1m", max="1m"), ) request = Request.parse("1", "0.5", gpus="GPU:1") - monkeypatch.setattr(gpu, "inventory", lambda: ()) - with pytest.raises(ComputeError, match="visible CUDA capacity for GPU"): + with pytest.raises(ComputeError, match="nonempty CUDA_VISIBLE_DEVICES"): provider.plan(offer, request) - monkeypatch.setattr(gpu, "inventory", lambda: (gpu.Device(device, "NVIDIA-A100"),)) - plan = provider.plan(offer, request) - monkeypatch.setattr(gpu, "inventory", lambda: (gpu.Device(f"GPU-{uuid4()}", "NVIDIA-A100"),)) - with pytest.raises(ComputeError, match="no longer visible"): - provider.launch(plan) assert not provider.root.exists() -def test_unavailable_local_gpu_driver_does_not_hide_a_later_slurm_offer( +def test_unconfigured_local_gpu_mask_does_not_hide_a_later_slurm_offer( provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, ) -> None: - monkeypatch.setattr(gpu, "inventory", MagicMock(side_effect=ProjectError("CUDA driver failed"))) + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) local_offer = Offer( name="local-gpu", connection="workstation", resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=1), @@ -891,39 +904,25 @@ def test_unavailable_local_gpu_driver_does_not_hide_a_later_slurm_offer( assert not provider.root.exists() -@pytest.mark.parametrize("requested,count,expected", [ - ("nvidia-a100", 1, (1,)), ("GPU", 2, (0, 1)), ("A100", 1, ()), -]) -def test_local_named_accelerators_match_native_models_without_guessing_aliases( +def test_local_gpu_offers_are_unavailable_outside_linux( provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, - requested: str, count: int, expected: tuple[int, ...], ) -> None: - devices = ( - gpu.Device(f"GPU-{uuid4()}", "NVIDIA-H100"), - gpu.Device(f"GPU-{uuid4()}", "NVIDIA-A100"), - gpu.Device(f"GPU-{uuid4()}", "NVIDIA-A100"), - ) - monkeypatch.setattr(gpu, "inventory", lambda: devices) + monkeypatch.setattr(sys, "platform", "darwin") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") offer = Offer( name="gpu", connection="workstation", resources=Resources.from_bytes( - cpus=1, memory_bytes=512 * 1024**2, gpus=count, accelerator_name=requested, + cpus=1, memory_bytes=512 * 1024**2, gpus=1, ), max_nodes=1, time=TimeLimits(default="1m", max="1m"), ) - request = Request.parse("1", "0.5", gpus=f"{requested}:{count}") - if not expected: - with pytest.raises(ComputeError, match="visible CUDA capacity for A100"): - provider.plan(offer, request) - return - plan = provider.plan(offer, request) - assert plan.details["gpu_devices"] == tuple(devices[index].uuid for index in expected) - assert plan.details["accelerator_name"] == ("NVIDIA-A100" if count == 1 else "GPU") - - -@pytest.mark.parametrize("visible_count", [0, 1, 2]) -def test_local_runtime_verifies_gpu_visibility_before_advertising_resources( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, visible_count: int, + with pytest.raises(ComputeError, match="GPU allocations require Linux"): + provider.plan(offer, Request.parse("1", "0.5", gpus="GPU:1")) + + +@pytest.mark.parametrize("gpus", [0, 2]) +def test_local_runtime_advertises_configured_resources_and_preserves_native_gpu_mask( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, gpus: int, ) -> None: import distributed @@ -933,7 +932,8 @@ def test_local_runtime_verifies_gpu_visibility_before_advertising_resources( "identity": "allocation", "deadline": time.monotonic() + 60, "task_slots": 1, "scratch": str(scratch), }) - write_private_json(directory / "identity.json", {"cpus": 1, "memory": 1024**3, "gpus": 1}) + write_private_json(directory / "identity.json", {"cpus": 1, "memory": 1024**3, "gpus": gpus}) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "3,1") monkeypatch.setattr(sys, "argv", ["local_runtime", str(directory)]) monkeypatch.setattr(os, "getsid", lambda _: os.getpid()) monkeypatch.setattr(os, "getpgrp", os.getpid) @@ -945,18 +945,10 @@ def test_local_runtime_verifies_gpu_visibility_before_advertising_resources( if signum == signal.SIGTERM else None, ) monkeypatch.setattr(local_runtime, "create_security", lambda _: None) - monkeypatch.setattr( - gpu, "visible_devices", lambda: tuple(f"GPU-{i}" for i in range(visible_count)), - ) cluster = MagicMock() cluster.return_value.__enter__.return_value.scheduler.id = "Scheduler-gpu" monkeypatch.setattr(distributed, "LocalCluster", cluster) - if visible_count != 1: - with pytest.raises(ComputeError, match="CUDA GPUs do not match"): - local_runtime.main() - cluster.assert_not_called() - assert not (directory / "connection.json").exists() - else: - local_runtime.main() - assert cluster.call_args.kwargs["resources"] == {"CPU": 1, "MEMORY": 1024**3, "GPU": 1} + local_runtime.main() + assert cluster.call_args.kwargs["resources"] == {"CPU": 1, "MEMORY": 1024**3, "GPU": gpus} + assert os.environ["CUDA_VISIBLE_DEVICES"] == ("3,1" if gpus else "") diff --git a/tests/test_compute_slurm.py b/tests/test_compute_slurm.py index 41402360..4d2ed937 100644 --- a/tests/test_compute_slurm.py +++ b/tests/test_compute_slurm.py @@ -18,7 +18,6 @@ import psutil import pytest -from lightcone.engine import gpu from lightcone.engine.compute import Compute, slurm, slurm_bootstrap from lightcone.engine.compute.catalog import Catalog from lightcone.engine.compute.model import ( @@ -1024,13 +1023,12 @@ def test_bootstrap_refuses_mismatched_native_envelope( slurm_bootstrap._allocation(_bootstrap_args(tmp_path)) -@pytest.mark.parametrize("native,visible_count,valid", [ - ("", 2, False), ("unknown", 2, False), ("1", 2, False), - ("2", 0, False), ("2", 1, False), ("2", 2, True), ("4", 4, True), +@pytest.mark.parametrize("native,valid", [ + ("", False), ("unknown", False), ("1", False), ("2", True), ("4", True), ]) -def test_gpu_bootstrap_requires_native_and_cuda_capacity_without_rewriting_the_mask( +def test_gpu_bootstrap_requires_native_capacity_without_probing_or_rewriting_the_mask( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, - native: str, visible_count: int, valid: bool, + native: str, valid: bool, ) -> None: for key, value in _bootstrap_env(0).items(): monkeypatch.setenv(key, value) @@ -1038,24 +1036,39 @@ def test_gpu_bootstrap_requires_native_and_cuda_capacity_without_rewriting_the_m monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "1,3") monkeypatch.setenv("CUDA_DEVICE_ORDER", "FASTEST_FIRST") - def devices() -> tuple[str, ...]: - assert os.environ["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" - assert os.environ["CUDA_VISIBLE_DEVICES"] == "1,3" - return tuple(f"GPU-{i}" for i in range(visible_count)) - - monkeypatch.setattr(gpu, "visible_devices", devices) + probe = MagicMock(side_effect=AssertionError("Slurm bootstrap must not probe CUDA")) + monkeypatch.setattr(subprocess, "run", probe) args = _bootstrap_args(tmp_path) args.gpus = 2 if valid: _, _, rank = slurm_bootstrap._allocation(args) assert rank == 0 + assert os.environ["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" else: with pytest.raises(ComputeError, match="GPUs"): slurm_bootstrap._allocation(args) assert os.environ["CUDA_VISIBLE_DEVICES"] == "1,3" + probe.assert_not_called() + + +@pytest.mark.parametrize("mask", [None, ""]) +def test_gpu_bootstrap_requires_a_nonempty_native_mask( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, mask: str | None, +) -> None: + for key, value in _bootstrap_env(0).items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("SLURM_GPUS_ON_NODE", "2") + if mask is None: + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + else: + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) + args = _bootstrap_args(tmp_path) + args.gpus = 2 + with pytest.raises(ComputeError, match="native CUDA_VISIBLE_DEVICES mask"): + slurm_bootstrap._allocation(args) -def test_gpu_worker_advertises_verified_capacity_with_the_native_mask( +def test_gpu_worker_advertises_native_capacity_with_the_native_mask( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: import distributed @@ -1066,7 +1079,6 @@ def test_gpu_worker_advertises_verified_capacity_with_the_native_mask( monkeypatch.setenv(key, value) monkeypatch.setenv("SLURM_GPUS_ON_NODE", "2") monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "1,3") - monkeypatch.setattr(gpu, "visible_devices", lambda: ("GPU-first", "GPU-second")) connection = Connection( namespace=NAMESPACE, provider="slurm", launch={"connection_root": args.connection_root}, ) diff --git a/tests/test_gpu.py b/tests/test_gpu.py deleted file mode 100644 index b5568397..00000000 --- a/tests/test_gpu.py +++ /dev/null @@ -1,144 +0,0 @@ -"""CUDA owns native device enumeration; only verified identities become capacity.""" - -from __future__ import annotations - -import ctypes -import json -import subprocess -from pathlib import Path -from types import SimpleNamespace -from unittest.mock import Mock -from uuid import UUID - -import pytest - -from lightcone.engine import gpu -from lightcone.engine.project import ProjectError - -GPU = "GPU-01234567-89ab-cdef-0123-456789abcdef" -DEVICE = {"uuid": GPU, "name": "NVIDIA-A100-SXM4-80GB"} - - -def test_empty_mask_never_loads_or_queries_cuda(monkeypatch: pytest.MonkeyPatch) -> None: - monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "") - run = Mock(side_effect=AssertionError("must not query hidden devices")) - monkeypatch.setattr(gpu.subprocess, "run", run) - assert gpu.visible_devices() == () - - -def test_uuid_probe_preserves_the_native_mask_and_uses_isolated_python( - monkeypatch: pytest.MonkeyPatch, -) -> None: - monkeypatch.setattr(gpu.sys, "platform", "linux") - monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") - run = Mock(return_value=subprocess.CompletedProcess([], 0, json.dumps([DEVICE]), "")) - monkeypatch.setattr(gpu.subprocess, "run", run) - assert gpu.visible_devices() == (GPU,) - argv = run.call_args.args[0] - assert argv[1] == "-I" - assert Path(argv[2]) == Path(gpu.__file__) - assert "env" not in run.call_args.kwargs # Native visibility is inherited unchanged. - assert run.call_args.kwargs["timeout"] > 0 - - -@pytest.mark.parametrize("payload", [ - {}, [42], [GPU], [{"uuid": "GPU-bad", "name": "A100"}], - [{"uuid": GPU, "name": "A100:4"}], [{"uuid": GPU, "name": ""}], -]) -def test_invalid_inventory_is_not_capacity( - monkeypatch: pytest.MonkeyPatch, payload: object, -) -> None: - monkeypatch.setattr(gpu.sys, "platform", "linux") - monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) - monkeypatch.setattr( - gpu.subprocess, "run", - Mock(return_value=subprocess.CompletedProcess([], 0, json.dumps(payload), "")), - ) - with pytest.raises(ProjectError, match="invalid GPU identities"): - gpu.visible_devices() - - -def test_inventory_preserves_models_and_rejects_duplicate_identities( - monkeypatch: pytest.MonkeyPatch, -) -> None: - monkeypatch.setattr(gpu.sys, "platform", "linux") - monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) - run = Mock(return_value=subprocess.CompletedProcess([], 0, json.dumps([DEVICE]), "")) - monkeypatch.setattr(gpu.subprocess, "run", run) - assert gpu.inventory() == (gpu.Device(GPU, DEVICE["name"]),) - run.return_value.stdout = json.dumps([DEVICE, DEVICE]) - with pytest.raises(ProjectError, match="duplicate GPU identities"): - gpu.inventory() - - -def test_probe_failure_is_not_reported_as_an_empty_machine(monkeypatch: pytest.MonkeyPatch) -> None: - monkeypatch.setattr(gpu.sys, "platform", "linux") - monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) - monkeypatch.setattr( - gpu.subprocess, "run", - Mock(return_value=subprocess.CompletedProcess([], 1, "", "CUDA driver returned error 999")), - ) - with pytest.raises(ProjectError, match="error 999"): - gpu.visible_devices() - - -def test_missing_cuda_is_a_cpu_only_machine(monkeypatch: pytest.MonkeyPatch) -> None: - monkeypatch.setattr(ctypes, "CDLL", Mock(side_effect=OSError("no CUDA driver"))) - assert gpu._probe() == [] - - -@pytest.mark.parametrize("partitioned", [False, True]) -@pytest.mark.parametrize("model_name", ["NVIDIA A100-SXM4-80GB", "TITAN X (Pascal)"]) -def test_probe_uses_cuda_device_handles_and_rejects_mig( - monkeypatch: pytest.MonkeyPatch, partitioned: bool, model_name: str, -) -> None: - # The visible ordinal maps to a driver handle, not a host GPU index. The - # real ctypes buffers and signatures exercise the probe's FFI boundary. - raw = UUID(GPU.removeprefix("GPU-")).bytes - - def count(pointer: object) -> int: - ctypes.cast(pointer, ctypes.POINTER(ctypes.c_int))[0] = 1 - return 0 - - def device(pointer: object, ordinal: int) -> int: - assert ordinal == 0 - ctypes.cast(pointer, ctypes.POINTER(ctypes.c_int))[0] = 7 - return 0 - - def identity(buffer: object, handle: ctypes.c_int, *, instance: bool = False) -> int: - assert handle.value == 7 - ctypes.memmove(buffer, bytes(16) if partitioned and instance else raw, 16) - return 0 - - def name(buffer: object, size: int, handle: ctypes.c_int) -> int: - assert handle.value == 7 - value = model_name.encode("ascii") + b"\x00" - assert size >= len(value) - ctypes.memmove(buffer, value, len(value)) - return 0 - - driver = SimpleNamespace( - cuInit=Mock(return_value=0), - cuDeviceGetCount=Mock(side_effect=count), - cuDeviceGet=Mock(side_effect=device), - cuDeviceGetUuid=Mock(side_effect=identity), - cuDeviceGetUuid_v2=Mock(side_effect=lambda b, d: identity(b, d, instance=True)), - cuDeviceGetName=Mock(side_effect=name), - ) - monkeypatch.setattr(ctypes, "CDLL", Mock(return_value=driver)) - if partitioned: - with pytest.raises(RuntimeError, match="MIG"): - gpu._probe() - else: - expected = "NVIDIA-A100-SXM4-80GB" if model_name.startswith("NVIDIA") else "TITAN-X-Pascal" - assert gpu._probe() == [{"uuid": GPU, "name": expected}] - - -def test_only_nvidia_character_nodes_are_granted( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, -) -> None: - for name in ("nvidia0", "nvidiactl", "nvidia-user-file", "unrelated"): - (tmp_path / name).touch() - monkeypatch.setattr(gpu, "_DEVICE_ROOT", tmp_path) - monkeypatch.setattr(Path, "is_char_device", lambda p: p.name in {"nvidia0", "nvidiactl"}) - assert gpu.device_paths() == (tmp_path / "nvidia0", tmp_path / "nvidiactl") diff --git a/tests/test_gpu_execution.py b/tests/test_gpu_execution.py index 06e45b83..49dfc2ac 100644 --- a/tests/test_gpu_execution.py +++ b/tests/test_gpu_execution.py @@ -1,4 +1,4 @@ -"""GPU requests reach each command without changing the shared worker environment.""" +"""GPU commands inherit the allocation mask without probing or assigning devices.""" from __future__ import annotations @@ -10,9 +10,10 @@ import pytest -from lightcone.engine import assets, container, gpu, identity, plan, worker +from lightcone.engine import assets, container, identity, plan, worker from lightcone.engine import run as engine_run from lightcone.engine.project import ProjectError +from lightcone.engine.sandbox import policy as policy_module _SPEC = """ version: "0.0.13" @@ -42,65 +43,82 @@ def project(analysis: Callable[..., Path]) -> tuple[Path, plan.Task, worker.RunC @pytest.mark.parametrize("requested", [0, 1, 2]) -def test_recipe_gets_requested_uuid_mask_in_its_actual_subprocess( +def test_recipe_inherits_the_whole_mask_without_changing_the_worker_environment( project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, requested: int, ) -> None: root, task, context = project - visible = ("GPU-second", "GPU-first") - monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "native-allocation") - monkeypatch.setattr(gpu, "visible_devices", lambda: visible) - monkeypatch.setattr(gpu, "device_paths", lambda: ()) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + monkeypatch.setattr(policy_module, "_gpu_device_paths", lambda: ()) result = worker.execute(root, replace(task, resources={"gpus": requested}), {}, context) assert result.status == "ok", result.reason - assert task.output_path.read_text() == ",".join(visible[:requested]) - assert os.environ["CUDA_VISIBLE_DEVICES"] == "native-allocation" + assert task.output_path.read_text() == ("2,0" if requested else "") + assert os.environ["CUDA_VISIBLE_DEVICES"] == "2,0" -def test_worker_gpu_mismatch_preserves_existing_output_and_manifest( +@pytest.mark.parametrize("mask", [None, ""]) +def test_missing_worker_gpu_mask_preserves_existing_output_and_manifest( project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + mask: str | None, ) -> None: root, task, context = project task.output_path.parent.mkdir(parents=True, exist_ok=True) task.output_path.write_text("previous output") task.manifest_path.write_text("previous manifest") - monkeypatch.setattr(gpu, "visible_devices", lambda: ("GPU-only",)) - with pytest.raises(ProjectError, match="requests 2 GPUs.*only 1 CUDA devices"): - worker.execute(root, replace(task, resources={"gpus": 2}), {}, context) + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + if mask is not None: + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) + with pytest.raises(ProjectError, match="nonempty allocation CUDA_VISIBLE_DEVICES"): + worker.execute(root, replace(task, resources={"gpus": 1}), {}, context) assert task.output_path.read_text() == "previous output" assert task.manifest_path.read_text() == "previous manifest" -@pytest.mark.parametrize("reserved", [0, 1, 2]) -def test_probe_exposes_only_reserved_devices_from_the_native_allocation( +@pytest.mark.parametrize("runtime", ["docker", "podman"]) +def test_unsupported_gpu_container_preserves_existing_output_and_manifest( project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, - reserved: int, + runtime: str, +) -> None: + root, task, context = project + task.output_path.parent.mkdir(parents=True, exist_ok=True) + task.output_path.write_text("previous output") + task.manifest_path.write_text("previous manifest") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + context = replace(context, runtime=replace( + context.runtime, mode="containerized", runtime=runtime, + )) + with pytest.raises(ProjectError, match="GPU containers require podman-hpc"): + worker.execute(root, replace(task, resources={"gpus": 1}), {}, context) + assert task.output_path.read_text() == "previous output" + assert task.manifest_path.read_text() == "previous manifest" + + +@pytest.mark.parametrize("use_gpus", [False, True]) +def test_probe_inherits_the_native_allocation_mask_or_hides_gpus( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + use_gpus: bool, ) -> None: _, _, context = project - visible = ("GPU-first", "GPU-second", "GPU-unreserved") - discover = Mock(return_value=visible) - monkeypatch.setattr(gpu, "visible_devices", discover) - monkeypatch.setattr(gpu, "device_paths", lambda: ()) - monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "native-allocation") + monkeypatch.setattr(policy_module, "_gpu_device_paths", lambda: ()) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") received: list[bytes] = [] outcome = engine_run._probe( context.runtime, [], ("sh", "-c", "printf '%s' \"$CUDA_VISIBLE_DEVICES\""), - reserved, + use_gpus, output=lambda stream, data: received.append(data) if stream == "stdout" else None, ) assert outcome.returncode == 0 - assert b"".join(received).decode() == ",".join(visible[:reserved]) - assert discover.call_count == bool(reserved) - assert os.environ["CUDA_VISIBLE_DEVICES"] == "native-allocation" + assert b"".join(received).decode() == ("2,0" if use_gpus else "") + assert os.environ["CUDA_VISIBLE_DEVICES"] == "2,0" -def test_probe_gpu_shortage_refuses_before_running_the_command( +def test_probe_missing_gpu_mask_refuses_before_running_the_command( project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, ) -> None: _, _, context = project - monkeypatch.setattr(gpu, "visible_devices", lambda: ("GPU-only",)) + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) execute = Mock() monkeypatch.setattr(engine_run.sandbox, "run", execute) - with pytest.raises(ProjectError, match="reserved 2 GPUs.*only 1 CUDA devices"): - engine_run._probe(context.runtime, [], ("true",), 2, output=lambda *_: None) + with pytest.raises(ProjectError, match="nonempty allocation CUDA_VISIBLE_DEVICES"): + engine_run._probe(context.runtime, [], ("true",), True, output=lambda *_: None) execute.assert_not_called() diff --git a/tests/test_materialize.py b/tests/test_materialize.py index ddaf7c86..ef98b42f 100644 --- a/tests/test_materialize.py +++ b/tests/test_materialize.py @@ -1176,10 +1176,7 @@ def test_real_dask_respects_recipe_resource_reservations( start, time.monotonic(), os.environ.get("CUDA_VISIBLE_DEVICES"), ])) """}) - from lightcone.engine import gpu - - monkeypatch.setattr(gpu, "visible_devices", lambda: ("GPU-first", "GPU-second")) - monkeypatch.setattr(gpu, "device_paths", lambda: ()) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") _resource_cluster(monkeypatch, gpus=gpus) report = engine.materialize(root, [], cluster_id=CLUSTER_ID) @@ -1188,7 +1185,7 @@ def test_real_dask_respects_recipe_resource_reservations( events = [] for path in (root / "results/baseline").glob("task*.json"): start, finish, visible = json.loads(path.read_text()) - assert visible == ("GPU-first" if gpus else "") + assert visible == ("2,0" if gpus else "") events.extend([(start, 1), (finish, -1)]) live = peak = 0 for _, change in sorted(events): @@ -1459,4 +1456,3 @@ def test_an_output_the_spec_dropped_is_excluded_and_named(root: Path, inline: No assert any(".second.manifest.json" in w for w in report.warnings) document = (root / "ro-crate-metadata.json").read_text() assert "results/baseline/second.txt" not in document - diff --git a/tests/test_sandbox_oci.py b/tests/test_sandbox_oci.py index d346fd5b..de6efac3 100644 --- a/tests/test_sandbox_oci.py +++ b/tests/test_sandbox_oci.py @@ -7,7 +7,6 @@ from __future__ import annotations -import csv import subprocess from dataclasses import replace from pathlib import Path @@ -15,6 +14,7 @@ import pytest +from lightcone.engine.project import ProjectError from lightcone.engine.sandbox import boundary, exec_policy from lightcone.engine.sandbox.boundary import Unavailable from lightcone.engine.sandbox.model import Policy @@ -187,26 +187,29 @@ def test_runtimes_differ_only_in_their_spellings(root: Path, policy: Policy) -> assert p == d == h -@pytest.mark.parametrize("runtime", ["podman", "docker", "podman-hpc"]) -def test_gpu_flags_name_only_the_requested_devices( - root: Path, policy: Policy, runtime: str, +def test_podman_hpc_preserves_the_allocation_mask_and_enables_native_gpu_support( + root: Path, policy: Policy, ) -> None: - devices = "GPU-first,GPU-second" - selected = replace(policy, env={**policy.env, "CUDA_VISIBLE_DEVICES": devices}) - backend = _backend(root, runtime) + devices = "2,0" + selected = replace(policy, env={ + **policy.env, "CUDA_VISIBLE_DEVICES": devices, "CUDA_DEVICE_ORDER": "PCI_BUS_ID", + }) + backend = _backend(root, "podman-hpc") argv = backend.wrap(selected, ["true"]) assert argv == backend.wrap(selected, ["true"]) assert f"--env=CUDA_VISIBLE_DEVICES={devices}" in argv - if runtime == "podman-hpc": - assert "--gpu" in argv - elif runtime == "docker": - assert next(csv.reader([argv[argv.index("--gpus") + 1]])) == [f"device={devices}"] - else: - assert [arg for arg in argv if arg.startswith("--device=")] == [ - "--device=nvidia.com/gpu=GPU-first", "--device=nvidia.com/gpu=GPU-second", - ] - assert "all" not in argv - assert "--device=nvidia.com/gpu=all" not in argv + assert "--env=CUDA_DEVICE_ORDER=PCI_BUS_ID" in argv + assert "--gpu" in argv + assert not any(arg.startswith("--device=") for arg in argv) + + +@pytest.mark.parametrize("runtime", ["podman", "docker"]) +def test_generic_gpu_containers_are_explicitly_refused( + root: Path, policy: Policy, runtime: str, +) -> None: + selected = replace(policy, env={**policy.env, "CUDA_VISIBLE_DEVICES": "2,0"}) + with pytest.raises(ProjectError, match="GPU containers require podman-hpc"): + _backend(root, runtime).wrap(selected, ["true"]) @pytest.mark.parametrize("runtime", ["podman", "docker", "podman-hpc"]) diff --git a/tests/test_sandbox_policy.py b/tests/test_sandbox_policy.py index dbaedcb2..3d926077 100644 --- a/tests/test_sandbox_policy.py +++ b/tests/test_sandbox_policy.py @@ -14,6 +14,7 @@ import pytest +from lightcone.engine.project import ProjectError from lightcone.engine.sandbox import policy as policy_module from lightcone.engine.sandbox.boundary import scope from lightcone.engine.sandbox.model import EXEC_ALLOWLIST_VERSION @@ -239,26 +240,47 @@ def test_the_entropy_sources_stay_read_only(built: policy_module.Policy) -> None @pytest.mark.parametrize("containerized", [False, True]) -@pytest.mark.parametrize("devices", [(), ("GPU-first", "GPU-second")]) +@pytest.mark.parametrize("use_gpus", [False, True]) def test_gpu_policy_sets_command_visibility_without_changing_the_host( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, - containerized: bool, devices: tuple[str, ...], + containerized: bool, use_gpus: bool, ) -> None: - from lightcone.engine import gpu - project = tmp_path / "project" project.mkdir() device = tmp_path / "nvidia0" device.touch() - monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "native-allocation") - monkeypatch.setattr(gpu, "device_paths", lambda: (device,)) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + monkeypatch.setenv("CUDA_DEVICE_ORDER", "PCI_BUS_ID") + monkeypatch.setattr(policy_module, "_gpu_device_paths", lambda: (device,)) with scope(policy_module.exec_policy( - project, containerized=containerized, gpu_devices=devices, + project, containerized=containerized, use_gpus=use_gpus, )) as built: - assert built.env["CUDA_VISIBLE_DEVICES"] == ",".join(devices) - assert (device in built.write) == (bool(devices) and not containerized) + assert built.env["CUDA_VISIBLE_DEVICES"] == ("2,0" if use_gpus else "") + if use_gpus: + assert built.env["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" + assert (device in built.write) == (use_gpus and not containerized) assert Path("/dev") not in built.write - assert os.environ["CUDA_VISIBLE_DEVICES"] == "native-allocation" + assert os.environ["CUDA_VISIBLE_DEVICES"] == "2,0" + assert os.environ["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" + + +def test_gpu_policy_refuses_a_missing_mask_before_creating_private_state( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + with pytest.raises(ProjectError, match="nonempty allocation CUDA_VISIBLE_DEVICES"): + policy_module.exec_policy(tmp_path, containerized=True, use_gpus=True) + assert not (tmp_path / ".lightcone").exists() + + +def test_only_nvidia_character_nodes_are_granted( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + for name in ("nvidia0", "nvidiactl", "nvidia-user-file", "unrelated"): + (tmp_path / name).touch() + monkeypatch.setattr(policy_module, "_DEVICE_ROOT", tmp_path) + monkeypatch.setattr(Path, "is_char_device", lambda p: p.name in {"nvidia0", "nvidiactl"}) + assert policy_module._gpu_device_paths() == (tmp_path / "nvidia0", tmp_path / "nvidiactl") def test_proc_and_sys_are_not_restricted(built: policy_module.Policy) -> None: From 1b614ca0e7e01005078a1584829cab091dace115 Mon Sep 17 00:00:00 2001 From: Francois Lanusse Date: Tue, 29 Sep 2026 05:10:29 -0700 Subject: [PATCH 3/3] Fix resource admission and address compute review findings Signed-off-by: Francois Lanusse --- CLAUDE.md | 47 +++++---- docs/api/compute.md | 35 ++++--- docs/api/container.md | 4 +- docs/api/materialize.md | 9 +- docs/api/plan.md | 4 +- docs/api/sandbox.md | 7 +- docs/api/worker.md | 4 +- docs/architecture.md | 19 ++-- docs/cli/materialize.md | 13 +-- docs/cli/run.md | 6 +- docs/user/cluster.md | 41 +++++--- src/lightcone/engine/compute/local.py | 2 + src/lightcone/engine/compute/local_runtime.py | 2 +- src/lightcone/engine/compute/model.py | 41 ++++---- src/lightcone/engine/compute/slurm.py | 10 +- .../engine/compute/slurm_bootstrap.py | 10 +- src/lightcone/engine/container.py | 15 ++- src/lightcone/engine/execution_resources.py | 88 +++++++++-------- src/lightcone/engine/materialize.py | 95 +++++++++++-------- src/lightcone/engine/run.py | 11 ++- src/lightcone/engine/sandbox/oci.py | 7 +- src/lightcone/engine/sandbox/policy.py | 6 +- src/lightcone/engine/units.py | 22 +---- src/lightcone/engine/worker.py | 4 +- tests/test_compute.py | 18 +++- tests/test_compute_local.py | 56 ++++++++++- tests/test_compute_slurm.py | 13 ++- tests/test_container.py | 22 +++++ tests/test_execution_resources.py | 54 ++++++----- tests/test_gpu_execution.py | 77 ++++++++++++++- tests/test_materialize.py | 61 ++++++++++++ tests/test_sandbox_policy.py | 2 +- 32 files changed, 564 insertions(+), 241 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 6263249c..ee4aaced 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -952,10 +952,10 @@ none; formats may, `tar.gz`), never `Path.stem`. The old `_HASH_EXCLUDE` is gone with the directory that made it necessary: the manifest cannot be inside the thing it describes any more. -**Dask owns the ordering.** Every task is submitted with its upstream -futures as arguments, so the dependency order, the parallelism, and the -scheduling all fall out of the argument graph. There is no ready-set loop -and no hand-rolled topological sort in the execution path. +**Dask owns the ordering.** Tasks that may execute receive upstream futures +or already-current `TaskResult` values as arguments, so dependency order, +parallelism, and scheduling fall out of the argument graph. There is no ready-set +loop and no hand-rolled topological sort in the execution path. `Graph.order()` exists for the read-only walk, which has to classify a task after everything upstream of it — and for submitting in an order where a task's upstream handles already exist. @@ -1165,7 +1165,7 @@ its sync touches only ignored paths — so both modes run one order.) **A run fetches its declared inputs; the read-only verbs never do.** `materialize` batch-runs `git annex get` over the graph's in-tree -declared inputs before anything hashes (driver-side — the storage +declared inputs before workers hash or execute (driver-side — the storage invariant that nobody is ever asked to run an annex command by hand), so a bytes-free clone materializes straight to up-to-date. A failed fetch is a *warning*, never a refusal: independent tasks still run and @@ -1797,18 +1797,27 @@ allocation was not stopped and unreported tasks may still be running". **Recipe resources use standard Dask admission.** Preserve ASTRA `recipe.resources` in `plan.Task` as raw mappings so `status` and `--check` remain independent of executor support. Parse `TaskResources` at execution admission: whole CPUs, memory -bytes, and whole GPU counts. Validate the whole selected graph before preparation -or submission, then pass reservations explicitly to submission. Recipe `time_limit` -is unsupported and must fail explicitly; allocation walltime remains supported. -Workers advertise CPU/MEMORY/GPU; tasks reserve their declarations, with omitted RAM -reserving a whole worker's memory and probes reserving all whole-worker budgets. +bytes, and whole GPU counts. Reuse the read-only classification walk before +admission: known current/behind outputs become values without Dask submission. +Validate tasks that may execute, including dependents of potentially rebuilt +outputs; workers recheck actual upstream digests. Normalize worker budgets once +with `worker_capacities`, then pass reservations explicitly to submission. Recipe +`time_limit` is unsupported and must fail explicitly; allocation walltime remains supported. +Workers advertise CPU/MEMORY/GPU; tasks reserve their declarations. Omitted RAM +adds no memory reservation; CPU requests and task slots govern concurrency. +Probes reserve all whole-worker budgets. GPU recipes reserve the worker's full GPU budget, one GPU recipe at a time, and inherit its whole allocation mask. Recipe `gpus` is a minimum capacity requirement, -not a visibility limit; it defaults to zero and does not select a model. Recipe memory retains ASTRA units (`8Gi` binary, -`8GB` decimal, no bare quantities), independently of compute's SkyPilot units. +not a visibility limit; it defaults to zero and does not select a model. Recipe +memory retains ASTRA units (`8Gi` binary, `8GB` decimal, no bare quantities), +independently of compute's SkyPilot units. Thread slots remain a separate concurrency cap. Reservations are cooperative, not -per-command OS CPU/RAM limits; unsupported disk/model requests and fractional -CPU/GPU counts fail explicitly. +per-command OS CPU/RAM limits or BLAS thread counts. Local Nanny defaults keep +OMP/MKL/OPENBLAS threads at one unless the launch environment overrides them; Slurm +uses direct workers and the job environment. Unsupported disk/model requests and +fractional CPU/GPU counts fail explicitly. Exact bytes are shared in `units.py`; +allocation durations are parsed in `compute.model`, with error conversion only +at the CLI request boundary. **GPU visibility comes from the allocation, not device discovery.** No CUDA probe, UUID inventory, MIG detection, model verification, or custom Dask worker. Local @@ -1817,9 +1826,13 @@ order. Slurm validates native GPU counts, preserves its mask, and sets `CUDA_DEVICE_ORDER=PCI_BUS_ID`. `exec_policy(use_gpus=True)` inherits that whole mask; CPU commands get an empty one. Never mutate the reusable worker's environment. Direct GPU policies grant native NVIDIA character nodes; OS permissions and cgroups -remain authoritative. Container GPU execution supports podman-hpc `--gpu` only; -ordinary Docker/Podman GPU requests fail explicitly. CPU containers remain supported -on all runtimes and set `NVIDIA_VISIBLE_DEVICES=void`. Physical GPU execution remains +remain authoritative. The host must initialize NVIDIA character devices including +UVM before launch; lc neither loads drivers nor creates nodes. Standalone GPU +reruns need an explicit CUDA mask in their own environment. Container GPU execution +supports podman-hpc `--gpu` only. Explicit GPU recipes on ordinary Docker/Podman +fail before image preparation; probes use CPU policy with a diagnostic note while +retaining their whole-worker reservation. CPU containers remain supported on all +runtimes and set `NVIDIA_VISIBLE_DEVICES=void`. Physical GPU execution remains unvalidated; tests check real subprocess masks and native argv without GPU hardware. **One catalog selector, `LC_COMPUTE_CONFIG` (2026-09).** `lc compute --config` diff --git a/docs/api/compute.md b/docs/api/compute.md index 3758504c..b780bd2c 100644 --- a/docs/api/compute.md +++ b/docs/api/compute.md @@ -127,21 +127,23 @@ Workers advertise standard Dask `CPU`, `MEMORY`, and `GPU` resources; memory is in bytes. `engine.execution_resources.TaskResources` validates ASTRA's `recipe.resources` into whole CPUs, bytes, and a whole GPU count at execution admission. `plan.Task` preserves the ASTRA mapping so read-only -classification does not impose executor restrictions. `requirements(workers)` -checks that one worker can satisfy it and returns the resource dictionary used -by `Client.submit`. -An omitted memory request reserves the full homogeneous worker budget; -`whole_worker=True` reserves CPU, memory, and GPUs for a probe. Recipe GPU counts -default to zero; a GPU recipe reserves the full GPU budget of a fitting worker +classification does not impose executor restrictions. `worker_capacities(workers)` +normalizes advertised budgets once; `requirements(capacities)` checks that one +worker can satisfy a task and returns its `Client.submit` resource dictionary. +Omitted memory adds no `MEMORY` reservation. `whole_worker=True` reserves CPU, +memory, and GPUs for a probe. Recipe GPU counts default to zero; a GPU recipe +reserves the full GPU budget of a fitting worker and inherits its whole allocation mask. The requested count is a minimum, not a visibility limit. This serializes GPU recipes per worker without device assignment. Unsupported disk/type requests and fractional CPU/GPU counts fail before execution. -The materialize scheduler validates every selected task before preparation or -submission, preventing earlier tasks from starting before a later impossible -request is discovered, then passes each task's reservation explicitly to -submission. Allocation and task requests share byte conversion utilities; their -models remain distinct because allocation selection supports minimum quantities +The driver reuses the read-only classification walk before admission. Known +current or unrefreshed behind outputs become `TaskResult` values, without Dask +submission or resource reservations. Tasks that may execute, including dependents +of potentially rebuilt outputs, have their resource requests validated before +preparation. Workers recheck actual upstream digests and may still skip a reserved +task if its inputs prove unchanged. Allocation and task requests share byte +conversion utilities; their models remain distinct because allocation selection supports minimum quantities and node counts. Standard Dask scheduling accounts for concurrent CPU, memory, and GPU reservations; Dask execution-thread counts remain a separate concurrency cap. Reservations do not impose hard limits on recipe @@ -150,7 +152,9 @@ preparation or execution; allocation walltime remains supported. Recipe memory remains ASTRA-style: `8Gi` is binary, `8GB` is decimal, and units are required. Allocation memory follows the compute convention above; keep the -two parsers' contracts explicit even though they share exact byte arithmetic. +two parsers' contracts explicit even though they share exact byte arithmetic in +`units.py`. Allocation duration parsing stays in `compute.model.duration`, raising +`ValueError` for Pydantic; `Request.parse` converts it to `ComputeError`. ## GPU allocation and visibility @@ -167,8 +171,11 @@ custom Dask worker. The sandbox's `use_gpus` policy option inherits the worker's mask for GPU commands and supplies an empty mask for CPU commands, without modifying the reusable worker's environment. Direct GPU policies grant native NVIDIA character devices. -Container GPU execution uses podman-hpc's `--gpu`; ordinary Docker and Podman GPU -requests are refused. Native permissions and cgroups remain authoritative. +Container GPU execution uses podman-hpc's `--gpu`. Explicit GPU recipes on ordinary +Docker or Podman are refused before image preparation; probes use a CPU policy and +report that GPU access is unavailable while retaining their whole-worker reservation. +Native permissions and cgroups remain authoritative. NVIDIA devices, including UVM, +must already exist; policy construction does not load drivers or create devices. See [GPU deployment requirements](../user/cluster.md#gpu-allocations). ## Execution output and teardown diff --git a/docs/api/container.md b/docs/api/container.md index bd360f6c..0bb0f553 100644 --- a/docs/api/container.md +++ b/docs/api/container.md @@ -18,10 +18,10 @@ Sources: `src/lightcone/engine/image.py`, | `image.tag(root)` | `lc-env-<16 hex>` over the rendered Containerfile *and* the identity document. | | `image.archive_path(root, tag)` | `.datalad/environments//image` — the `datalad containers-add` layout. | | `container.build(root)` | Build + save + commit, idempotent; returns `(Runtime, "built" \| "present")`. | -| `container.runtime_for_run(root, *, build)` | One function, two strictnesses: `lc build`/materialize-preflight may build and commit; the probe and worker only ever find, fetch, and load. | +| `container.runtime_for_run(root, *, build, use_gpus=False)` | Resolve the runtime, refusing unsupported explicit GPU requests before preparing the image. Materialize may build and commit; probes and reruns only find, fetch, and load. | | `container.backend(...)` | The single construction point for the exec backend — the only mode branch. | | `container.sync(...)` | The in-container environment converge: network on, project `:rw`, host uv cache mounted, into `.lightcone/venv`. | -| `Runtime` | Facts only — root/mode/name/tag/id/arch — never mechanism. | +| `Runtime` | Resolved execution facts; `supports_gpus` is true for direct mode and podman-hpc. | ## What must stay true diff --git a/docs/api/materialize.md b/docs/api/materialize.md index a7419172..d767404b 100644 --- a/docs/api/materialize.md +++ b/docs/api/materialize.md @@ -25,10 +25,13 @@ driver's stderr, independently of success or failure, leaving stdout for the rep ## The run's order, and why 1. **Read-only project checks before connecting** — tool, committer, dirty-tree, - spec and lock errors do not require a reachable cluster to report. + spec and lock errors do not require a reachable cluster to report. The shared + classification walk identifies outputs already current or left behind. 2. **Explicit cluster before preparing the environment** — validate native - allocation identity, connect, and validate every selected task's CPU/memory/GPU - request before fetching inputs or building an image. + allocation identity, connect, and validate CPU/memory/GPU requests for tasks + that may execute. Known skips become values without resource reservations; + dependents of potentially rebuilt outputs still need admission. Explicit GPU + recipes must also have a supported runtime before any image build. The dirty refusal has already run: in containerized mode the converge can commit an image archive, and `dataset.save` commits the whole index; on a dirty tree the user's diff --git a/docs/api/plan.md b/docs/api/plan.md index abc1641e..49f06a78 100644 --- a/docs/api/plan.md +++ b/docs/api/plan.md @@ -36,8 +36,8 @@ Source: `src/lightcone/engine/plan.py`. mapping. A valid declaration remains readable by `status` and `materialize --check` even when this executor cannot honor it. Execution validates supported requirements through `TaskResources.parse` and checks - cluster capacity before materialize prepares the project or submits any - task. No worker placement or executor-specific resource validation belongs + cluster capacity for tasks that may execute before preparing the project. + Already-current outputs need no resource admission. No worker placement or executor-specific resource validation belongs in this module. - **The layout is flat and path-addressed.** `results//.`, and the path in a diff --git a/docs/api/sandbox.md b/docs/api/sandbox.md index 6a61e0e0..813a1ac5 100644 --- a/docs/api/sandbox.md +++ b/docs/api/sandbox.md @@ -32,9 +32,10 @@ CUDA mask. Direct GPU policies grant existing NVIDIA character device nodes. Native permissions still apply; visibility is cooperative, and the reusable worker's environment is never modified. -The pure OCI rewrite adds podman-hpc's `--gpu` for GPU commands. Ordinary Docker -and Podman GPU execution is explicitly refused. CPU containers remain supported -on all runtimes and set `NVIDIA_VISIBLE_DEVICES=void` to override image defaults. +The pure OCI rewrite adds podman-hpc's `--gpu` for GPU commands. Runtime selection +refuses explicit GPU recipes with ordinary Docker or Podman; probes on those +runtimes receive a CPU policy and an explanatory note. CPU containers remain +supported on all runtimes and set `NVIDIA_VISIBLE_DEVICES=void` to override image defaults. See [GPU allocations](../user/cluster.md#gpu-allocations). ## What must stay true diff --git a/docs/api/worker.md b/docs/api/worker.md index 01d39541..63ecaba8 100644 --- a/docs/api/worker.md +++ b/docs/api/worker.md @@ -22,7 +22,9 @@ retain direct terminal output. The driver submits each cluster task with its CPU, memory, and GPU reservations. Before resetting outputs, `execute` validates resource syntax and builds the command policy with GPU access enabled only when the recipe requests it. Standalone reruns apply the same checks but do not perform -Dask resource admission. Recipe `time_limit` is explicitly refused. +Dask resource admission. GPU reruns require an explicit `CUDA_VISIBLE_DEVICES` in +the rerun environment, for example `CUDA_VISIBLE_DEVICES=0 datalad rerun`. Recipe +`time_limit` is explicitly refused. ## Key symbols diff --git a/docs/architecture.md b/docs/architecture.md index 5eff3927..4c47acf4 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -40,8 +40,9 @@ lc materialize "$CLUSTER" │ refuse: dirty tree │ plan: astra validate + resolve → Graph of Tasks │ (no tasks → converge the crate and stop; nothing connects) + │ classify: current/behind outputs become values; other outputs may run │ connect: native identity + Dask readiness - │ admit: each task's CPU, memory, and GPU request fits a worker + │ admit: resource requests for outputs that may run fit a worker │ fetch: git annex get (declared inputs not in this clone) │ converge: uv.lock ⇄ .venv (and the image, containerized) ├─► workers: reset output file → sandbox → recipe → hash → manifest @@ -58,20 +59,22 @@ The division of labor is strict and load-bearing: `TaskResult`; the driver commits as results arrive, in one thread. Concurrent git operations race on the index lock — this split is not a preference. -- **Dask owns the ordering.** Every task is submitted with its - upstream futures as arguments; there is no ready-set loop or +- **Dask owns the ordering.** Tasks that may execute are submitted with their + upstream futures or already-current values as arguments; there is no ready-set loop or hand-rolled topological sort on the execution path. - **The worker never raises.** A recipe failure, a gate failure, an unreadable manifest — all come back as a state, so one failure doesn't abort every task in flight, and a run reports *all* its independent failures. - **Dask accounts for task resources.** Workers advertise CPU, memory, and GPU - budgets; submissions reserve the recipe's requirements. CPU and memory - reservations coordinate scheduling rather than imposing per-recipe OS limits. + budgets; submissions reserve the recipe's requirements. Omitted memory adds no + RAM reservation. These coordinate scheduling rather than imposing per-recipe + OS limits or numerical-library thread counts. Recipe `time_limit` is explicitly refused; allocation walltime remains supported. A GPU recipe reserves the worker's full GPU budget and inherits its allocation mask, even when it requests fewer GPUs. CPU recipes expose none; probes reserve - the whole worker budget and inherit the allocation mask. + the whole worker budget. Direct and podman-hpc probes use the allocation mask, + while ordinary Docker and Podman probes use no GPUs. - **Values are resolved once and handed down.** HEAD, the container runtime, and the foreign-write facts are read by the driver and passed to workers as values — a worker that asked git itself could @@ -145,8 +148,8 @@ records what was *actually* enforced — never what should have been. There is one policy builder, `exec_policy`: probes and recipes share environment and filesystem rules, with write scope and GPU visibility supplied by the caller. -GPU recipes and probes inherit the worker's allocation mask; CPU commands get an -empty mask. +GPU recipes and supported probes inherit the worker's allocation mask; CPU +commands get an empty mask. ## The container hatch diff --git a/docs/cli/materialize.md b/docs/cli/materialize.md index 0f5fad9d..a323913c 100644 --- a/docs/cli/materialize.md +++ b/docs/cli/materialize.md @@ -20,8 +20,8 @@ cluster that is not active with every expected worker connected is refused cluster. Project validation and the dirty-tree check run before connecting to compute. A run whose spec selects no outputs still takes the CLUSTER argument but never connects to it: it only updates the publication view. A run whose -outputs are all current does connect, because each output's state is decided -on the cluster. +outputs are all current still validates the cluster connection, but submits no +Dask tasks and reserves no recipe resources. With no targets, everything the spec declares, across every universe. A target narrows the run to an output and whatever it depends on: @@ -59,12 +59,13 @@ never touched, under any flag. - **Honors recipe resources.** CPU, memory, and GPU requests must fit one worker and are reserved through standard Dask scheduling. GPU recipes run one at a time per worker and inherit the whole allocation's CUDA mask, which may expose - more GPUs than requested. Recipes without `gpus` see none. The whole selected - graph is checked before preparation or submission. Recipe `time_limit` is - unsupported and refused; allocation walltime remains supported. + more GPUs than requested. Recipes without `gpus` see none. Resource checks apply + to outputs that may rebuild; already-current outputs need no reservation. + Omitted memory reserves no RAM, so CPU requests and task slots control concurrency. + Recipe `time_limit` is unsupported and refused; allocation walltime remains supported. See [recipe resource requirements](../user/cluster.md#recipe-resource-requirements). - **Fetches what it needs.** Declared inputs whose annexed content is - not in this clone are fetched before anything hashes. + not in this clone are fetched before workers hash or execute. - **Commits as it goes.** Each output lands in its own commit, written by the driver in one thread while other recipes keep running. - **Forwards recipe diagnostics.** Recipe stdout and stderr reach the invoking diff --git a/docs/cli/run.md b/docs/cli/run.md index 07a66d7e..312fa587 100644 --- a/docs/cli/run.md +++ b/docs/cli/run.md @@ -40,8 +40,10 @@ command, for example `lc run "$CLUSTER" -- env NAME=value python script.py`. Containerized commands use the image's environment and the sandbox overlays. The command reserves one worker's full CPU, memory, and GPU budgets for its duration. -It inherits the allocation's whole CUDA mask; a CPU-only allocation exposes none, -even on a host with GPUs. A recipe declares its minimum GPU count explicitly. +Direct and podman-hpc probes inherit the allocation's whole CUDA mask. Ordinary +Docker and Podman probes run without GPUs and report that limitation, even on a +GPU allocation. CPU-only allocations expose no GPUs. A recipe declares its +minimum GPU count explicitly; unsupported GPU recipes fail before image preparation. See [GPU allocations](../user/cluster.md#gpu-allocations) for container prerequisites. Interrupting the CLI detaches its client; the remote command may still be running. diff --git a/docs/user/cluster.md b/docs/user/cluster.md index 98931195..35cd38a1 100644 --- a/docs/user/cluster.md +++ b/docs/user/cluster.md @@ -350,7 +350,10 @@ CUDA_VISIBLE_DEVICES=0 lc compute launch --cpus 4 --memory 8GB --gpus GPU:1 Lightcone freezes the nonempty mask and `CUDA_DEVICE_ORDER`, if set, at launch. You are responsible for matching the catalog's count and model to those devices; Lightcone does not verify them. Local allocations do not reserve GPUs exclusively -against other allocations or programs on the host. +against other allocations or programs on the host. Before launch, the host must +have loaded the NVIDIA driver and created its character devices, including UVM. +Lightcone grants existing device nodes and does not initialize them; see +[NVIDIA's device setup utility](https://github.com/NVIDIA/nvidia-modprobe/blob/main/nvidia-modprobe.1.m4). On Slurm, the offer's `config.gpu_type` maps its catalog label to a native GRES type. For example, add this offer with settings adjusted to your site: @@ -382,11 +385,17 @@ permissions remain authoritative. Containerized GPU execution currently supports **podman-hpc** through its native `--gpu` option. See [NERSC's GPU container guidance](https://docs.nersc.gov/development/containers/podman-hpc/overview/#using-nvidia-gpus-in-podman-hpc). -GPU requests with ordinary Docker or Podman are explicitly refused. CPU execution -continues to support all three runtimes and sets `NVIDIA_VISIBLE_DEVICES=void` to -override GPU-enabled image defaults. Physical GPU execution remains a deployment +Recipes explicitly requesting GPUs with ordinary Docker or Podman are refused +before image preparation. `lc run` probes on those runtimes remain usable on a +GPU cluster: they run without GPUs and report that limitation. CPU execution +supports all three runtimes and sets `NVIDIA_VISIBLE_DEVICES=void` to override +GPU-enabled image defaults. Physical GPU execution remains a deployment validation step. +A standalone GPU rerun needs a device mask in its own environment, for example +`CUDA_VISIBLE_DEVICES=0 datalad rerun`. It does not inherit an old allocation's +mask or reserve devices through Dask. + ## Recipe resource requirements Declare each recipe's needs in `astra.yaml`: @@ -408,9 +417,8 @@ limit how many CPUs a single recipe may request. CPUs must be positive whole numbers and default to one. Memory needs units: `512Mi` and `8Gi` are binary sizes; `8GB` is decimal, unlike compute memory. -Bare quantities are not accepted. Without a memory -declaration, a recipe reserves the worker's entire memory budget, so only -one such recipe runs per worker. +Bare quantities are not accepted. Without a memory declaration, no RAM is +reserved: CPU requests and `task_slots_per_node` control concurrency. Recipe `gpus` is a nonnegative whole count, defaulting to zero; accelerator type selection belongs to cluster allocation. A GPU recipe reserves the worker's @@ -419,17 +427,18 @@ requested count is a minimum capacity requirement: the command inherits the worker's whole allocated CUDA mask and may see more GPUs than requested. CPU recipes may still run alongside it when CPU, memory, and task slots permit; their CUDA mask is empty. `lc run` reserves the worker's entire CPU, memory, and GPU -budgets and inherits that same allocation mask. +budgets; direct and podman-hpc probes inherit that allocation mask. Recipe `time_limit` is not supported and is refused before preparation or execution. Set the allocation lifetime with `lc compute launch --time` instead. Fractional CPU/GPU counts, GPU model requests inside a recipe, and disk requests are also rejected rather than ignored. -`lc materialize` validates the complete selected graph against the cluster -before fetching inputs, preparing the environment, or starting a recipe. This -also validates currently complete outputs, which workers may need to rebuild -after an upstream change. Use `lc materialize --check` to inspect currency +`lc materialize` first classifies the selected graph, then checks resources for +outputs that may rebuild before preparation or submission. Already-current or +unrefreshed behind outputs reserve nothing. Dependents of an output that may +change still need resources; the worker may later skip them if the actual +upstream digest is unchanged. Use `lc materialize --check` to inspect currency without allocation. Read-only `status` and `--check` accept valid ASTRA resource declarations even when this executor cannot satisfy them. @@ -439,6 +448,14 @@ its request. Slurm enforces the overall allocation, while local execution uses cooperative budgets. Leave capacity for the scheduler, workers, and other overhead when declaring recipe requirements. +A CPU reservation does not set numerical-library thread counts. Local clusters +use Dask's Nanny defaults of `1` for `OMP_NUM_THREADS`, `MKL_NUM_THREADS`, and +`OPENBLAS_NUM_THREADS` when those variables are unset. Set the variables before +`lc compute launch`, or in the recipe command, to choose another value. See +[Dask's defaults](https://docs.dask.org/en/stable/configuration.html#distributed.nanny.pre-spawn-environ.OMP_NUM_THREADS) +and [environment precedence](https://distributed.dask.org/en/stable/_modules/distributed/nanny.html). +Slurm workers run directly without a Nanny and inherit the job's thread settings. + ## Execution requirements and limits Driver and workers must see the same project, prepared environment, and inputs diff --git a/src/lightcone/engine/compute/local.py b/src/lightcone/engine/compute/local.py index 9d83cc38..85b05272 100644 --- a/src/lightcone/engine/compute/local.py +++ b/src/lightcone/engine/compute/local.py @@ -252,6 +252,8 @@ def _record(self, identity: Identity) -> tuple[Path, dict[str, Any]]: raise ComputeError("the private locator does not match this local allocation identity") positive_int(record.get("cpus"), "recorded local cpus") positive_int(record.get("memory"), "recorded local memory") + record.setdefault("gpus", 0) + record.setdefault("accelerator_name", "GPU") if type(record.get("gpus")) is not int or record["gpus"] < 0: raise ComputeError("recorded local gpus must be a nonnegative integer") if not isinstance(record.get("accelerator_name"), str) or not record["accelerator_name"]: diff --git a/src/lightcone/engine/compute/local_runtime.py b/src/lightcone/engine/compute/local_runtime.py index 222a810f..1ca6d552 100644 --- a/src/lightcone/engine/compute/local_runtime.py +++ b/src/lightcone/engine/compute/local_runtime.py @@ -55,7 +55,7 @@ def expire(_signum: int, _frame: FrameType | None) -> None: security = create_security(directory) allocation = read_private_json(directory / "identity.json") - gpus = int(allocation["gpus"]) + gpus = int(allocation.get("gpus", 0)) if not gpus: os.environ["CUDA_VISIBLE_DEVICES"] = "" with dask.config.set(SCHEDULER_CONFIG), LocalCluster( # type: ignore[no-untyped-call] diff --git a/src/lightcone/engine/compute/model.py b/src/lightcone/engine/compute/model.py index b29cd2c2..eb16380f 100644 --- a/src/lightcone/engine/compute/model.py +++ b/src/lightcone/engine/compute/model.py @@ -19,12 +19,13 @@ Field, PlainSerializer, ValidationError, + field_validator, model_serializer, model_validator, ) from lightcone.engine.project import ProjectError -from lightcone.engine.units import duration_seconds, whole_bytes +from lightcone.engine.units import whole_bytes GIB = 1024**3 @@ -45,11 +46,15 @@ class UnavailableOfferError(ComputeError): def duration(value: object) -> int: - """Parse an explicit positive duration into seconds.""" - try: - return duration_seconds(value) - except ValueError as exc: - raise ComputeError(str(exc)) from exc + """Parse ordered day/hour/minute/second components, raising ValueError.""" + if not isinstance(value, str) or not ( + match := re.fullmatch(r"(?:(\d+)d)?(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?", value) + ): + raise ValueError("duration must use explicit units, e.g. 30m, 1h30m, or 45s") + seconds = sum(int(part or 0) * unit for part, unit in zip(match.groups(), (86400, 3600, 60, 1))) + if seconds <= 0: + raise ValueError("duration must be positive") + return seconds def memory_bytes(value: object) -> int: @@ -132,10 +137,7 @@ def _gib(value: object) -> Decimal: def _duration(value: str) -> str: - try: - duration(value) - except ComputeError as exc: - raise ValueError(str(exc)) from exc + duration(value) return value @@ -168,6 +170,13 @@ class Accelerator(ComputeModel): name: Annotated[str, Field(pattern=r"^[A-Za-z0-9][A-Za-z0-9_.-]*$")] count: PositiveInt = 1 + @field_validator("name") + @classmethod + def unambiguous_name(cls, value: str) -> str: + if value in {"name", "count"}: + raise ValueError("accelerator names cannot be reserved fields 'name' or 'count'") + return value + @model_validator(mode="before") @classmethod def shorthand(cls, value: Any) -> Any: @@ -176,7 +185,7 @@ def shorthand(cls, value: Any) -> Any: if match is None: raise ValueError("accelerators must be NAME[:COUNT] with a positive whole count") return {"name": match[1], "count": int(match[2] or 1)} - if isinstance(value, dict) and not ("name" in value and set(value) <= {"name", "count"}): + if isinstance(value, dict) and not value.keys() & {"name", "count"}: if len(value) != 1: raise ValueError("accelerators must specify exactly one type and whole count") name, count = next(iter(value.items())) @@ -246,14 +255,6 @@ class Request(ComputeModel): seconds: PositiveInt | None = None startup: Literal["fast"] | None = None - @property - def gpus(self) -> int: - return self.accelerators.count if self.accelerators is not None else 0 - - @property - def accelerator_name(self) -> str | None: - return self.accelerators.name if self.accelerators is not None else None - @classmethod def parse( cls, @@ -279,6 +280,8 @@ def parse( }) except ValidationError as exc: raise ComputeError(f"invalid compute request:\n{validation_message(exc)}") from exc + except ValueError as exc: + raise ComputeError(str(exc)) from exc def as_dict(self) -> dict[str, Any]: """Render the request without native provider settings.""" diff --git a/src/lightcone/engine/compute/slurm.py b/src/lightcone/engine/compute/slurm.py index cd43f510..7cad67aa 100644 --- a/src/lightcone/engine/compute/slurm.py +++ b/src/lightcone/engine/compute/slurm.py @@ -102,15 +102,17 @@ def _value(value: object, name: str) -> str: def _native_gpus(row: Mapping[str, str]) -> tuple[str, int] | None: """Read a per-node GPU count, never divide an aggregate into invented grants.""" - for field, prefix in (("TresPerNode", "gres/"), ("Gres", "")): + for field in ("TresPerNode", "Gres"): value = row.get(field, "") counts = [] names = set() for entry in value.split(","): - if not entry.startswith(f"{prefix}gpu"): + if field == "TresPerNode": + entry = entry.removeprefix("gres/").removeprefix("gres:") + if not entry.startswith("gpu"): continue match = re.fullmatch( - rf"{prefix}gpu(?::([A-Za-z0-9][A-Za-z0-9_.-]*))?[:=]([0-9]+)", entry, + r"gpu(?::([A-Za-z0-9][A-Za-z0-9_.-]*))?[:=]([0-9]+)", entry, ) if match is None: return None @@ -124,7 +126,7 @@ def _native_gpus(row: Mapping[str, str]) -> tuple[str, int] | None: return "GPU", 0 # Complete native TRES with no GPU entry proves a CPU-only job. An # aggregate GPU total does not prove a homogeneous per-node allocation. - for field in ("ReqTRES", "AllocTRES"): + for field in ("ReqTRES", "AllocTRES", "TRES"): value = row.get(field, "") if value and value not in {"(null)", "N/A"}: entries = value.split(",") diff --git a/src/lightcone/engine/compute/slurm_bootstrap.py b/src/lightcone/engine/compute/slurm_bootstrap.py index 3db01229..2ef71a3c 100644 --- a/src/lightcone/engine/compute/slurm_bootstrap.py +++ b/src/lightcone/engine/compute/slurm_bootstrap.py @@ -68,8 +68,6 @@ def _allocation(args: argparse.Namespace) -> tuple[Identity, int, int]: raise ComputeError("native per-node GPUs do not match the allocation envelope") if not os.environ.get("CUDA_VISIBLE_DEVICES"): raise ComputeError("GPU allocations require a native CUDA_VISIBLE_DEVICES mask") - # Slurm/NVML numbers devices in PCI order; CUDA's default is different. - os.environ["CUDA_DEVICE_ORDER"] = "PCI_BUS_ID" restarts = os.environ.get("SLURM_RESTART_COUNT", "0") if not restarts.isdigit(): raise ComputeError("invalid native Slurm restart count") @@ -85,7 +83,10 @@ async def run(args: argparse.Namespace) -> None: from distributed import Scheduler, Worker identity, restarts, rank = _allocation(args) - if not args.gpus: + if args.gpus: + # Slurm/NVML numbers devices in PCI order; CUDA's default is different. + os.environ["CUDA_DEVICE_ORDER"] = "PCI_BUS_ID" + else: os.environ["CUDA_VISIBLE_DEVICES"] = "" connection = Connection( namespace=identity.namespace, provider="slurm", @@ -170,8 +171,9 @@ def main() -> None: parser = argparse.ArgumentParser(description=__doc__) for name in ("submission", "namespace", "connection-root"): parser.add_argument(f"--{name}", required=True) - for name in ("num-nodes", "cpus", "memory-bytes", "gpus", "task-slots"): + for name in ("num-nodes", "cpus", "memory-bytes", "task-slots"): parser.add_argument(f"--{name}", required=True, type=int) + parser.add_argument("--gpus", default=0, type=int) parser.add_argument("--scratch-root") parser.add_argument("--interface") args = parser.parse_args() diff --git a/src/lightcone/engine/container.py b/src/lightcone/engine/container.py index 8f3f01d8..b109b482 100644 --- a/src/lightcone/engine/container.py +++ b/src/lightcone/engine/container.py @@ -66,6 +66,13 @@ class Runtime: #: The architecture the archive was built for. arch: str = "" + @property + def supports_gpus(self) -> bool: + """Whether this execution mode can expose the allocation's GPUs.""" + from lightcone.engine.sandbox.oci import supports_gpus + + return self.mode == "direct" or supports_gpus(self.runtime) + @property def archive(self) -> str: """The committed archive, project-relative — what the run record's @@ -87,7 +94,7 @@ def manifest_image(self) -> dict[str, str] | None: } -def runtime_for_run(root: Path, *, build: bool) -> Runtime: +def runtime_for_run(root: Path, *, build: bool, use_gpus: bool = False) -> Runtime: """Resolve the execution world, converging the image where allowed. The three image checks are repository questions first and runtime @@ -102,6 +109,7 @@ def runtime_for_run(root: Path, *, build: bool) -> Runtime: Args: root: The project root. build: Whether a missing archive may be built and committed. + use_gpus: Refuse unsupported GPU containers before preparing their image. Returns: The resolved runtime; a direct-mode one costs a TOML read. @@ -115,6 +123,10 @@ def runtime_for_run(root: Path, *, build: bool) -> Runtime: return Runtime(root=root, mode="direct", env_dir=project.env_dir(root)) name = runtime_name(root) + if use_gpus: + from lightcone.engine.sandbox.oci import require_gpu_runtime + + require_gpu_runtime(name) tag = image.tag(root) archive = image.archive_path(root, tag) if not _committed(archive): @@ -689,4 +701,3 @@ def _machine_preflight(root: Path) -> None: f"empty. Share it:\n podman machine stop\n" f" podman machine set --volume {root}\n podman machine start" ) - diff --git a/src/lightcone/engine/execution_resources.py b/src/lightcone/engine/execution_resources.py index 793e95ec..e9d71209 100644 --- a/src/lightcone/engine/execution_resources.py +++ b/src/lightcone/engine/execution_resources.py @@ -69,17 +69,17 @@ def parse(cls, value: object) -> Self: raise ProjectError(f"invalid recipe resources: {detail}") from exc def requirements( - self, workers: dict[str, Any], *, whole_worker: bool = False + self, capacities: set[tuple[float, float, float]], *, whole_worker: bool = False ) -> dict[str, float]: """Choose Dask resource reservations that fit an individual worker. Args: - workers: The ``workers`` mapping from Dask's scheduler information. + capacities: Validated worker budgets from :func:`worker_capacities`. whole_worker: Reserve a worker's entire CPU, memory, and GPU budget for an arbitrary command without declared resource requirements. Returns: - Dask's numeric ``CPU``, ``MEMORY``, and optional ``GPU`` reservations. + Dask's numeric ``CPU`` and optional ``MEMORY`` and ``GPU`` reservations. GPU recipes reserve the worker's full GPU budget, so only one GPU recipe uses that worker's native device mask at a time. The requested count is a minimum capacity, not a per-command visibility limit. @@ -88,55 +88,29 @@ def requirements( ProjectError: Capacity is unknown, a request cannot fit, or an unspecified budget is ambiguous across heterogeneous workers. """ - capacities: set[tuple[float, float, float]] = set() - for info in workers.values(): - resources = info.get("resources", {}) if isinstance(info, dict) else {} - values = [] - for name in ("CPU", "MEMORY", "GPU"): - value = ( - resources.get(name, 0 if name == "GPU" else None) - if isinstance(resources, dict) else None - ) - if ( - isinstance(value, bool) - or not isinstance(value, (int, float)) - or not math.isfinite(value) - or (value < 0 if name == "GPU" else value <= 0) - or not float(value).is_integer() - ): - raise ProjectError( - "cluster workers must advertise positive whole CPU and MEMORY budgets " - "and a nonnegative whole GPU count; " - "relaunch the cluster with the current Lightcone installation" - ) - values.append(float(value)) - capacities.add((values[0], values[1], values[2])) - if not capacities: - raise ProjectError("cluster has no workers available for execution") if ( - (whole_worker or self.memory_bytes is None) + whole_worker and len({(cpus, memory) for cpus, memory, _ in capacities}) != 1 ): raise ProjectError( - "unspecified task resources require workers with identical CPU and memory budgets" + "whole-worker probes require workers with identical CPU and memory budgets" ) available_cpus, available_memory, _ = next(iter(capacities)) - requested = { - "CPU": available_cpus if whole_worker else float(self.cpus), - "MEMORY": ( - available_memory - if whole_worker or self.memory_bytes is None - else float(self.memory_bytes) - ), - } + requested = {"CPU": available_cpus if whole_worker else float(self.cpus)} + if whole_worker: + requested["MEMORY"] = available_memory + elif self.memory_bytes is not None: + requested["MEMORY"] = float(self.memory_bytes) matches = { gpus for cpus, memory, gpus in capacities - if cpus >= requested["CPU"] and memory >= requested["MEMORY"] and gpus >= self.gpus + if cpus >= requested["CPU"] + and memory >= requested.get("MEMORY", 0) + and gpus >= self.gpus } if not matches: raise ProjectError( f"task needs {requested['CPU']:g} CPUs and " - f"{requested['MEMORY'] / 1024**3:g} GiB and {self.gpus} GPUs on one worker; " + f"{requested.get('MEMORY', 0) / 1024**3:g} GiB and {self.gpus} GPUs on one worker; " "no worker in this cluster can satisfy that request" ) if self.gpus or whole_worker: @@ -149,6 +123,40 @@ def requirements( return requested +def worker_capacities(workers: dict[str, Any]) -> set[tuple[float, float, float]]: + """Read distinct CPU, memory and GPU budgets once from scheduler information. + + Raises: + ProjectError: No workers are available or their resource budgets are invalid. + """ + capacities: set[tuple[float, float, float]] = set() + for info in workers.values(): + resources = info.get("resources", {}) if isinstance(info, dict) else {} + values = [] + for name in ("CPU", "MEMORY", "GPU"): + value = ( + resources.get(name, 0 if name == "GPU" else None) + if isinstance(resources, dict) else None + ) + if ( + isinstance(value, bool) + or not isinstance(value, (int, float)) + or not math.isfinite(value) + or (value < 0 if name == "GPU" else value <= 0) + or not float(value).is_integer() + ): + raise ProjectError( + "cluster workers must advertise positive whole CPU and MEMORY budgets " + "and a nonnegative whole GPU count; " + "relaunch the cluster with the current Lightcone installation" + ) + values.append(float(value)) + capacities.add((values[0], values[1], values[2])) + if not capacities: + raise ProjectError("cluster has no workers available for execution") + return capacities + + def _memory(value: object) -> int: if not isinstance(value, str) or not ( match := re.fullmatch(r"([0-9]+(?:\.[0-9]+)?)\s*([KMGTPE]i?B?|kB?|B)", value) diff --git a/src/lightcone/engine/materialize.py b/src/lightcone/engine/materialize.py index 8b956eac..60247ca1 100644 --- a/src/lightcone/engine/materialize.py +++ b/src/lightcone/engine/materialize.py @@ -6,10 +6,10 @@ committed together with the code that produced it, so a run that began with uncommitted changes could not honestly say which code that was. -**It hands the graph to Dask and gets out of the way.** Every task is -submitted with its upstream futures as arguments, so the ordering, the -parallelism, and the scheduling are Dask's — there is no ready-set loop -here to get wrong. +**It hands the work to Dask and gets out of the way.** Tasks that may run +receive upstream futures or already-current results as arguments, so the +ordering, parallelism, and scheduling are Dask's — there is no ready-set +loop here to get wrong. **It owns git, alone.** Workers execute and return; the driver commits, in one thread, as results arrive. That is not a preference: concurrent git @@ -42,7 +42,7 @@ from uuid import uuid4 from lightcone.engine import assets, container, dataset, identity, plan, project, worker -from lightcone.engine.execution_resources import TaskResources +from lightcone.engine.execution_resources import TaskResources, worker_capacities from lightcone.engine.plan import Graph, Key, Task from lightcone.engine.project import ProjectError @@ -174,10 +174,31 @@ def _classified( first. """ graph, env_version, _ = _graph(root, targets, report) - versions = assets.Versions() - would_run: set[Key] = set() unfetched: set[str] = set() + classified = _classify_graph( + root, graph, env_version, refresh=refresh, + versions=assets.Versions(), unfetched=unfetched, + ) + if unfetched: + report.warnings.append( + "reported as out of date because their content is not in this " + f"clone, not because they changed: {', '.join(sorted(unfetched))}. " + "`lc materialize` fetches declared inputs before executing " + "recipes. Compute is required for outputs whose inputs cannot yet be checked." + ) + return classified + + +def _classify_graph( + root: Path, graph: Graph, env_version: str, *, refresh: bool, + versions: assets.Versions, unfetched: set[str], +) -> list[tuple[Key, assets.Verdict, assets.Manifest | None, dataset.LastWrite | None]]: + """Predict which outputs may run, using the same walk for checks and admission. + Dependents of an output that may run are conservatively included: only the + worker can know whether rebuilding that input actually changed its bytes. + """ + would_run: set[Key] = set() classified = [] for key in graph.order(): task = graph.tasks[key] @@ -197,13 +218,6 @@ def _classified( would_run.add(key) classified.append((key, verdict, manifest, foreign)) - if unfetched: - report.warnings.append( - "reported as out of date because their content is not in this " - f"clone, not because they changed: {', '.join(sorted(unfetched))}. " - "`lc materialize` fetches declared inputs before it decides " - "anything, so there this resolves itself." - ) return classified @@ -516,13 +530,22 @@ def materialize( # maintainer. Nothing is submitted, so no allocation is needed. _converge_crate(root, report, full, dsid) return report + versions = assets.Versions() + classified = _classify_graph( + root, graph, env_version, refresh=refresh, versions=versions, unfetched=set(), + ) with cluster_for_run(cluster_id) as scheduler: - requirements = scheduler.validate(graph.tasks.values()) + requirements = scheduler.validate( + graph.tasks[key] for key, verdict, _, _ in classified + if verdict.calls_for_a_remake(refresh=refresh) + ) _fetch_inputs(root, graph, report) # Materialize is one of the two verbs allowed to build the image (the # other is `lc build`); the probe and the rerun entry point only find # one. Resolved once, then handed to every task — the HEAD discipline. - runtime = container.runtime_for_run(root, build=True) + runtime = container.runtime_for_run( + root, build=True, use_gpus=any(r.get("GPU", 0) for r in requirements.values()), + ) # Converge the environment: workers pass `--no-sync`, so this is the # only place on a run's path where it is made to match the lock. (A # rerun does not come through here; its entry point converges too.) @@ -533,7 +556,6 @@ def materialize( # because attestation is a fact about the run (and empty is an # answer, not a failure); one content-hash memo because a declared # input shared by several outputs is the same bytes every time. - versions = assets.Versions() for path in { path for task in graph.tasks.values() @@ -552,40 +574,38 @@ def materialize( runtime=runtime, uv_version=project.uv_version(root), ) - # The history question is the driver's to answer — workers have no - # git, by design — so each task is told up front whether its - # directory was last written by something other than its own run - # record. A foreign write contradicts the manifest, and a worker that - # trusted the recorded digest would skip the output forever. Guarded - # on the manifest's presence, as `_classified` is: without one the - # answer is dead — the output is remade regardless — and each ask is - # a git process. - foreign = { - key: _foreign_write(root, task) if task.manifest_path.is_file() else None - for key, task in graph.tasks.items() - } pending: dict[Key, Any] = {} - # Futures retain dependency ordering; task placement belongs to the - # selected cluster, while commits stay in this one driver thread. - for key in graph.order(): + handles = [] + # Current outputs are values, not Dask tasks: they need no allocation + # resources. Futures for everything that may run retain dependency order. + for key, verdict, manifest, foreign in classified: task = graph.tasks[key] + if not verdict.calls_for_a_remake(refresh=refresh): + assert manifest is not None and verdict.status != "stale" + result = worker.TaskResult( + key, verdict.status, data_version=manifest.data_version, reason=verdict.why, + ) + pending[key] = result + _consume(root, task, result, dsid, runtime, report) + continue pending[key] = scheduler.submit( worker.materialize, root, task, context, refresh, - foreign[key], + foreign, *[pending[dep] for dep in task.depends_on], key=_name(key), resources=requirements[key], ) + handles.append(pending[key]) from lightcone.engine.compute import UNSTOPPED # An unreported task can still have a running subprocess. Leave its # partial files in place on interruption rather than restoring over it. - outstanding = len(pending) - for result in scheduler.completed(list(pending.values())): + outstanding = len(handles) + for result in scheduler.completed(handles): outstanding -= 1 try: _consume(root, graph.tasks[result.key], result, dsid, runtime, report) @@ -690,10 +710,11 @@ class _Dask: def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: """Require each selected task to fit a worker before any task starts.""" + capacities = worker_capacities(self.workers) requests = {} for task in tasks: try: - requests[task.key] = TaskResources.parse(task.resources).requirements(self.workers) + requests[task.key] = TaskResources.parse(task.resources).requirements(capacities) except ProjectError as exc: raise ProjectError(f"{_name(task.key)}: {exc}") from exc return requests @@ -750,7 +771,7 @@ def cluster_for_run(cluster_id: str) -> Iterator[Scheduler]: def _fetch_inputs(root: Path, graph: Graph, report: MaterializeReport) -> None: - """Bring declared inputs' bytes into this clone before anything hashes. + """Bring declared inputs' bytes into this clone before workers hash or execute. lc fetches rather than telling anyone to — the storage invariant — and only here: ``--check`` and ``status`` are read-only verbs that diff --git a/src/lightcone/engine/run.py b/src/lightcone/engine/run.py index 2110cc20..cc4cbfb6 100644 --- a/src/lightcone/engine/run.py +++ b/src/lightcone/engine/run.py @@ -21,7 +21,7 @@ from uuid import uuid4 from lightcone.engine import container, sandbox -from lightcone.engine.execution_resources import TaskResources +from lightcone.engine.execution_resources import TaskResources, worker_capacities from lightcone.engine.project import ( SPEC_FILENAME, ProjectError, @@ -53,15 +53,20 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. paths = input_paths(project, read_spec(project)) with compute.connect(cluster_id) as client: resources = TaskResources().requirements( - client.scheduler_info()["workers"], whole_worker=True, + worker_capacities(client.scheduler_info()["workers"]), whole_worker=True, ) runtime = container.runtime_for_run(project, build=False) notes = [f"uv: {warning}" for warning in container.converge(runtime)] + use_gpus = resources.get("GPU", 0) > 0 and runtime.supports_gpus + if resources.get("GPU", 0) > 0 and not use_gpus: + notes.append( + f"GPU access is not supported by {runtime.runtime}; this probe runs without GPUs" + ) invocation = uuid4().hex with forwarding(client) as output: future = client.submit( call, _probe, output.topic, "probe", runtime, paths, tuple(command), - resources.get("GPU", 0) > 0, + use_gpus, key=f"lc-{invocation}-probe", pure=False, resources=resources, ) try: diff --git a/src/lightcone/engine/sandbox/oci.py b/src/lightcone/engine/sandbox/oci.py index dd206f69..ea9f2063 100644 --- a/src/lightcone/engine/sandbox/oci.py +++ b/src/lightcone/engine/sandbox/oci.py @@ -31,9 +31,14 @@ OCIRuntime = Literal["podman", "docker", "podman-hpc"] +def supports_gpus(runtime: str) -> bool: + """Whether the runtime can preserve the native allocation's GPU assignment.""" + return runtime == "podman-hpc" + + def require_gpu_runtime(runtime: str) -> None: """Refuse runtimes that need device translation outside the native allocation.""" - if runtime != "podman-hpc": + if not supports_gpus(runtime): raise ProjectError( "GPU containers require podman-hpc; Docker/Podman GPUs are not supported" ) diff --git a/src/lightcone/engine/sandbox/policy.py b/src/lightcone/engine/sandbox/policy.py index a6627e3b..6128af66 100644 --- a/src/lightcone/engine/sandbox/policy.py +++ b/src/lightcone/engine/sandbox/policy.py @@ -236,7 +236,11 @@ def exec_policy( """ gpu_mask = os.environ.get("CUDA_VISIBLE_DEVICES", "") if use_gpus else "" if use_gpus and not gpu_mask: - raise ProjectError("GPU execution requires a nonempty allocation CUDA_VISIBLE_DEVICES mask") + raise ProjectError( + "GPU execution requires a nonempty CUDA_VISIBLE_DEVICES mask. " + "Slurm sets it for GPU jobs; for a local rerun, select your devices explicitly, " + "for example: CUDA_VISIBLE_DEVICES=0 datalad rerun" + ) env_dir = env_dir if env_dir is not None else project / ".venv" # The containerized HOME lives under the project's own (gitignored) # `.lightcone/`, not the system temp dir: it is a mount source, and diff --git a/src/lightcone/engine/units.py b/src/lightcone/engine/units.py index 2d47130a..0bc1b48f 100644 --- a/src/lightcone/engine/units.py +++ b/src/lightcone/engine/units.py @@ -1,8 +1,7 @@ -"""Exact byte quantities and explicit walltime units used by execution and compute.""" +"""Exact byte quantities used by execution and compute.""" from __future__ import annotations -import re from decimal import Decimal, InvalidOperation @@ -24,22 +23,3 @@ def whole_bytes(amount: str, unit_bytes: int) -> int: if result <= 0 or remainder: raise ValueError("memory must be positive and exactly representable in bytes") return result - - -def duration_seconds(value: object) -> int: - """Parse a positive duration such as ``30m``, ``1h30m``, or ``45s``. - - Args: - value: Ordered day, hour, minute, and second components. - - Raises: - ValueError: The duration is malformed, empty, or zero. - """ - if not isinstance(value, str) or not ( - match := re.fullmatch(r"(?:(\d+)d)?(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?", value) - ): - raise ValueError("duration must use explicit units, e.g. 30m, 1h30m, or 45s") - seconds = sum(int(part or 0) * unit for part, unit in zip(match.groups(), (86400, 3600, 60, 1))) - if seconds <= 0: - raise ValueError("duration must be positive") - return seconds diff --git a/src/lightcone/engine/worker.py b/src/lightcone/engine/worker.py index e6662b8b..c9ffa014 100644 --- a/src/lightcone/engine/worker.py +++ b/src/lightcone/engine/worker.py @@ -417,7 +417,9 @@ def main(argv: list[str]) -> int: # This one-task run resolves its own runtime and HEAD, because it # *is* the driver here — the rule is that each is read once by # whoever owns the run, not that a worker never reads them. - runtime = container.runtime_for_run(root, build=False) + runtime = container.runtime_for_run( + root, build=False, use_gpus=TaskResources.parse(task.resources).gpus > 0, + ) container.converge(runtime) result = execute( root, diff --git a/tests/test_compute.py b/tests/test_compute.py index da51a264..29dfead0 100644 --- a/tests/test_compute.py +++ b/tests/test_compute.py @@ -399,6 +399,12 @@ def test_accelerator_alternatives_are_not_silently_treated_as_capacity(value: ob Resources.model_validate({"cpus": 1, "memory": 1, "accelerators": value}) +@pytest.mark.parametrize("value", [{"count": 2}, {"name": 2}, "count:2", "name:2"]) +def test_accelerator_field_names_are_not_misread_as_device_types(value: object) -> None: + with pytest.raises(ValidationError): + Resources.model_validate({"cpus": 1, "memory": 1, "accelerators": value}) + + @pytest.mark.parametrize("count", [-1, 0, True, 0.5, "1", None]) def test_accelerator_envelopes_require_positive_integer_counts(count: object) -> None: with pytest.raises(ValidationError): @@ -419,10 +425,10 @@ def test_accelerator_types_and_counts_survive_native_and_request_roundtrips() -> assert Resources.model_validate_json(resource.model_dump_json(by_alias=True)) == resource assert resource.replace(cpus=4).accelerators == Accelerator(name="A100", count=4) request = Request.parse("8", "16", gpus="A100:2") - assert request.gpus == 2 and request.accelerator_name == "A100" + assert request.accelerators == Accelerator(name="A100", count=2) assert request.as_dict()["resources"]["accelerators"] == {"A100": 2} - assert Request.parse("8", "16", gpus="A100").gpus == 1 - assert Request.parse("8", "16").gpus == 0 + assert Request.parse("8", "16", gpus="A100").accelerators == Accelerator(name="A100", count=1) + assert Request.parse("8", "16").accelerators is None def test_numeric_native_accelerator_names_remain_types() -> None: @@ -509,8 +515,12 @@ def test_allocation_durations_accept_compound_units(value: str, seconds: int) -> @pytest.mark.parametrize("value", ["", "0s", "1.5h", "30m1h", "1h30", 60, True]) def test_allocation_duration_refuses_ambiguous_or_zero_values(value: object) -> None: - with pytest.raises(ComputeError): + with pytest.raises(ValueError): duration(value) + with pytest.raises(ValidationError): + TimeLimits(default=value, max="3d") + with pytest.raises(ComputeError): + Request.parse("1", "1", time=value) @pytest.mark.parametrize("size", [1, GIB // 2, 8 * GIB, 2**80 + 1]) diff --git a/tests/test_compute_local.py b/tests/test_compute_local.py index 4aee5a15..c50ecf17 100644 --- a/tests/test_compute_local.py +++ b/tests/test_compute_local.py @@ -226,6 +226,51 @@ def expand(path): assert Compute().discover() == ([], {}) +def test_cpu_allocation_without_gpu_fields_remains_discoverable_and_stoppable( + provider: LocalProvider, +) -> None: + identity = _launch(provider) + try: + _ready(provider, identity) + path = provider.root / identity.token / "identity.json" + record = read_private_json(path) + del record["gpus"], record["accelerator_name"] + write_private_json(path, record) + + snapshot, = provider.discover() + assert snapshot.identity == identity + assert snapshot.resources is not None + assert snapshot.resources.gpus == 0 + assert snapshot.resources.accelerator_name is None + with provider.connect(identity) as client: + assert client.submit(sum, [1, 2]).result(timeout=5) == 3 + finally: + provider.terminate(identity) + _ended(provider, identity) + + +@pytest.mark.parametrize("threads", [None, "2"]) +def test_local_workers_keep_native_dask_thread_defaults_and_launch_overrides( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, threads: str | None, +) -> None: + names = ("OMP_NUM_THREADS", "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS") + for name in names: + if threads is None: + monkeypatch.delenv(name, raising=False) + else: + monkeypatch.setenv(name, threads) + identity = _launch(provider) + try: + _ready(provider, identity) + with provider.connect(identity) as client: + actual = client.submit(lambda: {name: os.environ.get(name) for name in names}).result( + timeout=5, + ) + assert actual == dict.fromkeys(names, threads or "1") + finally: + provider.terminate(identity) + + def test_named_local_allocation_is_discovered_and_name_can_be_reused_after_down( provider: LocalProvider, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -920,9 +965,9 @@ def test_local_gpu_offers_are_unavailable_outside_linux( provider.plan(offer, Request.parse("1", "0.5", gpus="GPU:1")) -@pytest.mark.parametrize("gpus", [0, 2]) +@pytest.mark.parametrize("gpus", [None, 0, 2]) def test_local_runtime_advertises_configured_resources_and_preserves_native_gpu_mask( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, gpus: int, + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, gpus: int | None, ) -> None: import distributed @@ -932,7 +977,10 @@ def test_local_runtime_advertises_configured_resources_and_preserves_native_gpu_ "identity": "allocation", "deadline": time.monotonic() + 60, "task_slots": 1, "scratch": str(scratch), }) - write_private_json(directory / "identity.json", {"cpus": 1, "memory": 1024**3, "gpus": gpus}) + record = {"cpus": 1, "memory": 1024**3} + if gpus is not None: + record["gpus"] = gpus + write_private_json(directory / "identity.json", record) monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "3,1") monkeypatch.setattr(sys, "argv", ["local_runtime", str(directory)]) monkeypatch.setattr(os, "getsid", lambda _: os.getpid()) @@ -950,5 +998,5 @@ def test_local_runtime_advertises_configured_resources_and_preserves_native_gpu_ monkeypatch.setattr(distributed, "LocalCluster", cluster) local_runtime.main() - assert cluster.call_args.kwargs["resources"] == {"CPU": 1, "MEMORY": 1024**3, "GPU": gpus} + assert cluster.call_args.kwargs["resources"] == {"CPU": 1, "MEMORY": 1024**3, "GPU": gpus or 0} assert os.environ["CUDA_VISIBLE_DEVICES"] == ("3,1" if gpus else "") diff --git a/tests/test_compute_slurm.py b/tests/test_compute_slurm.py index 4d2ed937..d51e40ec 100644 --- a/tests/test_compute_slurm.py +++ b/tests/test_compute_slurm.py @@ -752,14 +752,20 @@ def test_discovery_preserves_allocations_with_unknown_native_resource_evidence( @pytest.mark.parametrize("native,expected,name", [ ("TresPerNode=gres/gpu:4", 4, "GPU"), + ("TresPerNode=gres:gpu:4", 4, "GPU"), + ("TresPerNode=gpu:4", 4, "GPU"), ("TresPerNode=gres/gpu:a100:4", 4, "a100"), + ("TresPerNode=gres:gpu:a100:4", 4, "a100"), ("TresPerNode=gres/gpu:4090:2", 2, "4090"), ("TresPerNode=gres/gpu:a100:2,gres/gpu:v100:2", 4, "GPU"), ("Gres=gpu:a100:2", 2, "a100"), ("Gres=(null)", 0, None), ("ReqTRES=cpu=512,mem=960G,node=2", 0, None), + ("TRES=cpu=512,mem=960G,node=2", 0, None), ("ReqTRES=cpu=512,mem=960G,node=2,gres/gpu=8", None, None), + ("TRES=cpu=512,mem=960G,node=2,gres/gpu=8", None, None), ("TresPerNode=gres/gpu:unknown", None, None), + ("TresPerNode=gres:gpu:unknown TRES=cpu=512,mem=960G,node=2", None, None), ("TresPerNode=gres/gpu:a100/80gb:2", None, None), ("TresPerNode=gres/gpu:1.5", None, None), ("TresPerNode=gres/gpu:4,gres/gpu:a100:4", None, None), @@ -1043,11 +1049,11 @@ def test_gpu_bootstrap_requires_native_capacity_without_probing_or_rewriting_the if valid: _, _, rank = slurm_bootstrap._allocation(args) assert rank == 0 - assert os.environ["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" else: with pytest.raises(ComputeError, match="GPUs"): slurm_bootstrap._allocation(args) assert os.environ["CUDA_VISIBLE_DEVICES"] == "1,3" + assert os.environ["CUDA_DEVICE_ORDER"] == "FASTEST_FIRST" probe.assert_not_called() @@ -1079,6 +1085,7 @@ def test_gpu_worker_advertises_native_capacity_with_the_native_mask( monkeypatch.setenv(key, value) monkeypatch.setenv("SLURM_GPUS_ON_NODE", "2") monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "1,3") + monkeypatch.setenv("CUDA_DEVICE_ORDER", "FASTEST_FIRST") connection = Connection( namespace=NAMESPACE, provider="slurm", launch={"connection_root": args.connection_root}, ) @@ -1101,6 +1108,7 @@ def test_gpu_worker_advertises_native_capacity_with_the_native_mask( "CPU": 1, "MEMORY": args.memory_bytes, "GPU": 2, } assert os.environ["CUDA_VISIBLE_DEVICES"] == "1,3" + assert os.environ["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" def test_worker_rendezvous_has_a_finite_deadline( @@ -1166,7 +1174,8 @@ def test_standard_bootstrap_starts_scheduler_and_worker_on_rank_zero_and_worker_ ) argv = [sys.executable, "-P", "-m", "lightcone.engine.compute.slurm_bootstrap"] for key, value in vars(args).items(): - if value is not None: + # CPU-only submissions can omit the optional GPU count. + if value is not None and key != "gpus": argv += ["--" + key.replace("_", "-"), str(value)] connection = Connection( namespace=NAMESPACE, provider="slurm", launch={"connection_root": args.connection_root} diff --git a/tests/test_container.py b/tests/test_container.py index dc7a8f6a..fd28ed6a 100644 --- a/tests/test_container.py +++ b/tests/test_container.py @@ -193,6 +193,28 @@ def test_a_missing_archive_refuses_unless_the_caller_may_build( assert container.runtime_for_run(root, build=False).image_id == runtime.image_id +@pytest.mark.parametrize("runtime", ["docker", "podman"]) +@pytest.mark.parametrize("build", [False, True]) +def test_unsupported_gpu_runtime_refuses_before_any_image_preparation( + root: Path, fake: list[list[str]], monkeypatch: pytest.MonkeyPatch, + runtime: str, build: bool, +) -> None: + monkeypatch.setattr(container, "runtime_name", lambda _: runtime) + with pytest.raises(ProjectError, match="GPU containers require podman-hpc"): + container.runtime_for_run(root, build=build, use_gpus=True) + assert fake == [] + assert not (root / ".datalad").exists() + + +def test_supported_gpu_runtime_can_prepare_the_image_without_a_driver_gpu_mask( + root: Path, hpc: list[list[str]], monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + runtime = container.runtime_for_run(root, build=True, use_gpus=True) + assert runtime.supports_gpus + assert _argvs(hpc, "podman-hpc", "build") + + def test_unfetched_archive_content_is_fetched_by_lc_itself( root: Path, fake: list[list[str]], monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/tests/test_execution_resources.py b/tests/test_execution_resources.py index c81ec0b4..3bc1feeb 100644 --- a/tests/test_execution_resources.py +++ b/tests/test_execution_resources.py @@ -7,7 +7,7 @@ import pytest from pydantic import ValidationError -from lightcone.engine.execution_resources import TaskResources +from lightcone.engine.execution_resources import TaskResources, worker_capacities from lightcone.engine.project import ProjectError GIB = 1024**3 @@ -75,17 +75,19 @@ def test_recipe_memory_uses_exact_bytes_without_decimal_context_rounding() -> No def test_declared_requests_reserve_exact_cpu_and_memory_budgets() -> None: task = TaskResources.parse({"cpus": 4, "memory": "6Gi"}) - assert task.requirements(_workers((8, 16 * GIB))) == {"CPU": 4, "MEMORY": 6 * GIB} + assert task.requirements(worker_capacities(_workers((8, 16 * GIB)))) == { + "CPU": 4, "MEMORY": 6 * GIB, + } -def test_missing_memory_reserves_entire_worker_instead_of_guessing() -> None: - assert TaskResources().requirements(_workers((8, 16 * GIB), (8, 16 * GIB))) == { - "CPU": 1, "MEMORY": 16 * GIB, - } +def test_missing_memory_does_not_reserve_a_budget() -> None: + capacities = worker_capacities(_workers((8, 16 * GIB), (8, 16 * GIB))) + assert TaskResources().requirements(capacities) == {"CPU": 1} def test_probe_reserves_an_entire_worker() -> None: - assert TaskResources().requirements(_workers((8, 16 * GIB)), whole_worker=True) == { + capacities = worker_capacities(_workers((8, 16 * GIB))) + assert TaskResources().requirements(capacities, whole_worker=True) == { "CPU": 8, "MEMORY": 16 * GIB, } @@ -93,34 +95,33 @@ def test_probe_reserves_an_entire_worker() -> None: def test_cpu_count_is_independent_of_dask_execution_threads() -> None: workers = _workers((8, 16 * GIB)) workers["worker-0"]["nthreads"] = 1 - assert TaskResources(cpus=8).requirements(workers)["CPU"] == 8 + assert TaskResources(cpus=8).requirements(worker_capacities(workers))["CPU"] == 8 def test_a_task_must_fit_one_worker_not_the_sum_of_the_cluster() -> None: with pytest.raises(ProjectError, match="on one worker"): TaskResources(cpus=8, memory_bytes=20 * GIB).requirements( - _workers((4, 16 * GIB), (4, 16 * GIB)) + worker_capacities(_workers((4, 16 * GIB), (4, 16 * GIB))) ) def test_cpu_and_memory_must_fit_on_the_same_worker() -> None: with pytest.raises(ProjectError, match="no worker"): TaskResources(cpus=8, memory_bytes=16 * GIB).requirements( - _workers((8, 4 * GIB), (4, 16 * GIB)) + worker_capacities(_workers((8, 4 * GIB), (4, 16 * GIB))) ) def test_explicit_requests_can_select_a_fitting_worker() -> None: assert TaskResources(cpus=8, memory_bytes=8 * GIB).requirements( - _workers((4, 4 * GIB), (8, 16 * GIB)) + worker_capacities(_workers((4, 4 * GIB), (8, 16 * GIB))) ) == {"CPU": 8, "MEMORY": 8 * GIB} -@pytest.mark.parametrize("whole_worker", [False, True]) -def test_missing_budgets_are_not_guessed_for_heterogeneous_workers(whole_worker: bool) -> None: +def test_probes_require_identical_worker_budgets() -> None: with pytest.raises(ProjectError, match="identical"): TaskResources().requirements( - _workers((4, 4 * GIB), (8, 16 * GIB)), whole_worker=whole_worker + worker_capacities(_workers((4, 4 * GIB), (8, 16 * GIB))), whole_worker=True ) @@ -132,14 +133,14 @@ def test_missing_budgets_are_not_guessed_for_heterogeneous_workers(whole_worker: ) def test_absent_or_unknown_worker_capacity_refuses_execution(workers: dict[str, Any]) -> None: with pytest.raises(ProjectError): - TaskResources().requirements(workers) + TaskResources().requirements(worker_capacities(workers)) def test_gpu_recipe_reserves_the_whole_worker_gpu_budget() -> None: workers = _workers((8, 16 * GIB)) workers["worker-0"]["resources"]["GPU"] = 4 task = TaskResources.parse({"cpus": 2, "memory": "4Gi", "gpus": 1}) - assert task.requirements(workers) == {"CPU": 2, "MEMORY": 4 * GIB, "GPU": 4} + assert task.requirements(worker_capacities(workers)) == {"CPU": 2, "MEMORY": 4 * GIB, "GPU": 4} def test_gpu_request_must_fit_one_worker() -> None: @@ -147,24 +148,24 @@ def test_gpu_request_must_fit_one_worker() -> None: for worker in workers.values(): worker["resources"]["GPU"] = 1 with pytest.raises(ProjectError, match="2 GPUs on one worker"): - TaskResources(gpus=2).requirements(workers) + TaskResources(gpus=2).requirements(worker_capacities(workers)) def test_cpu_workers_cannot_satisfy_gpu_requests() -> None: with pytest.raises(ProjectError, match="1 GPUs on one worker"): - TaskResources(gpus=1).requirements(_workers((8, 16 * GIB))) + TaskResources(gpus=1).requirements(worker_capacities(_workers((8, 16 * GIB)))) def test_cpu_recipe_does_not_reserve_gpu_capacity() -> None: workers = _workers((8, 16 * GIB), (8, 16 * GIB)) workers["worker-0"]["resources"]["GPU"] = 4 - assert TaskResources().requirements(workers) == {"CPU": 1, "MEMORY": 16 * GIB} + assert TaskResources().requirements(worker_capacities(workers)) == {"CPU": 1} def test_probe_reserves_cpu_memory_and_all_gpus() -> None: workers = _workers((8, 16 * GIB)) workers["worker-0"]["resources"]["GPU"] = 4 - assert TaskResources().requirements(workers, whole_worker=True) == { + assert TaskResources().requirements(worker_capacities(workers), whole_worker=True) == { "CPU": 8, "MEMORY": 16 * GIB, "GPU": 4, } @@ -173,7 +174,8 @@ def test_gpu_requirements_ignore_workers_that_cannot_fit_the_recipe() -> None: workers = _workers((2, 2 * GIB), (8, 16 * GIB)) workers["worker-0"]["resources"]["GPU"] = 1 workers["worker-1"]["resources"]["GPU"] = 4 - assert TaskResources(cpus=4, memory_bytes=8 * GIB, gpus=1).requirements(workers) == { + task = TaskResources(cpus=4, memory_bytes=8 * GIB, gpus=1) + assert task.requirements(worker_capacities(workers)) == { "CPU": 4, "MEMORY": 8 * GIB, "GPU": 4, } @@ -184,7 +186,7 @@ def test_whole_gpu_reservations_refuse_ambiguous_worker_budgets(whole_worker: bo workers["worker-0"]["resources"]["GPU"] = 1 workers["worker-1"]["resources"]["GPU"] = 4 with pytest.raises(ProjectError, match="identical GPU budgets"): - TaskResources(gpus=1).requirements(workers, whole_worker=whole_worker) + TaskResources(gpus=1).requirements(worker_capacities(workers), whole_worker=whole_worker) @pytest.mark.parametrize("capacity", [-1, 0.5, True, "1", None, float("nan"), float("inf")]) @@ -192,4 +194,10 @@ def test_malformed_gpu_capacity_is_never_ignored(capacity: object) -> None: workers = _workers((8, 16 * GIB)) workers["worker-0"]["resources"]["GPU"] = capacity with pytest.raises(ProjectError, match="GPU count"): - TaskResources().requirements(workers) + TaskResources().requirements(worker_capacities(workers)) + + +def test_undeclared_memory_accepts_heterogeneous_workers() -> None: + assert TaskResources(cpus=4).requirements( + worker_capacities(_workers((4, 4 * GIB), (8, 16 * GIB))) + ) == {"CPU": 4} diff --git a/tests/test_gpu_execution.py b/tests/test_gpu_execution.py index 49dfc2ac..4ecf70b3 100644 --- a/tests/test_gpu_execution.py +++ b/tests/test_gpu_execution.py @@ -3,9 +3,11 @@ from __future__ import annotations import os -from collections.abc import Callable +from collections.abc import Callable, Iterator +from contextlib import contextmanager from dataclasses import replace from pathlib import Path +from typing import Any from unittest.mock import Mock import pytest @@ -68,7 +70,7 @@ def test_missing_worker_gpu_mask_preserves_existing_output_and_manifest( monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) if mask is not None: monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) - with pytest.raises(ProjectError, match="nonempty allocation CUDA_VISIBLE_DEVICES"): + with pytest.raises(ProjectError, match="nonempty CUDA_VISIBLE_DEVICES"): worker.execute(root, replace(task, resources={"gpus": 1}), {}, context) assert task.output_path.read_text() == "previous output" assert task.manifest_path.read_text() == "previous manifest" @@ -119,6 +121,75 @@ def test_probe_missing_gpu_mask_refuses_before_running_the_command( monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) execute = Mock() monkeypatch.setattr(engine_run.sandbox, "run", execute) - with pytest.raises(ProjectError, match="nonempty allocation CUDA_VISIBLE_DEVICES"): + with pytest.raises(ProjectError, match="nonempty CUDA_VISIBLE_DEVICES"): engine_run._probe(context.runtime, [], ("true",), True, output=lambda *_: None) execute.assert_not_called() + + +@pytest.mark.parametrize("runtime_name", ["docker", "podman", "podman-hpc"]) +def test_container_probe_on_gpu_cluster_exposes_only_supported_gpus_and_reports_cpu_mode( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + runtime_name: str, +) -> None: + from distributed import Client, LocalCluster + + from lightcone.engine import compute, sandbox + + root, _, context = project + runtime = replace(context.runtime, mode="containerized", runtime=runtime_name, image_id="test") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + monkeypatch.setattr(container, "runtime_for_run", lambda *args, **kwargs: runtime) + monkeypatch.setattr(container, "converge", lambda _: []) + commands = [] + reservations = [] + + def execute(backend: Any, policy: Any, argv: Any, **kwargs: Any) -> sandbox.Outcome: + commands.append(backend.wrap(policy, argv)) + return sandbox.Outcome(0, backend.attest(policy)) + + submit = Client.submit + + def record(client: Any, *args: Any, **kwargs: Any) -> Any: + reservations.append(kwargs["resources"]) + return submit(client, *args, **kwargs) + + @contextmanager + def connect(cluster_id: str) -> Iterator[Client]: + with LocalCluster( + n_workers=1, threads_per_worker=1, processes=False, dashboard_address=None, + resources={"CPU": 2, "MEMORY": 1024**3, "GPU": 2}, + ) as cluster, Client(cluster) as client: + yield client + + monkeypatch.setattr(sandbox, "run", execute) + monkeypatch.setattr(compute, "connect", connect) + monkeypatch.setattr(Client, "submit", record) + outcome = engine_run.probe(root, ["true"], cluster_id="gpu-cluster") + assert outcome.returncode == 0 + assert reservations == [{"CPU": 2, "MEMORY": 1024**3, "GPU": 2}] + assert len(commands) == 1 + if runtime_name == "podman-hpc": + assert "--gpu" in commands[0] + assert "--env=CUDA_VISIBLE_DEVICES=2,0" in commands[0] + assert not any("without GPUs" in note for note in outcome.notes) + else: + assert "--env=CUDA_VISIBLE_DEVICES=" in commands[0] + assert "--env=NVIDIA_VISIBLE_DEVICES=void" in commands[0] + assert "--gpu" not in commands[0] + assert any("this probe runs without GPUs" in note for note in outcome.notes) + + +def test_gpu_rerun_missing_mask_explains_how_to_select_local_devices( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture[str], +) -> None: + root, _, _ = project + (root / "astra.yaml").write_text(_SPEC.replace( + "command:", "resources: {gpus: 1}\n command:", + )) + monkeypatch.chdir(root) + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + assert worker.main(["baseline/visibility"]) == 2 + error = capsys.readouterr().err + assert "CUDA_VISIBLE_DEVICES=0 datalad rerun" in error + assert "Slurm sets it" in error diff --git a/tests/test_materialize.py b/tests/test_materialize.py index ef98b42f..44b323af 100644 --- a/tests/test_materialize.py +++ b/tests/test_materialize.py @@ -1127,6 +1127,30 @@ def unexpected(*args: object, **kwargs: object) -> None: assert not dataset.status(root) + +@pytest.mark.parametrize("runtime_name", ["docker", "podman"]) +def test_gpu_runtime_refusal_precedes_image_build_and_all_recipes( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, runtime_name: str, +) -> None: + spec = _SPEC.replace("command: cat", "resources: {gpus: 1}\n command: cat") + root = analysis(spec, universes={"baseline": _UNIVERSE}) + _resource_cluster(monkeypatch, gpus=1) + pyproject = root / "pyproject.toml" + pyproject.write_text(pyproject.read_text() + "\n[tool.lightcone.image]\napt-install = []\n") + dataset.save(root, [pyproject], "declare a container image") + monkeypatch.setattr(engine.container, "runtime_name", lambda _: runtime_name) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("image preparation began for an unsupported GPU runtime") + + monkeypatch.setattr(engine.container.image, "tag", unexpected) + before = dataset.head(root) + with pytest.raises(ProjectError, match="GPU containers require podman-hpc"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert dataset.head(root) == before + assert not dataset.status(root) + assert not (root / "results/baseline/first.txt").exists() + def test_empty_cluster_refuses_before_project_preparation( root: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -1144,6 +1168,7 @@ def unexpected(*args: object, **kwargs: object) -> None: @pytest.mark.parametrize( ("resource_spec", "gpus", "expected_parallelism"), [ + ("", 0, 4), ("cpus: 3, memory: 256Mi", 0, 1), ("cpus: 1, memory: 1Gi", 0, 2), ("cpus: 1, memory: 256Mi, gpus: 1", 2, 1), @@ -1195,6 +1220,42 @@ def test_real_dask_respects_recipe_resource_reservations( assert not dataset.status(root) + +def test_current_gpu_output_needs_no_gpu_to_build_a_cpu_dependent( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, +) -> None: + spec = _SPEC.replace("command: echo", "resources: {gpus: 1}\n command: echo") + root = analysis(spec, universes={"baseline": _UNIVERSE}) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") + _resource_cluster(monkeypatch, gpus=1) + assert engine.materialize(root, ["first"], cluster_id=CLUSTER_ID).made == ["baseline/first"] + original = (root / "results/baseline/.first.manifest.json").read_bytes() + + _resource_cluster(monkeypatch) + report = engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert report.ok + assert report.current == ["baseline/first"] + assert report.made == ["baseline/second"] + assert (root / "results/baseline/.first.manifest.json").read_bytes() == original + assert not dataset.status(root) + + +def test_current_outputs_are_not_submitted_to_dask( + root: Path, inline: None, monkeypatch: pytest.MonkeyPatch, +) -> None: + engine.materialize(root, [], cluster_id=CLUSTER_ID) + + class NoExecution(_Inline): + def validate(self, tasks: Any) -> dict[Any, Any]: + assert not list(tasks) + return {} + + def submit(self, *args: Any, **kwargs: Any) -> Any: + pytest.fail("a current output was submitted to Dask") + + _cluster(monkeypatch, NoExecution()) + assert len(engine.materialize(root, [], cluster_id=CLUSTER_ID).current) == 2 + def test_a_real_cluster_still_fits_through_the_seam(root: Path, cluster_id: str) -> None: """The one test that starts Dask. The seam is only worth having if the thing it abstracts still goes through it.""" diff --git a/tests/test_sandbox_policy.py b/tests/test_sandbox_policy.py index 3d926077..66978d7b 100644 --- a/tests/test_sandbox_policy.py +++ b/tests/test_sandbox_policy.py @@ -268,7 +268,7 @@ def test_gpu_policy_refuses_a_missing_mask_before_creating_private_state( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) - with pytest.raises(ProjectError, match="nonempty allocation CUDA_VISIBLE_DEVICES"): + with pytest.raises(ProjectError, match="nonempty CUDA_VISIBLE_DEVICES"): policy_module.exec_policy(tmp_path, containerized=True, use_gpus=True) assert not (tmp_path / ".lightcone").exists()