diff --git a/CLAUDE.md b/CLAUDE.md index 961880e6..ee4aaced 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -952,10 +952,10 @@ none; formats may, `tar.gz`), never `Path.stem`. The old `_HASH_EXCLUDE` is gone with the directory that made it necessary: the manifest cannot be inside the thing it describes any more. -**Dask owns the ordering.** Every task is submitted with its upstream -futures as arguments, so the dependency order, the parallelism, and the -scheduling all fall out of the argument graph. There is no ready-set loop -and no hand-rolled topological sort in the execution path. +**Dask owns the ordering.** Tasks that may execute receive upstream futures +or already-current `TaskResult` values as arguments, so dependency order, +parallelism, and scheduling fall out of the argument graph. There is no ready-set +loop and no hand-rolled topological sort in the execution path. `Graph.order()` exists for the read-only walk, which has to classify a task after everything upstream of it — and for submitting in an order where a task's upstream handles already exist. @@ -1165,7 +1165,7 @@ its sync touches only ignored paths — so both modes run one order.) **A run fetches its declared inputs; the read-only verbs never do.** `materialize` batch-runs `git annex get` over the graph's in-tree -declared inputs before anything hashes (driver-side — the storage +declared inputs before workers hash or execute (driver-side — the storage invariant that nobody is ever asked to run an annex command by hand), so a bytes-free clone materializes straight to up-to-date. A failed fetch is a *warning*, never a refusal: independent tasks still run and @@ -1727,6 +1727,10 @@ and use it for every native ownership check and filter. **Local compute needs no setup.** An absent implicit `~/.lightcone/compute.yaml` selects a built-in local catalog: one CPU, 1 GiB, one node, fast startup, 30-minute default and two-hour maximum lifetime. It writes no catalog and starts no cluster. +GPU offers require an explicit catalog and, for local launches, a nonempty +`CUDA_VISIBLE_DEVICES` mask on Linux. No GPU auto-discovery. Local GPU capacity and +model labels are configured, not hardware-verified; allocations do not reserve +devices exclusively against other host programs or allocations. Configured catalogs replace it completely; missing explicit paths and invalid files are errors. Execution still requires an explicitly launched cluster's name or ID. @@ -1753,6 +1757,17 @@ Connection names live only in the catalog's mapping keys. Use validated `replace for updates and the explicit `as_dict` allowlists for public output. Preserve duplicate-key rejection in YAML; providers validate their own `launch` and `config`. +**Allocation syntax follows SkyPilot without depending on SkyPilot.** Compute +CPU/memory requests support exact quantities or `+` minimums. Compute memory +uses binary units: bare `32`, `32GB`, and `32GiB` agree. Catalog resources use +one `accelerators: NAME[:COUNT]` or a one-entry mapping; CLI `--gpus A100:4`, +`A100`, or generic `GPU:4` selects an exact positive whole count, while `0` +means CPU only. Type matching is case-insensitive; no GPU `+`, fractions, or +global model alias registry. Local accelerator labels are trusted configuration. +Named Slurm offers must map their public label to the site's GRES type through +`config.gpu_type`; generic `GPU` offers may omit that setting. Preserve native +evidence in observations. + **Configured compute roots may be filesystem aliases.** Resolve connection and scratch roots before appending managed namespace, submission, or attempt paths. Keep symlink rejection within those managed paths and enforce private directory @@ -1779,6 +1794,47 @@ Any driver failure while tasks are outstanding (a failed commit included, not only a cluster error) carries `compute.UNSTOPPED`, the one wording for "the allocation was not stopped and unreported tasks may still be running". +**Recipe resources use standard Dask admission.** Preserve ASTRA `recipe.resources` +in `plan.Task` as raw mappings so `status` and `--check` remain independent of +executor support. Parse `TaskResources` at execution admission: whole CPUs, memory +bytes, and whole GPU counts. Reuse the read-only classification walk before +admission: known current/behind outputs become values without Dask submission. +Validate tasks that may execute, including dependents of potentially rebuilt +outputs; workers recheck actual upstream digests. Normalize worker budgets once +with `worker_capacities`, then pass reservations explicitly to submission. Recipe +`time_limit` is unsupported and must fail explicitly; allocation walltime remains supported. +Workers advertise CPU/MEMORY/GPU; tasks reserve their declarations. Omitted RAM +adds no memory reservation; CPU requests and task slots govern concurrency. +Probes reserve all whole-worker budgets. +GPU recipes reserve the worker's full GPU budget, one GPU recipe at a time, and +inherit its whole allocation mask. Recipe `gpus` is a minimum capacity requirement, +not a visibility limit; it defaults to zero and does not select a model. Recipe +memory retains ASTRA units (`8Gi` binary, `8GB` decimal, no bare quantities), +independently of compute's SkyPilot units. +Thread slots remain a separate concurrency cap. Reservations are cooperative, not +per-command OS CPU/RAM limits or BLAS thread counts. Local Nanny defaults keep +OMP/MKL/OPENBLAS threads at one unless the launch environment overrides them; Slurm +uses direct workers and the job environment. Unsupported disk/model requests and +fractional CPU/GPU counts fail explicitly. Exact bytes are shared in `units.py`; +allocation durations are parsed in `compute.model`, with error conversion only +at the CLI request boundary. + +**GPU visibility comes from the allocation, not device discovery.** No CUDA probe, +UUID inventory, MIG detection, model verification, or custom Dask worker. Local +launch freezes the externally supplied nonempty CUDA mask and optional device +order. Slurm validates native GPU counts, preserves its mask, and sets +`CUDA_DEVICE_ORDER=PCI_BUS_ID`. `exec_policy(use_gpus=True)` inherits that whole +mask; CPU commands get an empty one. Never mutate the reusable worker's environment. +Direct GPU policies grant native NVIDIA character nodes; OS permissions and cgroups +remain authoritative. The host must initialize NVIDIA character devices including +UVM before launch; lc neither loads drivers nor creates nodes. Standalone GPU +reruns need an explicit CUDA mask in their own environment. Container GPU execution +supports podman-hpc `--gpu` only. Explicit GPU recipes on ordinary Docker/Podman +fail before image preparation; probes use CPU policy with a diagnostic note while +retaining their whole-worker reservation. CPU containers remain supported on all +runtimes and set `NVIDIA_VISIBLE_DEVICES=void`. Physical GPU execution remains +unvalidated; tests check real subprocess masks and native argv without GPU hardware. + **One catalog selector, `LC_COMPUTE_CONFIG` (2026-09).** `lc compute --config` was removed: `run` and `materialize` resolve clusters through the catalog too, and a per-invocation override on one command group launched allocations those diff --git a/docs/api/compute.md b/docs/api/compute.md index c6fa9429..b780bd2c 100644 --- a/docs/api/compute.md +++ b/docs/api/compute.md @@ -6,7 +6,7 @@ It owns no service, registry, or saved current-cluster selection. | Symbol | Contract | |---|---| -| `Request.parse(...)` | Common exact/minimum CPU and memory requests, node count, walltime, startup class. | +| `Request.parse(...)` | Exact/minimum CPU and memory requests, exact accelerator type/count, node count, walltime, startup class. | | `Catalog.load(path)` | Ordered fixed shapes and stable connection namespaces; use the built-in local catalog only when the implicit default file is absent. | | `Compute.plan(request, *, name=None)` | Select an eligible offer and freeze its native launch settings and optional name without allocation. | | `Compute.launch(plan)` | Check names across native authorities, generate one if omitted, submit once, and return a self-contained `Identity`. | @@ -17,14 +17,15 @@ It owns no service, registry, or saved current-cluster selection. | `Provider` | `plan`, `launch`, `discover`, `inspect`, `connect`, `terminate`. | `Catalog.load()` defaults to `~/.lightcone/compute.yaml`. When that implicit file -is absent, the built-in catalog exposes one `local` offer: one CPU, 1 GiB, one node, -fast startup, 30-minute default and two-hour maximum lifetime. It creates no -configuration file or allocation. Configured catalogs replace it completely. +is absent, the built-in catalog exposes a `local` offer: one CPU, 1 GiB, one node, +fast startup, 30-minute default and two-hour maximum lifetime. GPU offers require +an explicit catalog. It creates no configuration file or allocation. Configured +catalogs replace it completely. Missing paths selected through an argument or `LC_COMPUTE_CONFIG`, unreadable files, and invalid catalogs remain errors. Stable connection namespaces let separate invocations discover and attach to the same local allocations. -`model.py` defines the shared Pydantic models: `Connection`, `Offer`, `Resources`, +`model.py` defines the shared Pydantic models: `Connection`, `Offer`, `Resources`, `Accelerator`, `TimeLimits`, `Startup`, `Request`, `Identity`, `LaunchPlan`, and `Snapshot`. `Catalog` validates YAML directly into these objects, which providers also use. Unknown common fields are rejected; schema errors identify paths such as @@ -39,6 +40,14 @@ keeps the configured `default` and `max` duration strings and exposes `default_seconds` and `max_seconds`. `Startup.class_` corresponds to YAML `class`. Connection names exist only as catalog mapping keys, referenced by `Offer.connection`. +Compute memory accepts bare GiB quantities and SkyPilot-style binary units: +`32`, `32GB`, and `32GiB` agree. CPU and memory requests accept a trailing `+`. +`Accelerator` accepts one `NAME[:COUNT]` or one-entry mapping, such as `A100:4` +or `{A100: 4}`, and serializes to that mapping. Counts are exact positive integers; +type matching is case-insensitive, and the generic name `GPU` accepts any model. +No accelerator registry or model alias expansion is maintained. Slurm's +`config.gpu_type` maps a named catalog accelerator to its native GRES type. + Model constructors take keyword arguments. `replace(...)` validates updates; `model_dump()` and `model_validate()` support internal roundtrips without changing units. Keep the explicit `as_dict()` methods for public CLI output so internal @@ -113,6 +122,64 @@ Dask chooses the workers and handles dependencies; invocation-specific keys prev unintended reuse across commands. There is no worker-selection layer, per-worker preflight orchestration, source fingerprinting, or login-node guard. Driver-side preparation and the existing task runtime/sandbox checks remain in their owners. + +Workers advertise standard Dask `CPU`, `MEMORY`, and `GPU` resources; memory is measured +in bytes. `engine.execution_resources.TaskResources` validates ASTRA's +`recipe.resources` into whole CPUs, bytes, and a whole GPU count at +execution admission. `plan.Task` preserves the ASTRA mapping so read-only +classification does not impose executor restrictions. `worker_capacities(workers)` +normalizes advertised budgets once; `requirements(capacities)` checks that one +worker can satisfy a task and returns its `Client.submit` resource dictionary. +Omitted memory adds no `MEMORY` reservation. `whole_worker=True` reserves CPU, +memory, and GPUs for a probe. Recipe GPU counts default to zero; a GPU recipe +reserves the full GPU budget of a fitting worker +and inherits its whole allocation mask. The requested count is a minimum, not a +visibility limit. This serializes GPU recipes per worker without device assignment. +Unsupported disk/type requests and fractional CPU/GPU counts fail before execution. + +The driver reuses the read-only classification walk before admission. Known +current or unrefreshed behind outputs become `TaskResult` values, without Dask +submission or resource reservations. Tasks that may execute, including dependents +of potentially rebuilt outputs, have their resource requests validated before +preparation. Workers recheck actual upstream digests and may still skip a reserved +task if its inputs prove unchanged. Allocation and task requests share byte +conversion utilities; their models remain distinct because allocation selection supports minimum quantities +and node counts. Standard Dask scheduling accounts for +concurrent CPU, memory, and GPU reservations; Dask execution-thread counts remain a +separate concurrency cap. Reservations do not impose hard limits on recipe +subprocesses. Recipe `time_limit` is unsupported and explicitly refused before +preparation or execution; allocation walltime remains supported. + +Recipe memory remains ASTRA-style: `8Gi` is binary, `8GB` is decimal, and units +are required. Allocation memory follows the compute convention above; keep the +two parsers' contracts explicit even though they share exact byte arithmetic in +`units.py`. Allocation duration parsing stays in `compute.model.duration`, raising +`ValueError` for Pydantic; `Request.parse` converts it to `ComputeError`. + +## GPU allocation and visibility + +Local GPU offers require Linux and an explicit nonempty `CUDA_VISIBLE_DEVICES`. +Planning freezes that mask and optional `CUDA_DEVICE_ORDER`; launch passes them +to the worker unchanged. Count and model are catalog declarations, not hardware +observations. The built-in local offer remains CPU-only. + +Slurm requests native GPU GRES and validates `SLURM_GPUS_ON_NODE` before +advertising the worker's GPU budget. Bootstrap preserves Slurm's CUDA mask and +sets `CUDA_DEVICE_ORDER=PCI_BUS_ID`. There is no CUDA probe, device inventory, or +custom Dask worker. + +The sandbox's `use_gpus` policy option inherits the worker's mask for GPU commands +and supplies an empty mask for CPU commands, without modifying the reusable +worker's environment. Direct GPU policies grant native NVIDIA character devices. +Container GPU execution uses podman-hpc's `--gpu`. Explicit GPU recipes on ordinary +Docker or Podman are refused before image preparation; probes use a CPU policy and +report that GPU access is unavailable while retaining their whole-worker reservation. +Native permissions and cgroups remain authoritative. NVIDIA devices, including UVM, +must already exist; policy construction does not load drivers or create devices. +See [GPU deployment requirements](../user/cluster.md#gpu-allocations). + +## Execution output and teardown + `output.py` transports byte chunks through standard Dask events so detached workers' output reaches the invoking CLI. It uses the borrowed client's event topic, which the schedulers lc launches drop as soon as the client disconnects diff --git a/docs/api/container.md b/docs/api/container.md index bd360f6c..0bb0f553 100644 --- a/docs/api/container.md +++ b/docs/api/container.md @@ -18,10 +18,10 @@ Sources: `src/lightcone/engine/image.py`, | `image.tag(root)` | `lc-env-<16 hex>` over the rendered Containerfile *and* the identity document. | | `image.archive_path(root, tag)` | `.datalad/environments//image` — the `datalad containers-add` layout. | | `container.build(root)` | Build + save + commit, idempotent; returns `(Runtime, "built" \| "present")`. | -| `container.runtime_for_run(root, *, build)` | One function, two strictnesses: `lc build`/materialize-preflight may build and commit; the probe and worker only ever find, fetch, and load. | +| `container.runtime_for_run(root, *, build, use_gpus=False)` | Resolve the runtime, refusing unsupported explicit GPU requests before preparing the image. Materialize may build and commit; probes and reruns only find, fetch, and load. | | `container.backend(...)` | The single construction point for the exec backend — the only mode branch. | | `container.sync(...)` | The in-container environment converge: network on, project `:rw`, host uv cache mounted, into `.lightcone/venv`. | -| `Runtime` | Facts only — root/mode/name/tag/id/arch — never mechanism. | +| `Runtime` | Resolved execution facts; `supports_gpus` is true for direct mode and podman-hpc. | ## What must stay true diff --git a/docs/api/index.md b/docs/api/index.md index 8674ea0b..d79ee66b 100644 --- a/docs/api/index.md +++ b/docs/api/index.md @@ -18,6 +18,7 @@ is responsibility and contract, not every signature. | [`worker`](worker.md) | Making one output; the rerun entry point | impure | | [`materialize`](materialize.md) | The driver: gates, scheduling, the save/restore loop, status | impure | | [`compute`](compute.md) | Resource requests, native allocation lifecycle, borrowed Dask clients | impure | +| [`execution_resources`](compute.md) | Task resource admission | pure | | [`sandbox`](sandbox.md) | The exec boundary: policy, backends, attestation, denials | mixed | | [`image` & `container`](container.md) | The container hatch: declaration → image → archive → runtime | pure / impure | | [`crate`](crate.md) | The publication view: the repo as an RO-Crate | pure | diff --git a/docs/api/materialize.md b/docs/api/materialize.md index 39450bcb..d767404b 100644 --- a/docs/api/materialize.md +++ b/docs/api/materialize.md @@ -18,16 +18,20 @@ driver's stderr, independently of success or failure, leaving stdout for the rep | `check(root, targets, *, refresh)` | The same classification without executing, committing, or fetching. Exempt from the dirty refusal. | | `status(root)` | The report: every output's state and provenance commit, plus the mode/image/sandbox header facts. | | `MaterializeReport` / `StatusReport` | The JSON surfaces; `ok` and `up_to_date` first. | -| `cluster_for_run(cluster_id)` | Borrow the cluster; the submit/completed scheduler seam (`submit`, `completed`). | +| `cluster_for_run(cluster_id)` | Borrow the cluster; expose resource validation, submission, and completion. | | `run_record(...)` / `datalad_run_subject(...)` | The commit message `datalad rerun` replays, and the one spelling of its subject line — shared with the foreign-write comparator, because two strings here would drift. | | `_engine_requirement()` | How a record pins its engine: by version for a release, by source commit (hatch-vcs) for a dev build. | ## The run's order, and why 1. **Read-only project checks before connecting** — tool, committer, dirty-tree, - spec and lock errors do not require a reachable cluster to report. + spec and lock errors do not require a reachable cluster to report. The shared + classification walk identifies outputs already current or left behind. 2. **Explicit cluster before preparing the environment** — validate native - allocation identity and connect before fetching inputs or building an image. + allocation identity, connect, and validate CPU/memory/GPU requests for tasks + that may execute. Known skips become values without resource reservations; + dependents of potentially rebuilt outputs still need admission. Explicit GPU + recipes must also have a supported runtime before any image build. The dirty refusal has already run: in containerized mode the converge can commit an image archive, and `dataset.save` commits the whole index; on a dirty tree the user's diff --git a/docs/api/plan.md b/docs/api/plan.md index 93cf8e35..49f06a78 100644 --- a/docs/api/plan.md +++ b/docs/api/plan.md @@ -3,8 +3,8 @@ The spec, read as a graph of tasks. `astra.yaml` × `universes/*.yaml` gives one task per `(universe, output)` pair that has a recipe; a task carries everything executing it needs — the rendered command, where its -bytes go, what it reads, its decisions, its `definition_version` — and -nothing about *how* it will be executed. +bytes go, what it reads, its decisions, its `definition_version`, and its +resource requirements — and nothing about *how* it will be executed. Source: `src/lightcone/engine/plan.py`. @@ -14,7 +14,7 @@ Source: `src/lightcone/engine/plan.py`. |---|---| | `build(root)` | Validate the spec with ASTRA's own validators, resolve every universe, return the `Graph`. | | `Graph` | Tasks keyed on `(universe_id, output_id)`; `order()` for the read-only topological walk, `resolve(targets)` for what a user typed, `closure(keys)` to narrow a run. | -| `Task` | One output in one universe, frozen. | +| `Task` | One output in one universe, frozen, retaining ASTRA's resource declaration in `resources`. | | `declared_path(root, path)` | The one rule that names a path: project-relative inside the tree, absolute outside, never resolved. | ## What must stay true @@ -31,6 +31,14 @@ Source: `src/lightcone/engine/plan.py`. schema, file, and universe validators before resolving anything — resolution answers what a *valid* spec means and does not re-check that it is one. +- **Resource declarations survive resolution.** `build` reads + `recipe.resources` from ASTRA's resolved output definition and preserves the + mapping. A valid declaration remains readable by `status` and + `materialize --check` even when this executor cannot honor it. Execution + validates supported requirements through `TaskResources.parse` and checks + cluster capacity for tasks that may execute before preparing the project. + Already-current outputs need no resource admission. No worker placement or executor-specific resource validation belongs + in this module. - **The layout is flat and path-addressed.** `results//.`, and the path in a rendered recipe *is* the path on disk — no staging, no relocation. @@ -50,8 +58,8 @@ Source: `src/lightcone/engine/plan.py`. ## Tests `tests/test_plan.py` — pure; tests what lc *adds* (directories, edges, -versions, the validation gate), never what a spec means — that -coverage lives in astra-tools' own suite, and re-asserting it here +versions, resource preservation, the validation gate), never what a spec means — +that coverage lives in astra-tools' own suite, and re-asserting it here would recreate the second implementation this module deleted. Every fixture must be a spec `astra validate` accepts; the gate enforces it for free. diff --git a/docs/api/sandbox.md b/docs/api/sandbox.md index afd87882..813a1ac5 100644 --- a/docs/api/sandbox.md +++ b/docs/api/sandbox.md @@ -18,7 +18,7 @@ plus `lightcone/_sandbox_exec.py`, the Landlock shim. | `Capability` | What this host can do — `detect()`'s answer, the only `sys.platform` branch. | | `Attestation` | What was actually enforced, derived from the flags applied — never from what the matrix says should have happened. | | `Backend.wrap(policy, argv)` | The pure rewrite. `contains_prefix` declares whether the uv hop rides inside (a container is a world; a host mechanism trusts host plumbing). | -| `exec_policy(...)` | The one policy: probe and recipe get the same thing. Building it is where the impurity lives (the per-run private `$HOME`); `scope()` owns its cleanup. | +| `exec_policy(...)` | Shared policy builder with caller-supplied write scope and `use_gpus` setting. Building it creates a private `$HOME`; `scope()` owns its cleanup. | | `Unavailable` | A real backend that wraps to the same argv and attests `fs: open`. Saying so is the caller's job; pretending is nobody's. | | `denial.explain()` / `denial.trailer()` | Best-guess remedies (allowed to return nothing) and the unconditional trailer on every nonzero sandboxed exit. | @@ -26,6 +26,18 @@ An optional output receiver gets stdout/stderr byte chunks. Capturing output nev decodes or normalizes stdout; only the retained stderr tail is decoded for denial classification. Without a receiver, stdout remains inherited. +`exec_policy(..., use_gpus=True)` passes the allocation's `CUDA_VISIBLE_DEVICES` +and optional `CUDA_DEVICE_ORDER` through unchanged. CPU commands receive an empty +CUDA mask. Direct GPU policies grant existing NVIDIA character device nodes. +Native permissions still apply; visibility is cooperative, and the reusable +worker's environment is never modified. + +The pure OCI rewrite adds podman-hpc's `--gpu` for GPU commands. Runtime selection +refuses explicit GPU recipes with ordinary Docker or Podman; probes on those +runtimes receive a CPU policy and an explanatory note. CPU containers remain +supported on all runtimes and set `NVIDIA_VISIBLE_DEVICES=void` to override image defaults. +See [GPU allocations](../user/cluster.md#gpu-allocations). + ## What must stay true - **`wrap` stays pure** — no temp files, no FDs, no global state diff --git a/docs/api/worker.md b/docs/api/worker.md index 3d6e255d..63ecaba8 100644 --- a/docs/api/worker.md +++ b/docs/api/worker.md @@ -18,14 +18,21 @@ Source: `src/lightcone/engine/worker.py`. Cluster execution supplies an output receiver to `materialize`/`execute`, which passes byte chunks from the sandbox back to the invocation. Standalone reruns -retain direct terminal output. +retain direct terminal output. The driver submits each cluster task with its +CPU, memory, and GPU reservations. Before resetting outputs, `execute` validates +resource syntax and builds the command policy with GPU access enabled only when +the recipe requests it. Standalone reruns apply the same checks but do not perform +Dask resource admission. GPU reruns require an explicit `CUDA_VISIBLE_DEVICES` in +the rerun environment, for example `CUDA_VISIBLE_DEVICES=0 datalad rerun`. Recipe +`time_limit` is explicitly refused. ## Key symbols | Symbol | Role | |---|---| -| `materialize(task, versions, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns a `TaskResult`, always. | -| `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, and the attestation. `.usable` is what dependents check. | +| `materialize(root, task, context, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns a `TaskResult`, always. | +| `execute(root, task, input_versions, context)` | Validate resource syntax and GPU policy, run a recipe unconditionally, then record its payload and manifest. | +| `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, reason, and diagnostic notes. `.usable` is what dependents check. | | `main(argv)` | The rerun entry point: guards, converges the project environment from the commit's own lock, resolves its own HEAD and runtime, executes. | | `lc_version()` | The engine version every manifest records. | @@ -35,6 +42,11 @@ retain direct terminal output. contract holds for failure modes nobody enumerated. Raising would make Dask abort every task in flight; reporting all independent failures in one run is most of what owning the loop buys. +- **Device visibility belongs to each command.** CPU recipes receive an empty + `CUDA_VISIBLE_DEVICES`; GPU recipes inherit the allocation's whole mask through + the sandbox policy. Never modify the reusable worker's shared environment. + Admission reserves the worker's full GPU budget for one GPU recipe at a time, + even when its minimum requested count is smaller. - **`data_version` is computed here, before anything is staged** — the dependent's argument *is* this return value, so the digest must exist while the files are still unannexed. Deriving it from @@ -67,3 +79,5 @@ retain direct terminal output. `tests/test_worker.py` — real recipes through the real boundary against a real repository (the `analysis` fixture): whether gates hold and bytes land are not questions a stub can answer. +`tests/test_gpu_execution.py` checks real subprocess masks and unsupported runtime +refusal before output deletion; it requires no physical GPU. diff --git a/docs/architecture.md b/docs/architecture.md index 5e55fb62..4c47acf4 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -40,7 +40,9 @@ lc materialize "$CLUSTER" │ refuse: dirty tree │ plan: astra validate + resolve → Graph of Tasks │ (no tasks → converge the crate and stop; nothing connects) + │ classify: current/behind outputs become values; other outputs may run │ connect: native identity + Dask readiness + │ admit: resource requests for outputs that may run fit a worker │ fetch: git annex get (declared inputs not in this clone) │ converge: uv.lock ⇄ .venv (and the image, containerized) ├─► workers: reset output file → sandbox → recipe → hash → manifest @@ -57,13 +59,22 @@ The division of labor is strict and load-bearing: `TaskResult`; the driver commits as results arrive, in one thread. Concurrent git operations race on the index lock — this split is not a preference. -- **Dask owns the ordering.** Every task is submitted with its - upstream futures as arguments; there is no ready-set loop or +- **Dask owns the ordering.** Tasks that may execute are submitted with their + upstream futures or already-current values as arguments; there is no ready-set loop or hand-rolled topological sort on the execution path. - **The worker never raises.** A recipe failure, a gate failure, an unreadable manifest — all come back as a state, so one failure doesn't abort every task in flight, and a run reports *all* its independent failures. +- **Dask accounts for task resources.** Workers advertise CPU, memory, and GPU + budgets; submissions reserve the recipe's requirements. Omitted memory adds no + RAM reservation. These coordinate scheduling rather than imposing per-recipe + OS limits or numerical-library thread counts. + Recipe `time_limit` is explicitly refused; allocation walltime remains supported. + A GPU recipe reserves the worker's full GPU budget and inherits its allocation + mask, even when it requests fewer GPUs. CPU recipes expose none; probes reserve + the whole worker budget. Direct and podman-hpc probes use the allocation mask, + while ordinary Docker and Podman probes use no GPUs. - **Values are resolved once and handed down.** HEAD, the container runtime, and the foreign-write facts are read by the driver and passed to workers as values — a worker that asked git itself could @@ -135,9 +146,10 @@ Because every backend is a pure argv rewrite, all of them are testable on a host that can't run them, and the manifest's `hermeticity` field records what was *actually* enforced — never what should have been. -There is one policy, `exec_policy`: probe and recipe get exactly the -same thing (tree read-only apart from `results/`), so "works under -`lc run`" and "works as a recipe" stay the same fact. +There is one policy builder, `exec_policy`: probes and recipes share environment +and filesystem rules, with write scope and GPU visibility supplied by the caller. +GPU recipes and supported probes inherit the worker's allocation mask; CPU +commands get an empty mask. ## The container hatch @@ -157,17 +169,26 @@ config-blob id, never a tag. `engine.compute` owns allocation lifecycle through a small provider protocol. A YAML catalog supplies ordered resource offers and stable native service namespaces. When the implicit default file is absent, a built-in local catalog -provides one CPU and 1 GiB without setup. An explicit catalog replaces that default; -missing explicit paths and invalid files remain errors. No catalog is written and +provides one CPU and 1 GiB without setup. GPU offers need an explicit catalog, +which replaces the built-in defaults. Missing explicit paths and invalid files +remain errors. No catalog is written and no allocation starts until `compute launch` resolves resources and submits once. Slurm queries and validated local OS identities are authoritative for allocations; Dask is the authority for connected workers. Private scheduler/TLS files are connection material, not a registry. +Allocation requests use SkyPilot-style CPU/memory exact or minimum quantities +and one accelerator type/count. Providers translate those requests into native +allocations; Lightcone does not depend on SkyPilot or carry its GPU alias registry. +Stock Dask workers inherit the allocation's native CUDA mask; Lightcone does not +probe GPU hardware. Container GPU access uses podman-hpc's native `--gpu` option. +See [GPU setup](user/cluster.md#gpu-allocations). + `compute.connect(CLUSTER_ID)` borrows a standard Dask client and closes only that client on exit. Both execution commands require a cluster ID. The materialization -scheduler keeps its `submit`/`completed` seam. Driver preparation and existing -task runtime/sandbox checks remain unchanged. Tasks use ordinary Dask scheduling; +scheduler validates resource requests, then keeps its `submit`/`completed` seam. +Driver preparation and existing task runtime/sandbox checks remain unchanged. +Tasks use ordinary Dask scheduling; there is no separate worker-selection or preflight layer, or site-marker guard. No execution command implicitly allocates compute. See [compute internals](api/compute.md) and [deployment limits](user/cluster.md). diff --git a/docs/cli/compute.md b/docs/cli/compute.md index e4d697eb..b082434b 100644 --- a/docs/cli/compute.md +++ b/docs/cli/compute.md @@ -5,7 +5,7 @@ No project is required for these commands. ```text lc compute resources [--json] -lc compute launch --cpus VALUE --memory VALUE +lc compute launch --cpus VALUE --memory VALUE [--gpus NAME[:COUNT]|0] [--name NAME] [--num-nodes N] [--time DURATION] [--startup fast] [--dry-run] [--json] lc compute status [CLUSTER] [--wait] [--timeout SECONDS] [--json] lc compute down CLUSTER [--json] @@ -15,6 +15,9 @@ Without configuration, `resources` exposes a built-in `local` offer: one CPU, 1 GiB, one node, fast startup, and a 30-minute default lifetime (two-hour maximum). Launch it with `lc compute launch --cpus 1 --memory 1`; execution still requires the returned cluster name or its full immutable ID. +GPU offers require an explicit catalog. On Linux, local GPU launches also require +an externally configured `CUDA_VISIBLE_DEVICES` mask; Lightcone does not discover +GPU hardware. See [GPU allocations](../user/cluster.md#gpu-allocations). `~/.lightcone/compute.yaml`, when present, replaces this built-in catalog. `LC_COMPUTE_CONFIG` selects another file for both compute and execution commands, @@ -53,9 +56,21 @@ durable reference to that allocation. Use the full immutable `id` from launch or status JSON to address one allocation directly, including when unrelated connections are unavailable. No name registry is maintained. -CPU quantities are logical CPUs **per node**, memory is **GiB per node**, and -`--num-nodes` defaults to one. Bare quantities are exact; `4+` means at least four. -Time accepts positive whole minutes or hours, such as `30m` or `2h`. Without +Resource quantities are **per node**, and `--num-nodes` defaults to one. CPU and +memory requests follow SkyPilot's exact/minimum convention: `4` is exact and `4+` +means at least four. Compute memory uses binary units: `16`, `16GB`, and `16GiB` +all mean 16 GiB; `16GB+` permits a larger offer. + +`--gpus A100:4` requests exactly four GPUs from an offer whose accelerator type is `A100`; +`--gpus A100` means one, `--gpus GPU:4` accepts any GPU model, and the default +`--gpus 0` selects CPU-only offers. Names match case-insensitively. GPU counts +are positive whole numbers, with no `+` or fractional form. Lightcone does not +maintain SkyPilot's accelerator alias registry: use the labels configured in +`resources` or use `GPU:N`. Catalog shapes use `accelerators: A100:4` or +`accelerators: {A100: 4}`. See [GPU allocations](../user/cluster.md#gpu-allocations). + +Time accepts positive durations with day/hour/minute/second units, such as `30m`, +`1h30m`, or `45s`. Without `--time`, the chosen offer's default applies. `fast` is a service class, not a queue-time promise. Limits apply to each allocation; aggregate quotas remain with the native backend. @@ -86,7 +101,9 @@ scheduler credentials: `phase` is `pending`, `active`, `stopping`, `ended`, or `unknown`. `allocation` holds `num_nodes`, per-node `resources`, and their `evidence` (`configured`, -`requested`, or `unknown`). `dask` is observed separately: `observation` +`requested`, or `unknown`). Resource objects contain `cpus`, `memory` in GiB, +and `accelerators` as a one-entry type/count mapping, or `null` for CPU-only shapes. +`dask` is observed separately: `observation` (`unverified`, `reachable`, or `unreachable`), `ready`, and `workers`. A discovery that partially succeeds still exits 1. Native errors, invalid requests, and readiness timeouts also exit 1. diff --git a/docs/cli/materialize.md b/docs/cli/materialize.md index e75a72b5..a323913c 100644 --- a/docs/cli/materialize.md +++ b/docs/cli/materialize.md @@ -20,8 +20,8 @@ cluster that is not active with every expected worker connected is refused cluster. Project validation and the dirty-tree check run before connecting to compute. A run whose spec selects no outputs still takes the CLUSTER argument but never connects to it: it only updates the publication view. A run whose -outputs are all current does connect, because each output's state is decided -on the cluster. +outputs are all current still validates the cluster connection, but submits no +Dask tasks and reserves no recipe resources. With no targets, everything the spec declares, across every universe. A target narrows the run to an output and whatever it depends on: @@ -56,8 +56,16 @@ never touched, under any flag. already belong to a newer allocation), and confirm its recipes have stopped before cleaning results. Local containers may need separate termination through their runtime; see [execution limits](../user/cluster.md#execution-requirements-and-limits). +- **Honors recipe resources.** CPU, memory, and GPU requests must fit one worker + and are reserved through standard Dask scheduling. GPU recipes run one at a + time per worker and inherit the whole allocation's CUDA mask, which may expose + more GPUs than requested. Recipes without `gpus` see none. Resource checks apply + to outputs that may rebuild; already-current outputs need no reservation. + Omitted memory reserves no RAM, so CPU requests and task slots control concurrency. + Recipe `time_limit` is unsupported and refused; allocation walltime remains supported. + See [recipe resource requirements](../user/cluster.md#recipe-resource-requirements). - **Fetches what it needs.** Declared inputs whose annexed content is - not in this clone are fetched before anything hashes. + not in this clone are fetched before workers hash or execute. - **Commits as it goes.** Each output lands in its own commit, written by the driver in one thread while other recipes keep running. - **Forwards recipe diagnostics.** Recipe stdout and stderr reach the invoking diff --git a/docs/cli/run.md b/docs/cli/run.md index 1e21c53e..312fa587 100644 --- a/docs/cli/run.md +++ b/docs/cli/run.md @@ -2,9 +2,8 @@ Run an ad-hoc command in the project environment, under isolation. This is the probe verb: it executes exactly one command the way a -recipe would be executed — same environment, same sandbox — so "does -it work under `lc run`?" and "will it work as a recipe?" are the same -question. +recipe would be executed — same environment, same sandbox. Recipes must also +declare the resources they need, including their GPU count. ## Synopsis @@ -40,6 +39,13 @@ that variable to an existing cluster. Set command-specific values inside the command, for example `lc run "$CLUSTER" -- env NAME=value python script.py`. Containerized commands use the image's environment and the sandbox overlays. +The command reserves one worker's full CPU, memory, and GPU budgets for its duration. +Direct and podman-hpc probes inherit the allocation's whole CUDA mask. Ordinary +Docker and Podman probes run without GPUs and report that limitation, even on a +GPU allocation. CPU-only allocations expose no GPUs. A recipe declares its +minimum GPU count explicitly; unsupported GPU recipes fail before image preparation. +See [GPU allocations](../user/cluster.md#gpu-allocations) for container prerequisites. + Interrupting the CLI detaches its client; the remote command may still be running. Stop the allocation with `lc compute down` and its full ID (a name can already belong to a newer allocation) before working with files the interrupted command diff --git a/docs/user/cluster.md b/docs/user/cluster.md index 0250908d..35cd38a1 100644 --- a/docs/user/cluster.md +++ b/docs/user/cluster.md @@ -8,10 +8,12 @@ present. `lc materialize --check` and `lc status` remain local project inspectio ## Start locally No configuration is needed on a fresh installation. When -`~/.lightcone/compute.yaml` is absent, Lightcone exposes one built-in `local` offer: +`~/.lightcone/compute.yaml` is absent, Lightcone exposes a built-in `local` CPU offer: one logical CPU, 1 GiB, one node, and fast startup. Its default lifetime is 30 minutes, with a maximum of two hours. This creates no catalog file and starts no processes until you launch a cluster. +Use a custom catalog for larger CPU or RAM budgets and for GPU offers. +See [GPU allocations](#gpu-allocations). ```bash lc compute resources @@ -134,15 +136,20 @@ list: optional `context`, and optional provider `launch` settings. Namespaces must be unique, and so must each provider/`context` pair. - An offer has a unique `name`, the `connection` it uses, per-node `resources` - (`cpus` and `memory` in GiB), `max_nodes`, and `time` with a `default` no - longer than its `max`. `startup` is optional (`fast`, `batch`, or the default + (`cpus`, `memory`, and optional `accelerators`), `max_nodes`, and `time` with a + `default` no longer than its `max`. `startup` is optional (`fast`, `batch`, or the default `unknown`), written either as a bare class or as `{class: …, source: …}`. `config` holds provider-specific settings. Catalog errors identify the invalid field, for example `offers.0.resources.cpus`. Unknown common fields and duplicate YAML keys are rejected. CPU and node counts -must be positive integers; memory is in GiB and may be fractional if it is an -exact number of bytes, and durations use minutes or hours such as `30m` or `2h`. +must be positive integers. Compute memory follows SkyPilot's binary-unit +convention: `32`, `32GB`, and `32GiB` mean 32 GiB. Fractional quantities must +represent an exact number of bytes. An accelerator declaration names one type +and a positive whole count: `accelerators: A100:4` or `accelerators: {A100: 4}`; +`accelerators: A100` means one. Omit it for CPU-only offers. Durations use ordered +day/hour/minute/second units, such as +`30m`, `1h30m`, or `45s`. Selection takes the first offer in catalog order that matches the request. An offer this host cannot provide is skipped: a local offer with more nodes, CPUs or @@ -203,8 +210,11 @@ offers: ``` An offer's `config` accepts `submit` (`sbatch`, the default, or `salloc`), -`account`, `partition`, `qos`, `constraint`, and `reservation`. Slurm offers must -state memory as a whole number of MiB. +`account`, `partition`, `qos`, `constraint`, `reservation`, and `gpu_type`. +For a named accelerator offer, `gpu_type` maps the public type to the site's +native Slurm GRES name; it is required even when the spellings happen to match. +A generic `GPU` offer may omit it. Slurm offers must state memory as a whole +number of MiB. Every setting under a Slurm connection's `launch` mapping is optional. The defaults assume a home directory that the login and compute nodes share: @@ -303,11 +313,149 @@ Each worker's files go under `//attempt-/`. Before starting Dask, every rank checks that Slurm gave it what the plan requested: the node count, CPUs per task and its actual CPU affinity, and memory -per node. Ranks other than zero wait up to 120 seconds for the scheduler, which +per node. GPU allocations also validate Slurm's native GPU count. +Ranks other than zero wait up to 120 seconds for the scheduler, which has as long to start. A failed check or timeout logs `Slurm Dask startup failed: …` to the submission log and exits nonzero. Look there when a job is active but never becomes ready. +## GPU allocations + +GPU support uses NVIDIA CUDA devices on Linux. `lc compute resources` shows +configured accelerator types and counts; Lightcone does not probe CUDA or discover +local hardware. There is no fractional GPU or MIG management. + +Requests use SkyPilot-style `NAME[:COUNT]`: `--gpus A100` means one A100, +`--gpus A100:4` means exactly four, and `--gpus GPU:4` accepts any configured model +with exactly four. Names are case-insensitive catalog labels; GPU counts do not +accept `+`. Omitting `--gpus`, or passing `0`, selects CPU-only offers. + +For local GPUs, add an offer to the [workstation catalog above](#customize-resource-offers): + +```yaml + - name: workstation-gpu + connection: workstation + resources: {cpus: 4, memory: 8GB, accelerators: 'GPU:1'} + max_nodes: 1 + time: {default: 30m, max: 2h} + startup: fast +``` + +Set the devices available to that allocation when launching it: + +```bash +CUDA_VISIBLE_DEVICES=0 lc compute launch --cpus 4 --memory 8GB --gpus GPU:1 +``` + +Lightcone freezes the nonempty mask and `CUDA_DEVICE_ORDER`, if set, at launch. +You are responsible for matching the catalog's count and model to those devices; +Lightcone does not verify them. Local allocations do not reserve GPUs exclusively +against other allocations or programs on the host. Before launch, the host must +have loaded the NVIDIA driver and created its character devices, including UVM. +Lightcone grants existing device nodes and does not initialize them; see +[NVIDIA's device setup utility](https://github.com/NVIDIA/nvidia-modprobe/blob/main/nvidia-modprobe.1.m4). + +On Slurm, the offer's `config.gpu_type` maps its catalog label to a native GRES +type. For example, add this offer with settings adjusted to your site: + +```yaml +- name: gpu-batch + connection: perlmutter + resources: {cpus: 32, memory: 128GB, accelerators: 'A100:4'} + max_nodes: 2 + time: {default: 1h, max: 4h} + startup: batch + config: + account: myproject + constraint: gpu + gpu_type: a100 +``` + +```bash +lc compute launch --cpus 32 --memory 128GB --gpus A100:4 --dry-run +``` + +This is an illustrative shape, not a tested site configuration. Slurm allocates +GPUs through GRES; Lightcone checks the native count and passes Slurm's CUDA mask +through unchanged, using `CUDA_DEVICE_ORDER=PCI_BUS_ID`. + +GPU commands inherit the allocation's whole CUDA mask. CPU commands receive an +empty mask. These are cooperative visibility settings; native OS and cgroup +permissions remain authoritative. + +Containerized GPU execution currently supports **podman-hpc** through its native +`--gpu` option. See [NERSC's GPU container guidance](https://docs.nersc.gov/development/containers/podman-hpc/overview/#using-nvidia-gpus-in-podman-hpc). +Recipes explicitly requesting GPUs with ordinary Docker or Podman are refused +before image preparation. `lc run` probes on those runtimes remain usable on a +GPU cluster: they run without GPUs and report that limitation. CPU execution +supports all three runtimes and sets `NVIDIA_VISIBLE_DEVICES=void` to override +GPU-enabled image defaults. Physical GPU execution remains a deployment +validation step. + +A standalone GPU rerun needs a device mask in its own environment, for example +`CUDA_VISIBLE_DEVICES=0 datalad rerun`. It does not inherit an old allocation's +mask or reserve devices through Dask. + +## Recipe resource requirements + +Declare each recipe's needs in `astra.yaml`: + +```yaml +recipe: + command: python src/fit.py {output} + resources: + cpus: 4 + memory: 8Gi + gpus: 1 +``` + +Each recipe runs on one worker. Its CPU, memory, and GPU request must fit that +worker, even when the cluster has several nodes. Dask reserves CPU and memory +while the task runs, so recipes can run together only when their combined +requests fit. `task_slots_per_node` also caps concurrent tasks; it does not +limit how many CPUs a single recipe may request. + +CPUs must be positive whole numbers and default to one. Memory needs units: +`512Mi` and `8Gi` are binary sizes; `8GB` is decimal, unlike compute memory. +Bare quantities are not accepted. Without a memory declaration, no RAM is +reserved: CPU requests and `task_slots_per_node` control concurrency. + +Recipe `gpus` is a nonnegative whole count, defaulting to zero; accelerator type +selection belongs to cluster allocation. A GPU recipe reserves the worker's +entire GPU budget, so only one GPU recipe runs on that worker at a time. The +requested count is a minimum capacity requirement: the command inherits the +worker's whole allocated CUDA mask and may see more GPUs than requested. CPU +recipes may still run alongside it when CPU, memory, and task slots permit; their +CUDA mask is empty. `lc run` reserves the worker's entire CPU, memory, and GPU +budgets; direct and podman-hpc probes inherit that allocation mask. + +Recipe `time_limit` is not supported and is refused before preparation or +execution. Set the allocation lifetime with `lc compute launch --time` instead. +Fractional CPU/GPU counts, GPU model requests inside a recipe, and disk requests +are also rejected rather than ignored. + +`lc materialize` first classifies the selected graph, then checks resources for +outputs that may rebuild before preparation or submission. Already-current or +unrefreshed behind outputs reserve nothing. Dependents of an output that may +change still need resources; the worker may later skip them if the actual +upstream digest is unchanged. Use `lc materialize --check` to inspect currency +without allocation. Read-only `status` and `--check` accept valid ASTRA resource +declarations even when this executor cannot satisfy them. + +These are scheduling reservations, not per-recipe CPU or RAM enforcement. +Recipes must respect their declarations; a subprocess can otherwise exceed +its request. Slurm enforces the overall allocation, while local execution +uses cooperative budgets. Leave capacity for the scheduler, workers, and other +overhead when declaring recipe requirements. + +A CPU reservation does not set numerical-library thread counts. Local clusters +use Dask's Nanny defaults of `1` for `OMP_NUM_THREADS`, `MKL_NUM_THREADS`, and +`OPENBLAS_NUM_THREADS` when those variables are unset. Set the variables before +`lc compute launch`, or in the recipe command, to choose another value. See +[Dask's defaults](https://docs.dask.org/en/stable/configuration.html#distributed.nanny.pre-spawn-environ.OMP_NUM_THREADS) +and [environment precedence](https://distributed.dask.org/en/stable/_modules/distributed/nanny.html). +Slurm workers run directly without a Nanny and inherit the job's thread settings. + ## Execution requirements and limits Driver and workers must see the same project, prepared environment, and inputs diff --git a/src/lightcone/cli/compute.py b/src/lightcone/cli/compute.py index 59fb2517..aa139375 100644 --- a/src/lightcone/cli/compute.py +++ b/src/lightcone/cli/compute.py @@ -49,6 +49,11 @@ def _table(headers: list[str], rows: list[list[str]]) -> None: Console(markup=False).print(table) +def _duration(seconds: int) -> str: + minutes, remainder = divmod(seconds, 60) + return (f"{minutes}m" if minutes else "") + (f"{remainder}s" if remainder else "") + + @click.group() def compute() -> None: """Allocate resources, inspect clusters, and end allocations. @@ -70,15 +75,19 @@ def resources(as_json: bool) -> None: click.echo(json.dumps(data)) return _table( - ["OFFER", "CPUS", "MEMORY", "MAX NODES", "DEFAULT", "MAX TIME", "STARTUP"], + ["OFFER", "CPUS", "MEMORY", "GPUS", "MAX NODES", "DEFAULT", "MAX TIME", "STARTUP"], [ [ offer["name"], str(offer["resources"]["cpus"]), f"{offer['resources']['memory']:g} GiB", + ", ".join( + f"{name}:{count}" + for name, count in (offer["resources"]["accelerators"] or {}).items() + ) or "-", str(offer["max_nodes"]), - f"{offer['time']['default_seconds'] // 60}m", - f"{offer['time']['max_seconds'] // 60}m", + _duration(offer["time"]["default_seconds"]), + _duration(offer["time"]["max_seconds"]), offer["startup"], ] for offer in data["offers"] @@ -89,10 +98,13 @@ def resources(as_json: bool) -> None: @compute.command() @click.option("--name", help="Cluster name; defaults to a generated short name.") @click.option("--cpus", required=True, help="Logical CPUs per node; suffix + requests a minimum.") -@click.option("--memory", required=True, help="GiB per node; suffix + requests a minimum.") +@click.option("--memory", required=True, + help="Memory per node, e.g. 16 or 16GB; suffix + requests a minimum.") +@click.option("--gpus", default="0", show_default=True, + help="Accelerator NAME[:COUNT] per node, e.g. A100:4 or GPU:1; 0 requests CPU only.") @click.option("--num-nodes", default=1, type=click.IntRange(min=1), show_default=True) @click.option( - "--time", "walltime", help="Requested walltime, e.g. 30m or 2h; defaults to the offer." + "--time", "walltime", help="Requested walltime, e.g. 30m or 1h30m; defaults to the offer." ) @click.option( "--startup", type=click.Choice(["fast"]), help="Require a fast startup service class." @@ -103,6 +115,7 @@ def launch( name: str | None, cpus: str, memory: str, + gpus: str, num_nodes: int, walltime: str | None, startup: str | None, @@ -119,6 +132,7 @@ def launch( Request.parse( cpus, memory, + gpus=gpus, num_nodes=num_nodes, time=walltime, startup=startup, diff --git a/src/lightcone/engine/compute/__init__.py b/src/lightcone/engine/compute/__init__.py index 0f3c40c6..2139ca9b 100644 --- a/src/lightcone/engine/compute/__init__.py +++ b/src/lightcone/engine/compute/__init__.py @@ -102,7 +102,10 @@ def resources(self) -> dict[str, Any]: """Describe configured policy, without inventing live free capacity.""" return { "schema_version": 1, - "units": {"cpus": "logical CPUs per node", "memory": "GiB per node"}, + "units": { + "cpus": "logical CPUs per node", "memory": "GiB per node", + "accelerators": "type and count per node", + }, "offers": [ { "name": offer.name, @@ -142,6 +145,12 @@ def plan(self, request: Request, *, name: str | None = None) -> LaunchPlan: else offer.resources.memory_bytes != request.memory_bytes ): continue + if request.accelerators is None: + matches_accelerators = offer.resources.accelerators is None + else: + matches_accelerators = request.accelerators.matches(offer.resources.accelerators) + if not matches_accelerators: + continue try: provider = self.provider(self.catalog.connections[offer.connection]) plan = provider.plan(offer, request) diff --git a/src/lightcone/engine/compute/local.py b/src/lightcone/engine/compute/local.py index 0474d146..85b05272 100644 --- a/src/lightcone/engine/compute/local.py +++ b/src/lightcone/engine/compute/local.py @@ -104,6 +104,15 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: if offer.resources.cpus > CPU_COUNT or offer.resources.memory_bytes > MEMORY_LIMIT: raise UnavailableOfferError("the local offer exceeds this host's CPU or RAM capacity") + mask = "" + if offer.resources.gpus: + if sys.platform != "linux": + raise UnavailableOfferError("local GPU allocations require Linux") + mask = os.environ.get("CUDA_VISIBLE_DEVICES", "") + if not mask: + raise UnavailableOfferError( + "local GPU offers require an explicit nonempty CUDA_VISIBLE_DEVICES mask" + ) seconds = request.seconds if request.seconds is not None else offer.time.default_seconds if seconds <= 0 or seconds > offer.time.max_seconds: raise ComputeError("local allocations require a finite time within the offer's limit") @@ -123,7 +132,9 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: "connection_root": str(self.root), "scratch_root": str(scratch), "task_slots_per_node": slots, - "resource_enforcement": "cooperative; no exclusive CPU or RAM reservation", + "cuda_visible_devices": mask, + "cuda_device_order": os.environ.get("CUDA_DEVICE_ORDER"), + "resource_enforcement": "cooperative; no exclusive CPU, RAM, or GPU reservation", "termination_grace_seconds": _STOP_GRACE, }, ) @@ -150,6 +161,11 @@ def launch(self, plan: LaunchPlan) -> Identity: ) process: subprocess.Popen[bytes] | None = None identity: Identity | None = None + environment = {**os.environ, "CUDA_VISIBLE_DEVICES": plan.details["cuda_visible_devices"]} + if plan.details["cuda_device_order"] is None: + environment.pop("CUDA_DEVICE_ORDER", None) + else: + environment["CUDA_DEVICE_ORDER"] = plan.details["cuda_device_order"] try: # This allocation outlives a command; the ordinary run-to-completion # subprocess seam cannot own it. Logs are discarded rather than grow. @@ -160,6 +176,7 @@ def launch(self, plan: LaunchPlan) -> Identity: stderr=subprocess.DEVNULL, start_new_session=True, close_fds=True, + env=environment, ) identity = Identity( namespace=self.connection.namespace, native_id=str(process.pid), token=token, @@ -174,6 +191,8 @@ def launch(self, plan: LaunchPlan) -> Identity: "host": identity.host, "cpus": plan.resources.cpus, "memory": plan.resources.memory_bytes, + "gpus": plan.resources.gpus, + "accelerator_name": plan.resources.accelerator_name or "GPU", } write_private_json(directory / "identity.json", record) # The child waits for this file before publishing its TLS connection. @@ -233,6 +252,12 @@ def _record(self, identity: Identity) -> tuple[Path, dict[str, Any]]: raise ComputeError("the private locator does not match this local allocation identity") positive_int(record.get("cpus"), "recorded local cpus") positive_int(record.get("memory"), "recorded local memory") + record.setdefault("gpus", 0) + record.setdefault("accelerator_name", "GPU") + if type(record.get("gpus")) is not int or record["gpus"] < 0: + raise ComputeError("recorded local gpus must be a nonnegative integer") + if not isinstance(record.get("accelerator_name"), str) or not record["accelerator_name"]: + raise ComputeError("recorded local accelerator_name must be a nonempty string") return directory, record def _process( @@ -347,7 +372,7 @@ def inspect(self, identity: Identity) -> Snapshot: except psutil.NoSuchProcess: process = None native_state = "not-running" - reason = "CPU and RAM budgets are cooperative, not exclusive OS reservations" + reason = "CPU, RAM, and GPU budgets are cooperative, not exclusive OS reservations" if process is None and (directory / "error.json").exists(): reason = str(read_private_json(directory / "error.json").get("error", "")) return Snapshot( @@ -355,6 +380,8 @@ def inspect(self, identity: Identity) -> Snapshot: phase="active" if process is not None else "ended", resources=Resources.from_bytes( cpus=int(record["cpus"]), memory_bytes=int(record["memory"]), + gpus=record["gpus"], + accelerator_name=record["accelerator_name"], ), num_nodes=1, evidence="configured", diff --git a/src/lightcone/engine/compute/local_runtime.py b/src/lightcone/engine/compute/local_runtime.py index dbb360e6..1ca6d552 100644 --- a/src/lightcone/engine/compute/local_runtime.py +++ b/src/lightcone/engine/compute/local_runtime.py @@ -54,6 +54,10 @@ def expire(_signum: int, _frame: FrameType | None) -> None: from distributed import LocalCluster security = create_security(directory) + allocation = read_private_json(directory / "identity.json") + gpus = int(allocation.get("gpus", 0)) + if not gpus: + os.environ["CUDA_VISIBLE_DEVICES"] = "" with dask.config.set(SCHEDULER_CONFIG), LocalCluster( # type: ignore[no-untyped-call] n_workers=1, threads_per_worker=int(launch["task_slots"]), @@ -73,6 +77,9 @@ def expire(_signum: int, _frame: FrameType | None) -> None: # Recipes use subprocesses: Dask's Python-process RSS cannot enforce # their RAM envelope. Local resource limits are explicitly cooperative. memory_limit=0, + resources={ + "CPU": int(allocation["cpus"]), "MEMORY": int(allocation["memory"]), "GPU": gpus, + }, silence_logs=50, ) as cluster: write_private_json( diff --git a/src/lightcone/engine/compute/model.py b/src/lightcone/engine/compute/model.py index ba7a25d1..eb16380f 100644 --- a/src/lightcone/engine/compute/model.py +++ b/src/lightcone/engine/compute/model.py @@ -7,7 +7,7 @@ import re from collections.abc import Callable, Sequence from contextlib import AbstractContextManager -from decimal import Decimal, InvalidOperation, localcontext +from decimal import Decimal, localcontext from typing import Annotated, Any, Literal, Protocol, Self from uuid import UUID @@ -19,10 +19,13 @@ Field, PlainSerializer, ValidationError, + field_validator, + model_serializer, model_validator, ) from lightcone.engine.project import ProjectError +from lightcone.engine.units import whole_bytes GIB = 1024**3 @@ -43,25 +46,28 @@ class UnavailableOfferError(ComputeError): def duration(value: object) -> int: - """Parse an explicit positive whole-minute/hour duration into seconds.""" - match = re.fullmatch(r"([1-9][0-9]*)([mh])", str(value)) - if match is None: - raise ComputeError("duration must be a positive number of minutes or hours, e.g. 30m or 1h") - return int(match[1]) * (60 if match[2] == "m" else 3600) + """Parse ordered day/hour/minute/second components, raising ValueError.""" + if not isinstance(value, str) or not ( + match := re.fullmatch(r"(?:(\d+)d)?(?:(\d+)h)?(?:(\d+)m)?(?:(\d+)s)?", value) + ): + raise ValueError("duration must use explicit units, e.g. 30m, 1h30m, or 45s") + seconds = sum(int(part or 0) * unit for part, unit in zip(match.groups(), (86400, 3600, 60, 1))) + if seconds <= 0: + raise ValueError("duration must be positive") + return seconds def memory_bytes(value: object) -> int: - """Convert positive decimal GiB to an exact integer number of bytes.""" - if isinstance(value, bool) or not re.fullmatch(r"[0-9]+(?:\.[0-9]+)?", str(value)): - raise ComputeError("memory must be a positive number of GiB") + """Parse SkyPilot compute memory: bare GiB or binary KB/MB/GB/TB/PB units.""" + match = re.fullmatch(r"([0-9]+(?:\.[0-9]+)?)([KMGTPE]I?B|B)?", str(value), re.IGNORECASE) + if isinstance(value, bool) or match is None: + raise ComputeError("memory must be positive GiB or a quantity with KB/MB/GB/TB/PB units") + unit = (match[2] or "GB").upper().replace("I", "") + units = {"B": 1, **{f"{prefix}B": 1024**index for index, prefix in enumerate("KMGTPE", 1)}} try: - numerator, denominator = Decimal(str(value)).as_integer_ratio() - except InvalidOperation as exc: - raise ComputeError("memory must be a positive number of GiB") from exc - amount, remainder = divmod(numerator * GIB, denominator) - if amount <= 0 or remainder: - raise ComputeError("memory must be positive GiB exactly representable in bytes") - return amount + return whole_bytes(match[1], units[unit]) + except ValueError as exc: + raise ComputeError("memory must be positive GiB exactly representable in bytes") from exc def gib_from_bytes(value: int) -> Decimal: @@ -124,17 +130,14 @@ def _gib(value: object) -> Decimal: # digits when applying the same quantity rules as the CLI. literal = format(value, "f") if isinstance(value, Decimal) else value try: - memory_bytes(literal) + size = memory_bytes(literal) except ComputeError as exc: raise ValueError(str(exc)) from exc - return Decimal(str(literal)) + return gib_from_bytes(size) def _duration(value: str) -> str: - try: - duration(value) - except ComputeError as exc: - raise ValueError(str(exc)) from exc + duration(value) return value @@ -161,24 +164,83 @@ def replace(self, **changes: Any) -> Self: return type(self).model_validate({**self.model_dump(), **changes}) +class Accelerator(ComputeModel): + """One accelerator type and whole-device count, using SkyPilot's notation.""" + + name: Annotated[str, Field(pattern=r"^[A-Za-z0-9][A-Za-z0-9_.-]*$")] + count: PositiveInt = 1 + + @field_validator("name") + @classmethod + def unambiguous_name(cls, value: str) -> str: + if value in {"name", "count"}: + raise ValueError("accelerator names cannot be reserved fields 'name' or 'count'") + return value + + @model_validator(mode="before") + @classmethod + def shorthand(cls, value: Any) -> Any: + if isinstance(value, str): + match = re.fullmatch(r"([A-Za-z0-9][A-Za-z0-9_.-]*)(?::([1-9][0-9]*))?", value) + if match is None: + raise ValueError("accelerators must be NAME[:COUNT] with a positive whole count") + return {"name": match[1], "count": int(match[2] or 1)} + if isinstance(value, dict) and not value.keys() & {"name", "count"}: + if len(value) != 1: + raise ValueError("accelerators must specify exactly one type and whole count") + name, count = next(iter(value.items())) + return {"name": name, "count": count} + return value + + @model_serializer + def serialize(self) -> dict[str, int]: + """Use the same single-type mapping in native records and public output.""" + return {self.name: self.count} + + def matches(self, available: Accelerator | None) -> bool: + """Match exact counts and types; the generic GPU name accepts any type.""" + return available is not None and self.count == available.count and ( + self.name.casefold() == "gpu" or self.name.casefold() == available.name.casefold() + ) + + class Resources(ComputeModel): """A per-node resource envelope with explicitly named memory units.""" cpus: Count memory_gib: GiB = Field(validation_alias="memory", serialization_alias="memory") + accelerators: Accelerator | None = None + + @property + def gpus(self) -> int: + return self.accelerators.count if self.accelerators is not None else 0 + + @property + def accelerator_name(self) -> str | None: + return self.accelerators.name if self.accelerators is not None else None @property def memory_bytes(self) -> int: return memory_bytes(format(self.memory_gib, "f")) @classmethod - def from_bytes(cls, *, cpus: int, memory_bytes: int) -> Self: + def from_bytes( + cls, *, cpus: int, memory_bytes: int, gpus: int = 0, accelerator_name: str = "GPU", + ) -> Self: """Represent native byte counts exactly, independently of Decimal precision.""" - return cls(cpus=cpus, memory_gib=gib_from_bytes(memory_bytes)) + if type(gpus) is not int or gpus < 0: + raise ValueError("gpus must be a nonnegative whole count") + return cls( + cpus=cpus, memory_gib=gib_from_bytes(memory_bytes), + accelerators=Accelerator(name=accelerator_name, count=gpus) if gpus else None, + ) - def as_dict(self) -> dict[str, int | float]: + def as_dict(self) -> dict[str, Any]: """Render public memory in GiB.""" - return {"cpus": self.cpus, "memory": self.memory_bytes / GIB} + return { + "cpus": self.cpus, "memory": self.memory_bytes / GIB, + "accelerators": self.accelerators.model_dump() if self.accelerators else None, + } class Request(ComputeModel): @@ -186,6 +248,7 @@ class Request(ComputeModel): cpus: PositiveInt memory_bytes: PositiveInt + accelerators: Accelerator | None = None num_nodes: PositiveInt = 1 min_cpus: bool = False min_memory: bool = False @@ -198,6 +261,7 @@ def parse( cpus: str, memory: str, *, + gpus: str = "0", num_nodes: int = 1, time: str | None = None, startup: str | None = None, @@ -207,6 +271,7 @@ def parse( return cls.model_validate({ "cpus": positive_int(cpus.removesuffix("+"), "cpus"), "memory_bytes": memory_bytes(memory.removesuffix("+")), + "accelerators": None if gpus == "0" else gpus, "num_nodes": positive_int(num_nodes, "num_nodes"), "min_cpus": cpus.endswith("+"), "min_memory": memory.endswith("+"), @@ -215,6 +280,8 @@ def parse( }) except ValidationError as exc: raise ComputeError(f"invalid compute request:\n{validation_message(exc)}") from exc + except ValueError as exc: + raise ComputeError(str(exc)) from exc def as_dict(self) -> dict[str, Any]: """Render the request without native provider settings.""" @@ -223,6 +290,7 @@ def as_dict(self) -> dict[str, Any]: "resources": { "cpus": f"{self.cpus}{'+' if self.min_cpus else ''}", "memory": f"{gib_from_bytes(self.memory_bytes):f}{'+' if self.min_memory else ''}", + "accelerators": self.accelerators.model_dump() if self.accelerators else None, }, "time_seconds": self.seconds, "startup": self.startup, diff --git a/src/lightcone/engine/compute/slurm.py b/src/lightcone/engine/compute/slurm.py index 86d66729..7cad67aa 100644 --- a/src/lightcone/engine/compute/slurm.py +++ b/src/lightcone/engine/compute/slurm.py @@ -100,6 +100,46 @@ def _value(value: object, name: str) -> str: return value +def _native_gpus(row: Mapping[str, str]) -> tuple[str, int] | None: + """Read a per-node GPU count, never divide an aggregate into invented grants.""" + for field in ("TresPerNode", "Gres"): + value = row.get(field, "") + counts = [] + names = set() + for entry in value.split(","): + if field == "TresPerNode": + entry = entry.removeprefix("gres/").removeprefix("gres:") + if not entry.startswith("gpu"): + continue + match = re.fullmatch( + r"gpu(?::([A-Za-z0-9][A-Za-z0-9_.-]*))?[:=]([0-9]+)", entry, + ) + if match is None: + return None + names.add(match[1] or "GPU") + counts.append(int(match[2])) + if counts: + if "GPU" in names and len(names) > 1: + return None # A total plus typed subcounts must not be double-counted. + return next(iter(names)) if len(names) == 1 else "GPU", sum(counts) + if field == "Gres" and value in {"(null)", "N/A", "none"}: + return "GPU", 0 + # Complete native TRES with no GPU entry proves a CPU-only job. An + # aggregate GPU total does not prove a homogeneous per-node allocation. + for field in ("ReqTRES", "AllocTRES", "TRES"): + value = row.get(field, "") + if value and value not in {"(null)", "N/A"}: + entries = value.split(",") + if any(entry.startswith("gres/gpu") for entry in entries): + return None + if all("=" in entry for entry in entries) and any( + entry.startswith("cpu=") for entry in entries + ): + return "GPU", 0 + return None + return None + + class SlurmProvider: """Submit, observe, and cancel allocations using the selected Slurm authority.""" @@ -185,11 +225,26 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: "qos", "constraint", "reservation", + "gpu_type", }: raise ComputeError(f"unknown Slurm offer settings: {', '.join(sorted(extra))}") submit = config.get("submit", "sbatch") if submit not in {"sbatch", "salloc"}: raise ComputeError("Slurm submit must be sbatch or salloc") + gpu_type = config.get("gpu_type") + if gpu_type is not None: + if not isinstance(gpu_type, str) or not re.fullmatch( + r"[A-Za-z0-9][A-Za-z0-9_.-]*", gpu_type, + ): + raise ComputeError("Slurm gpu_type must name one native GPU GRES type") + if not offer.resources.gpus: + raise ComputeError("Slurm gpu_type requires accelerator resources") + elif (offer.resources.accelerator_name or "GPU").casefold() != "gpu": + raise ComputeError("a named Slurm accelerator offer requires its native gpu_type") + gres = ( + f"gpu:{gpu_type + ':' if gpu_type else ''}{offer.resources.gpus}" + if offer.resources.gpus else "none" + ) cpus, memory = offer.resources.cpus, offer.resources.memory_bytes if memory % _MIB: raise ComputeError("Slurm offer memory must be an exact whole number of MiB") @@ -223,6 +278,8 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: f"--time={hours:02}:{minutes:02}:{seconds_part:02}", f"--chdir={paths['cwd']}", ] + if offer.resources.gpus: + args.append(f"--gres={gres}") return LaunchPlan( connection=self.connection, offer=offer, @@ -231,6 +288,7 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: details={ "submit": submit, "native_args": args, + "gres": gres, **paths, "task_slots_per_node": slots, "interface": interface, @@ -244,6 +302,7 @@ def _payload(self, plan: LaunchPlan, token: str) -> list[str]: f"--ntasks={plan.num_nodes}", "--ntasks-per-node=1", f"--cpus-per-task={plan.resources.cpus}", + f"--gres={details['gres']}", # One process per node holds the whole allocation, so binding to # exactly its allocated hardware threads is the only useful mask. "--cpu-bind=threads", @@ -264,6 +323,8 @@ def _payload(self, plan: LaunchPlan, token: str) -> list[str]: str(plan.resources.cpus), "--memory-bytes", str(plan.resources.memory_bytes), + "--gpus", + str(plan.resources.gpus), "--task-slots", str(details["task_slots_per_node"]), ] @@ -495,11 +556,16 @@ def _snapshot(self, identity: Identity, row: Mapping[str, str]) -> Snapshot: nodes = int(nodes_text) if nodes_text.isdigit() and int(nodes_text) > 0 else None cpus = row.get("CPUs/Task", "") memory = re.fullmatch(r"([0-9]+)([KMGT]?)", row.get("MinMemoryNode", "")) + accelerators = _native_gpus(row) resources = None - if cpus.isdigit() and int(cpus) > 0 and memory and int(memory[1]) > 0: + if ( + cpus.isdigit() and int(cpus) > 0 and memory and int(memory[1]) > 0 + and accelerators is not None + ): scale = {"": _MIB, "K": 1024, "M": _MIB, "G": 1024**3, "T": 1024**4} resources = Resources.from_bytes( - cpus=int(cpus), memory_bytes=int(memory[1]) * scale[memory[2]] + cpus=int(cpus), memory_bytes=int(memory[1]) * scale[memory[2]], + accelerator_name=accelerators[0], gpus=accelerators[1], ) return Snapshot( identity=identity, diff --git a/src/lightcone/engine/compute/slurm_bootstrap.py b/src/lightcone/engine/compute/slurm_bootstrap.py index 6e35cdf0..2ef71a3c 100644 --- a/src/lightcone/engine/compute/slurm_bootstrap.py +++ b/src/lightcone/engine/compute/slurm_bootstrap.py @@ -48,6 +48,7 @@ def _allocation(args: argparse.Namespace) -> tuple[Identity, int, int]: args.num_nodes < 1 or args.cpus < 1 or args.memory_bytes < 1 + or args.gpus < 0 or args.task_slots < 1 or values["SLURM_NTASKS"] != args.num_nodes or values["SLURM_JOB_NUM_NODES"] != args.num_nodes @@ -61,6 +62,12 @@ def _allocation(args: argparse.Namespace) -> tuple[Identity, int, int]: memory = os.environ.get("SLURM_MEM_PER_NODE", "") if not memory.isdigit() or int(memory) * 1024**2 < args.memory_bytes: raise ComputeError("native per-node memory does not match the allocation envelope") + if args.gpus: + native_gpus = os.environ.get("SLURM_GPUS_ON_NODE", "") + if not native_gpus.isdigit() or int(native_gpus) < args.gpus: + raise ComputeError("native per-node GPUs do not match the allocation envelope") + if not os.environ.get("CUDA_VISIBLE_DEVICES"): + raise ComputeError("GPU allocations require a native CUDA_VISIBLE_DEVICES mask") restarts = os.environ.get("SLURM_RESTART_COUNT", "0") if not restarts.isdigit(): raise ComputeError("invalid native Slurm restart count") @@ -76,6 +83,11 @@ async def run(args: argparse.Namespace) -> None: from distributed import Scheduler, Worker identity, restarts, rank = _allocation(args) + if args.gpus: + # Slurm/NVML numbers devices in PCI order; CUDA's default is different. + os.environ["CUDA_DEVICE_ORDER"] = "PCI_BUS_ID" + else: + os.environ["CUDA_VISIBLE_DEVICES"] = "" connection = Connection( namespace=identity.namespace, provider="slurm", launch={"connection_root": args.connection_root}, @@ -94,6 +106,7 @@ async def run(args: argparse.Namespace) -> None: "num_nodes": args.num_nodes, "cpus": args.cpus, "memory_bytes": args.memory_bytes, + "gpus": args.gpus, "task_slots": args.task_slots, } address = {"interface": args.interface} if args.interface else {"host": socket.gethostname()} @@ -101,6 +114,7 @@ async def run(args: argparse.Namespace) -> None: **address, "nthreads": args.task_slots, "memory_limit": 0, + "resources": {"CPU": args.cpus, "MEMORY": args.memory_bytes, "GPU": args.gpus}, "local_directory": str(scratch), "dashboard_address": "127.0.0.1:0", "dashboard": False, @@ -159,6 +173,7 @@ def main() -> None: parser.add_argument(f"--{name}", required=True) for name in ("num-nodes", "cpus", "memory-bytes", "task-slots"): parser.add_argument(f"--{name}", required=True, type=int) + parser.add_argument("--gpus", default=0, type=int) parser.add_argument("--scratch-root") parser.add_argument("--interface") args = parser.parse_args() diff --git a/src/lightcone/engine/container.py b/src/lightcone/engine/container.py index d7f995de..b109b482 100644 --- a/src/lightcone/engine/container.py +++ b/src/lightcone/engine/container.py @@ -66,6 +66,13 @@ class Runtime: #: The architecture the archive was built for. arch: str = "" + @property + def supports_gpus(self) -> bool: + """Whether this execution mode can expose the allocation's GPUs.""" + from lightcone.engine.sandbox.oci import supports_gpus + + return self.mode == "direct" or supports_gpus(self.runtime) + @property def archive(self) -> str: """The committed archive, project-relative — what the run record's @@ -87,7 +94,7 @@ def manifest_image(self) -> dict[str, str] | None: } -def runtime_for_run(root: Path, *, build: bool) -> Runtime: +def runtime_for_run(root: Path, *, build: bool, use_gpus: bool = False) -> Runtime: """Resolve the execution world, converging the image where allowed. The three image checks are repository questions first and runtime @@ -102,6 +109,7 @@ def runtime_for_run(root: Path, *, build: bool) -> Runtime: Args: root: The project root. build: Whether a missing archive may be built and committed. + use_gpus: Refuse unsupported GPU containers before preparing their image. Returns: The resolved runtime; a direct-mode one costs a TOML read. @@ -115,6 +123,10 @@ def runtime_for_run(root: Path, *, build: bool) -> Runtime: return Runtime(root=root, mode="direct", env_dir=project.env_dir(root)) name = runtime_name(root) + if use_gpus: + from lightcone.engine.sandbox.oci import require_gpu_runtime + + require_gpu_runtime(name) tag = image.tag(root) archive = image.archive_path(root, tag) if not _committed(archive): @@ -408,7 +420,8 @@ def converge(runtime: Runtime) -> list[str]: def policy_for( - runtime: Runtime, read_paths: list[Path], *, write_dir: Path | None = None + runtime: Runtime, read_paths: list[Path], *, write_dir: Path | None = None, + use_gpus: bool = False, ) -> sandbox.Policy: """Build the exec policy for a resolved runtime. @@ -422,16 +435,22 @@ def policy_for( runtime: The resolved runtime. read_paths: Declared inputs, as :func:`sandbox.exec_policy` takes. write_dir: The directory a recipe's output lands in; absent for a probe. + use_gpus: Inherit the allocation's CUDA mask; otherwise hide GPUs. Returns: The policy for this world. """ + if use_gpus and runtime.mode == "containerized": + from lightcone.engine.sandbox.oci import require_gpu_runtime + + require_gpu_runtime(runtime.runtime) return sandbox.exec_policy( runtime.root, read_paths=read_paths, env_dir=runtime.env_dir, containerized=runtime.mode == "containerized", write_dir=write_dir, + use_gpus=use_gpus, ) @@ -682,5 +701,3 @@ def _machine_preflight(root: Path) -> None: f"empty. Share it:\n podman machine stop\n" f" podman machine set --volume {root}\n podman machine start" ) - - diff --git a/src/lightcone/engine/execution_resources.py b/src/lightcone/engine/execution_resources.py new file mode 100644 index 00000000..e9d71209 --- /dev/null +++ b/src/lightcone/engine/execution_resources.py @@ -0,0 +1,171 @@ +"""Recipe resource requests and admission to stock Dask workers.""" + +from __future__ import annotations + +import math +import re +from typing import Any, Self + +from pydantic import BaseModel, ConfigDict, Field, ValidationError, field_validator + +from lightcone.engine.project import ProjectError +from lightcone.engine.units import whole_bytes + + +class TaskResources(BaseModel): + """Reserve CPU, memory, and minimum GPU capacity on stock Dask workers.""" + + model_config = ConfigDict(frozen=True, strict=True, extra="forbid") + + cpus: int = Field(default=1, gt=0) + gpus: int = Field(default=0, ge=0) + memory_bytes: int | None = Field(default=None, gt=0) + + @field_validator("cpus", mode="before") + @classmethod + def _whole_cpus(cls, value: object) -> object: + # ASTRA permits fractional CPUs. This executor reserves whole CPUs; + # accepting 4.0 is exact, whereas rounding 0.5 would hide a policy change. + if isinstance(value, float): + if not math.isfinite(value) or not value.is_integer(): + raise ValueError("fractional CPUs are not supported; request whole CPUs") + return int(value) + return value + + @classmethod + def parse(cls, value: object) -> Self: + """Parse ASTRA's ``recipe.resources`` into explicit execution units. + + Args: + value: The recipe resource mapping, or ``None`` when omitted. + + Returns: + A validated CPU, memory, and GPU request. + + Raises: + ProjectError: A requirement is invalid or cannot be honored. + """ + if value is None: + return cls() + if not isinstance(value, dict): + raise ProjectError("recipe.resources must be a mapping") + if "time_limit" in value: + raise ProjectError( + "recipe time_limit is not supported; use cluster allocation walltime" + ) + if extra := value.keys() - {"cpus", "memory", "gpus"}: + names = ", ".join(sorted(map(str, extra))) + raise ProjectError(f"unsupported recipe resource requirements: {names}") + parsed = {"cpus": value.get("cpus", 1), "gpus": value.get("gpus", 0)} + if "memory" in value: + parsed["memory_bytes"] = _memory(value["memory"]) + try: + return cls.model_validate(parsed) + except ValidationError as exc: + detail = "; ".join( + f"{'.'.join(map(str, item['loc']))}: {item['msg']}" + for item in exc.errors(include_url=False, include_input=False) + ) + raise ProjectError(f"invalid recipe resources: {detail}") from exc + + def requirements( + self, capacities: set[tuple[float, float, float]], *, whole_worker: bool = False + ) -> dict[str, float]: + """Choose Dask resource reservations that fit an individual worker. + + Args: + capacities: Validated worker budgets from :func:`worker_capacities`. + whole_worker: Reserve a worker's entire CPU, memory, and GPU budget for + an arbitrary command without declared resource requirements. + + Returns: + Dask's numeric ``CPU`` and optional ``MEMORY`` and ``GPU`` reservations. + GPU recipes reserve the worker's full GPU budget, so only one GPU + recipe uses that worker's native device mask at a time. The requested + count is a minimum capacity, not a per-command visibility limit. + + Raises: + ProjectError: Capacity is unknown, a request cannot fit, or an + unspecified budget is ambiguous across heterogeneous workers. + """ + if ( + whole_worker + and len({(cpus, memory) for cpus, memory, _ in capacities}) != 1 + ): + raise ProjectError( + "whole-worker probes require workers with identical CPU and memory budgets" + ) + available_cpus, available_memory, _ = next(iter(capacities)) + requested = {"CPU": available_cpus if whole_worker else float(self.cpus)} + if whole_worker: + requested["MEMORY"] = available_memory + elif self.memory_bytes is not None: + requested["MEMORY"] = float(self.memory_bytes) + matches = { + gpus for cpus, memory, gpus in capacities + if cpus >= requested["CPU"] + and memory >= requested.get("MEMORY", 0) + and gpus >= self.gpus + } + if not matches: + raise ProjectError( + f"task needs {requested['CPU']:g} CPUs and " + f"{requested.get('MEMORY', 0) / 1024**3:g} GiB and {self.gpus} GPUs on one worker; " + "no worker in this cluster can satisfy that request" + ) + if self.gpus or whole_worker: + if len(matches) != 1: + raise ProjectError( + "GPU execution requires matching workers with identical GPU budgets" + ) + if gpus := next(iter(matches)): + requested["GPU"] = gpus + return requested + + +def worker_capacities(workers: dict[str, Any]) -> set[tuple[float, float, float]]: + """Read distinct CPU, memory and GPU budgets once from scheduler information. + + Raises: + ProjectError: No workers are available or their resource budgets are invalid. + """ + capacities: set[tuple[float, float, float]] = set() + for info in workers.values(): + resources = info.get("resources", {}) if isinstance(info, dict) else {} + values = [] + for name in ("CPU", "MEMORY", "GPU"): + value = ( + resources.get(name, 0 if name == "GPU" else None) + if isinstance(resources, dict) else None + ) + if ( + isinstance(value, bool) + or not isinstance(value, (int, float)) + or not math.isfinite(value) + or (value < 0 if name == "GPU" else value <= 0) + or not float(value).is_integer() + ): + raise ProjectError( + "cluster workers must advertise positive whole CPU and MEMORY budgets " + "and a nonnegative whole GPU count; " + "relaunch the cluster with the current Lightcone installation" + ) + values.append(float(value)) + capacities.add((values[0], values[1], values[2])) + if not capacities: + raise ProjectError("cluster has no workers available for execution") + return capacities + + +def _memory(value: object) -> int: + if not isinstance(value, str) or not ( + match := re.fullmatch(r"([0-9]+(?:\.[0-9]+)?)\s*([KMGTPE]i?B?|kB?|B)", value) + ): + raise ProjectError("recipe memory must include units, e.g. 512Mi, 16Gi, or 8GB") + unit = match[2].lower() + exponent = 0 if unit == "b" else "kmgtpe".index(unit[0]) + 1 + factor: int = (1024 if "i" in unit else 1000) ** exponent + try: + return whole_bytes(match[1], factor) + except ValueError as exc: + raise ProjectError(f"recipe {exc}") from exc diff --git a/src/lightcone/engine/materialize.py b/src/lightcone/engine/materialize.py index e9791540..60247ca1 100644 --- a/src/lightcone/engine/materialize.py +++ b/src/lightcone/engine/materialize.py @@ -6,10 +6,10 @@ committed together with the code that produced it, so a run that began with uncommitted changes could not honestly say which code that was. -**It hands the graph to Dask and gets out of the way.** Every task is -submitted with its upstream futures as arguments, so the ordering, the -parallelism, and the scheduling are Dask's — there is no ready-set loop -here to get wrong. +**It hands the work to Dask and gets out of the way.** Tasks that may run +receive upstream futures or already-current results as arguments, so the +ordering, parallelism, and scheduling are Dask's — there is no ready-set +loop here to get wrong. **It owns git, alone.** Workers execute and return; the driver commits, in one thread, as results arrive. That is not a preference: concurrent git @@ -34,7 +34,7 @@ import functools import json import re -from collections.abc import Iterator, Sequence +from collections.abc import Iterable, Iterator, Sequence from contextlib import contextmanager from dataclasses import asdict, dataclass, field, replace from pathlib import Path @@ -42,6 +42,7 @@ from uuid import uuid4 from lightcone.engine import assets, container, dataset, identity, plan, project, worker +from lightcone.engine.execution_resources import TaskResources, worker_capacities from lightcone.engine.plan import Graph, Key, Task from lightcone.engine.project import ProjectError @@ -173,10 +174,31 @@ def _classified( first. """ graph, env_version, _ = _graph(root, targets, report) - versions = assets.Versions() - would_run: set[Key] = set() unfetched: set[str] = set() + classified = _classify_graph( + root, graph, env_version, refresh=refresh, + versions=assets.Versions(), unfetched=unfetched, + ) + if unfetched: + report.warnings.append( + "reported as out of date because their content is not in this " + f"clone, not because they changed: {', '.join(sorted(unfetched))}. " + "`lc materialize` fetches declared inputs before executing " + "recipes. Compute is required for outputs whose inputs cannot yet be checked." + ) + return classified + +def _classify_graph( + root: Path, graph: Graph, env_version: str, *, refresh: bool, + versions: assets.Versions, unfetched: set[str], +) -> list[tuple[Key, assets.Verdict, assets.Manifest | None, dataset.LastWrite | None]]: + """Predict which outputs may run, using the same walk for checks and admission. + + Dependents of an output that may run are conservatively included: only the + worker can know whether rebuilding that input actually changed its bytes. + """ + would_run: set[Key] = set() classified = [] for key in graph.order(): task = graph.tasks[key] @@ -196,13 +218,6 @@ def _classified( would_run.add(key) classified.append((key, verdict, manifest, foreign)) - if unfetched: - report.warnings.append( - "reported as out of date because their content is not in this " - f"clone, not because they changed: {', '.join(sorted(unfetched))}. " - "`lc materialize` fetches declared inputs before it decides " - "anything, so there this resolves itself." - ) return classified @@ -515,12 +530,22 @@ def materialize( # maintainer. Nothing is submitted, so no allocation is needed. _converge_crate(root, report, full, dsid) return report + versions = assets.Versions() + classified = _classify_graph( + root, graph, env_version, refresh=refresh, versions=versions, unfetched=set(), + ) with cluster_for_run(cluster_id) as scheduler: + requirements = scheduler.validate( + graph.tasks[key] for key, verdict, _, _ in classified + if verdict.calls_for_a_remake(refresh=refresh) + ) _fetch_inputs(root, graph, report) # Materialize is one of the two verbs allowed to build the image (the # other is `lc build`); the probe and the rerun entry point only find # one. Resolved once, then handed to every task — the HEAD discipline. - runtime = container.runtime_for_run(root, build=True) + runtime = container.runtime_for_run( + root, build=True, use_gpus=any(r.get("GPU", 0) for r in requirements.values()), + ) # Converge the environment: workers pass `--no-sync`, so this is the # only place on a run's path where it is made to match the lock. (A # rerun does not come through here; its entry point converges too.) @@ -531,7 +556,6 @@ def materialize( # because attestation is a fact about the run (and empty is an # answer, not a failure); one content-hash memo because a declared # input shared by several outputs is the same bytes every time. - versions = assets.Versions() for path in { path for task in graph.tasks.values() @@ -550,39 +574,38 @@ def materialize( runtime=runtime, uv_version=project.uv_version(root), ) - # The history question is the driver's to answer — workers have no - # git, by design — so each task is told up front whether its - # directory was last written by something other than its own run - # record. A foreign write contradicts the manifest, and a worker that - # trusted the recorded digest would skip the output forever. Guarded - # on the manifest's presence, as `_classified` is: without one the - # answer is dead — the output is remade regardless — and each ask is - # a git process. - foreign = { - key: _foreign_write(root, task) if task.manifest_path.is_file() else None - for key, task in graph.tasks.items() - } pending: dict[Key, Any] = {} - # Futures retain dependency ordering; task placement belongs to the - # selected cluster, while commits stay in this one driver thread. - for key in graph.order(): + handles = [] + # Current outputs are values, not Dask tasks: they need no allocation + # resources. Futures for everything that may run retain dependency order. + for key, verdict, manifest, foreign in classified: task = graph.tasks[key] + if not verdict.calls_for_a_remake(refresh=refresh): + assert manifest is not None and verdict.status != "stale" + result = worker.TaskResult( + key, verdict.status, data_version=manifest.data_version, reason=verdict.why, + ) + pending[key] = result + _consume(root, task, result, dsid, runtime, report) + continue pending[key] = scheduler.submit( worker.materialize, root, task, context, refresh, - foreign[key], + foreign, *[pending[dep] for dep in task.depends_on], key=_name(key), + resources=requirements[key], ) + handles.append(pending[key]) from lightcone.engine.compute import UNSTOPPED # An unreported task can still have a running subprocess. Leave its # partial files in place on interruption rather than restoring over it. - outstanding = len(pending) - for result in scheduler.completed(list(pending.values())): + outstanding = len(handles) + for result in scheduler.completed(handles): outstanding -= 1 try: _consume(root, graph.tasks[result.key], result, dsid, runtime, report) @@ -646,13 +669,18 @@ class Scheduler(Protocol): they land, keeping the commit logic independent of the provider. """ - def submit(self, fn: Any, *args: Any, key: str) -> Any: + def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: + """Validate all requests and return their Dask resource reservations.""" + ... + + def submit(self, fn: Any, *args: Any, key: str, resources: dict[str, float]) -> Any: """Schedule a call. Args: fn: The function to run. *args: Its arguments, upstream handles included. key: A display name for the task. + resources: Validated reservations for this task. Returns: A handle to pass to dependents. @@ -678,14 +706,26 @@ class _Dask: client: Any invocation: str output: Forwarder + workers: dict[str, Any] + + def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: + """Require each selected task to fit a worker before any task starts.""" + capacities = worker_capacities(self.workers) + requests = {} + for task in tasks: + try: + requests[task.key] = TaskResources.parse(task.resources).requirements(capacities) + except ProjectError as exc: + raise ProjectError(f"{_name(task.key)}: {exc}") from exc + return requests - def submit(self, fn: Any, *args: Any, key: str) -> Any: + def submit(self, fn: Any, *args: Any, key: str, resources: dict[str, float]) -> Any: """Submit an ordinary Dask task with a unique key and forwarded output.""" from lightcone.engine.compute.output import call return self.client.submit( call, fn, self.output.topic, key, *args, - key=f"lc-{self.invocation}-{key}", pure=False, + key=f"lc-{self.invocation}-{key}", pure=False, resources=resources, ) def completed(self, handles: list[Any]) -> Iterator[worker.TaskResult]: @@ -727,11 +767,11 @@ def cluster_for_run(cluster_id: str) -> Iterator[Scheduler]: with compute.connect(cluster_id) as client: invocation = uuid4().hex with forwarding(client, stdout="stderr") as output: - yield _Dask(client, invocation, output) + yield _Dask(client, invocation, output, client.scheduler_info()["workers"]) def _fetch_inputs(root: Path, graph: Graph, report: MaterializeReport) -> None: - """Bring declared inputs' bytes into this clone before anything hashes. + """Bring declared inputs' bytes into this clone before workers hash or execute. lc fetches rather than telling anyone to — the storage invariant — and only here: ``--check`` and ``status`` are read-only verbs that diff --git a/src/lightcone/engine/plan.py b/src/lightcone/engine/plan.py index 46067016..513f410c 100644 --- a/src/lightcone/engine/plan.py +++ b/src/lightcone/engine/plan.py @@ -23,9 +23,10 @@ from __future__ import annotations -from dataclasses import dataclass +from dataclasses import dataclass, field from graphlib import CycleError, TopologicalSorter from pathlib import Path +from typing import Any from lightcone.engine import assets, identity from lightcone.engine.project import SPEC_FILENAME, ProjectError @@ -51,6 +52,8 @@ class Task: produced_by: dict[str, Key] decisions: dict[str, str] definition_version: str + #: ASTRA's declaration; executor support is checked only when executing. + resources: dict[str, Any] = field(default_factory=dict) @property def manifest_path(self) -> Path: @@ -331,6 +334,7 @@ def file_of(out: object) -> Path: ) except ValueError as e: raise ProjectError(f"output `{out.id}`: {e}") from e + resources = dict((out.definition.get("recipe") or {}).get("resources") or {}) tasks.append( Task( @@ -344,6 +348,7 @@ def file_of(out: object) -> Path: definition_version=identity.definition_version( recipe=recipe, decisions=out.decisions, fmt=str(out.format) ), + resources=resources, ) ) return tasks diff --git a/src/lightcone/engine/run.py b/src/lightcone/engine/run.py index d54b7ed2..cc4cbfb6 100644 --- a/src/lightcone/engine/run.py +++ b/src/lightcone/engine/run.py @@ -21,6 +21,7 @@ from uuid import uuid4 from lightcone.engine import container, sandbox +from lightcone.engine.execution_resources import TaskResources, worker_capacities from lightcone.engine.project import ( SPEC_FILENAME, ProjectError, @@ -51,13 +52,22 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. require_uv() paths = input_paths(project, read_spec(project)) with compute.connect(cluster_id) as client: + resources = TaskResources().requirements( + worker_capacities(client.scheduler_info()["workers"]), whole_worker=True, + ) runtime = container.runtime_for_run(project, build=False) notes = [f"uv: {warning}" for warning in container.converge(runtime)] + use_gpus = resources.get("GPU", 0) > 0 and runtime.supports_gpus + if resources.get("GPU", 0) > 0 and not use_gpus: + notes.append( + f"GPU access is not supported by {runtime.runtime}; this probe runs without GPUs" + ) invocation = uuid4().hex with forwarding(client) as output: future = client.submit( call, _probe, output.topic, "probe", runtime, paths, tuple(command), - key=f"lc-{invocation}-probe", pure=False, + use_gpus, + key=f"lc-{invocation}-probe", pure=False, resources=resources, ) try: outcome: sandbox.Outcome = future.result() @@ -76,10 +86,11 @@ def probe(project: Path, command: Sequence[str], *, cluster_id: str) -> sandbox. def _probe( runtime: container.Runtime, paths: list[Path], command: tuple[str, ...], + use_gpus: bool, *, output: Callable[[str, bytes], None], ) -> sandbox.Outcome: """Execute the prepared probe; the driver alone converges its environment.""" - built = container.policy_for(runtime, paths) + built = container.policy_for(runtime, paths, use_gpus=use_gpus) with sandbox.scope(built) as policy: outcome = sandbox.run( container.backend(runtime), policy, command, cwd=runtime.root, diff --git a/src/lightcone/engine/sandbox/oci.py b/src/lightcone/engine/sandbox/oci.py index 0acbf4e9..ea9f2063 100644 --- a/src/lightcone/engine/sandbox/oci.py +++ b/src/lightcone/engine/sandbox/oci.py @@ -22,6 +22,7 @@ from pathlib import Path from typing import Literal +from lightcone.engine.project import ProjectError from lightcone.engine.sandbox.boundary import SANDBOX_ENV from lightcone.engine.sandbox.model import Attestation, Capability, Policy @@ -30,6 +31,19 @@ OCIRuntime = Literal["podman", "docker", "podman-hpc"] +def supports_gpus(runtime: str) -> bool: + """Whether the runtime can preserve the native allocation's GPU assignment.""" + return runtime == "podman-hpc" + + +def require_gpu_runtime(runtime: str) -> None: + """Refuse runtimes that need device translation outside the native allocation.""" + if not supports_gpus(runtime): + raise ProjectError( + "GPU containers require podman-hpc; Docker/Podman GPUs are not supported" + ) + + @dataclass(frozen=True) class OCIBackend: """A container runtime, expressed as an argv rewrite.""" @@ -81,7 +95,16 @@ def wrap(self, policy: Policy, argv: Sequence[str]) -> list[str]: # host-layout collision `_write_roots` documents for direct mode. mounts = [f"--volume={path.resolve()}:{path}:ro" for path in policy.read] mounts += [f"--volume={path.resolve()}:{path}:rw" for path in policy.write] - overlay = [f"--env={k}={v}" for k, v in sorted(policy.env.items())] + gpu_mask = policy.env.get("CUDA_VISIBLE_DEVICES", "") + environment = dict(policy.env) + if not gpu_mask: + # CUDA images may default to all devices under an NVIDIA runtime. + environment["NVIDIA_VISIBLE_DEVICES"] = "void" + overlay = [f"--env={k}={v}" for k, v in sorted(environment.items())] + gpu_flags = [] + if gpu_mask: + require_gpu_runtime(self.runtime) + gpu_flags = ["--gpu"] return [ self.runtime, "run", "--rm", "--entrypoint", "", @@ -97,6 +120,7 @@ def wrap(self, policy: Policy, argv: Sequence[str]) -> list[str]: # would rewrite the user's own file contexts on disk. "--security-opt", "label=disable", *self.user_flags, + *gpu_flags, *mounts, "--tmpfs", "/tmp:rw,exec", "--shm-size", "1g", diff --git a/src/lightcone/engine/sandbox/policy.py b/src/lightcone/engine/sandbox/policy.py index 0c91ba68..6128af66 100644 --- a/src/lightcone/engine/sandbox/policy.py +++ b/src/lightcone/engine/sandbox/policy.py @@ -32,6 +32,7 @@ from fnmatch import fnmatch from pathlib import Path +from lightcone.engine.project import ProjectError from lightcone.engine.sandbox.model import Policy #: The utility tier of the exec allowlist. A maintained policy @@ -174,6 +175,18 @@ def utility(name: str) -> Path | None: "TMPDIR": ".tmp", } +_DEVICE_ROOT = Path("/dev") + + +def _gpu_device_paths() -> tuple[Path, ...]: + """Grant existing NVIDIA character devices, retaining native OS/cgroup limits.""" + candidates = [ + *(_DEVICE_ROOT / name for name in ("nvidiactl", "nvidia-uvm", "nvidia-uvm-tools")), + *_DEVICE_ROOT.glob("nvidia[0-9]*"), + *(_DEVICE_ROOT / "nvidia-caps").glob("*"), + ] + return tuple(sorted(path for path in candidates if path.is_char_device())) + def exec_policy( project: Path, @@ -182,6 +195,7 @@ def exec_policy( env_dir: Path | None = None, containerized: bool = False, write_dir: Path | None = None, + use_gpus: bool = False, ) -> Policy: """Build what a sandboxed command may touch. @@ -211,6 +225,7 @@ def exec_policy( directory holding its output file, shared with the siblings declared beside it. Absent for a probe, which has no analysis node and gets the project's own ``results/`` whole. + use_gpus: Inherit the allocation's CUDA mask; otherwise hide GPUs. Returns: The policy. The in-tree write scope is granted only if it exists — @@ -219,6 +234,13 @@ def exec_policy( disk; the caller owns removing it (see :func:`~lightcone.engine.sandbox.boundary.scope`). """ + gpu_mask = os.environ.get("CUDA_VISIBLE_DEVICES", "") if use_gpus else "" + if use_gpus and not gpu_mask: + raise ProjectError( + "GPU execution requires a nonempty CUDA_VISIBLE_DEVICES mask. " + "Slurm sets it for GPU jobs; for a local rerun, select your devices explicitly, " + "for example: CUDA_VISIBLE_DEVICES=0 datalad rerun" + ) env_dir = env_dir if env_dir is not None else project / ".venv" # The containerized HOME lives under the project's own (gitignored) # `.lightcone/`, not the system temp dir: it is a mount source, and @@ -234,6 +256,10 @@ def exec_policy( (tmp_home / sub).mkdir(parents=True, exist_ok=True) in_tree_write = write_dir if write_dir is not None else project / "results" + overlay = home_overlay(tmp_home, env_dir, containerized=containerized) + overlay["CUDA_VISIBLE_DEVICES"] = gpu_mask + if use_gpus and "CUDA_DEVICE_ORDER" in os.environ: + overlay["CUDA_DEVICE_ORDER"] = os.environ["CUDA_DEVICE_ORDER"] if containerized: # Declared spellings, not realpaths — the one shape that keeps # its paths unresolved. These become mount *destinations*, and a @@ -246,14 +272,15 @@ def exec_policy( write=_declared([tmp_home, in_tree_write]), execute=(), tmp_home=tmp_home, - env=home_overlay(tmp_home, env_dir, containerized=True), + env=overlay, ) python = _venv_python(env_dir) # EXECUTE on the interpreter *file*; READ on the install root beside # it, for the stdlib. See :func:`_venv_python` and :func:`_stdlib_root`. stdlib = _stdlib_root(python) - write = _existing([tmp_home, in_tree_write, *_write_roots(project)]) + devices = _gpu_device_paths() if use_gpus else () + write = _existing([tmp_home, in_tree_write, *_write_roots(project), *devices]) read = _existing([project, *read_paths, *stdlib, *(Path(p) for p in _OS_READ_BASELINE)]) return Policy( @@ -261,7 +288,7 @@ def exec_policy( write=write, execute=_existing(_exec_set(env_dir, python)), tmp_home=tmp_home, - env=home_overlay(tmp_home, env_dir), + env=overlay, ) diff --git a/src/lightcone/engine/units.py b/src/lightcone/engine/units.py new file mode 100644 index 00000000..0bc1b48f --- /dev/null +++ b/src/lightcone/engine/units.py @@ -0,0 +1,25 @@ +"""Exact byte quantities used by execution and compute.""" + +from __future__ import annotations + +from decimal import Decimal, InvalidOperation + + +def whole_bytes(amount: str, unit_bytes: int) -> int: + """Scale a decimal memory amount into positive bytes without rounding. + + Args: + amount: A decimal quantity in the caller's units. + unit_bytes: The number of bytes represented by one unit. + + Raises: + ValueError: The quantity is not positive or exactly representable in bytes. + """ + try: + numerator, denominator = Decimal(amount).as_integer_ratio() + except (InvalidOperation, ValueError, OverflowError) as exc: + raise ValueError("memory must be positive and exactly representable in bytes") from exc + result, remainder = divmod(numerator * unit_bytes, denominator) + if result <= 0 or remainder: + raise ValueError("memory must be positive and exactly representable in bytes") + return result diff --git a/src/lightcone/engine/worker.py b/src/lightcone/engine/worker.py index af1be88a..c9ffa014 100644 --- a/src/lightcone/engine/worker.py +++ b/src/lightcone/engine/worker.py @@ -36,6 +36,7 @@ from typing import Literal from lightcone.engine import assets, container, dataset, identity, plan, project, sandbox +from lightcone.engine.execution_resources import TaskResources from lightcone.engine.plan import Key, Task from lightcone.engine.project import ( ProjectError, @@ -227,6 +228,7 @@ def execute( ``ok`` with the output's ``data_version``, or ``failed``. Commits nothing and never touches git beyond reading HEAD. """ + resources = TaskResources.parse(task.resources) if moved := _gate(root, context.env_version): return TaskResult(task.key, "failed", reason=moved) @@ -239,17 +241,18 @@ def execute( # cannot reach a sibling, a longer id, another output's sidecar, or a # scope directory of the same name. task.output_path.parent.mkdir(parents=True, exist_ok=True) - task.manifest_path.unlink(missing_ok=True) - for stale in task.output_path.parent.glob(f"{task.output_id}.*"): - if stale.is_file() or stale.is_symlink(): - stale.unlink() - read_paths = [p for p in task.inputs.values() if p.exists()] policy = container.policy_for( - context.runtime, read_paths, write_dir=task.output_path.parent + context.runtime, read_paths, write_dir=task.output_path.parent, + use_gpus=resources.gpus > 0, ) - started_at = _now() with sandbox.scope(policy): + # Validate device visibility and container support before removing outputs. + task.manifest_path.unlink(missing_ok=True) + for stale in task.output_path.parent.glob(f"{task.output_id}.*"): + if stale.is_file() or stale.is_symlink(): + stale.unlink() + started_at = _now() outcome = sandbox.run( container.backend(context.runtime), policy, @@ -414,7 +417,9 @@ def main(argv: list[str]) -> int: # This one-task run resolves its own runtime and HEAD, because it # *is* the driver here — the rule is that each is read once by # whoever owns the run, not that a worker never reads them. - runtime = container.runtime_for_run(root, build=False) + runtime = container.runtime_for_run( + root, build=False, use_gpus=TaskResources.parse(task.resources).gpus > 0, + ) container.converge(runtime) result = execute( root, diff --git a/tests/conftest.py b/tests/conftest.py index 2a4d164a..260e2cc8 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -5,7 +5,7 @@ import shutil import subprocess import textwrap -from collections.abc import Callable, Iterator +from collections.abc import Callable, Iterable, Iterator from contextlib import contextmanager from pathlib import Path from unittest.mock import MagicMock @@ -15,6 +15,7 @@ from lightcone.engine import dataset, project, templates from lightcone.engine.compute.model import Identity +from lightcone.engine.plan import Key, Task from lightcone.engine.project import _run as _real_run CLUSTER_ID = Identity( @@ -151,7 +152,13 @@ class _Inline: are the upstream results themselves, exactly what the worker expects. """ - def submit(self, fn: Callable[..., object], *args: object, key: str) -> object: + def validate(self, tasks: Iterable[Task]) -> dict[Key, dict[str, float]]: + """Run fixture tasks without a finite cluster resource envelope.""" + return {task.key: {} for task in tasks} + + def submit( + self, fn: Callable[..., object], *args: object, key: str, resources: dict[str, float], + ) -> object: return fn(*args) def completed(self, handles: list[object]) -> Iterator[object]: @@ -180,7 +187,8 @@ def cluster_id(monkeypatch: pytest.MonkeyPatch) -> Iterator[str]: from lightcone.engine import compute with LocalCluster( - n_workers=1, threads_per_worker=2, processes=False, dashboard_address=None + n_workers=1, threads_per_worker=2, processes=False, dashboard_address=None, + resources={"CPU": 2, "MEMORY": 1024**3}, ) as cluster: @contextmanager def connect(value: str) -> Iterator[Client]: diff --git a/tests/test_compute.py b/tests/test_compute.py index 7e56ab99..29dfead0 100644 --- a/tests/test_compute.py +++ b/tests/test_compute.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import subprocess from decimal import Decimal, localcontext from pathlib import Path from unittest.mock import MagicMock @@ -18,6 +19,7 @@ from lightcone.engine.compute.catalog import Catalog from lightcone.engine.compute.model import ( GIB, + Accelerator, ComputeError, Connection, Identity, @@ -29,6 +31,7 @@ Startup, TimeLimits, UnavailableOfferError, + duration, memory_bytes, validate_name, ) @@ -364,6 +367,162 @@ def test_memory_conversion_does_not_round_fractional_bytes() -> None: memory_bytes("0.000000000931322574615478515625000000000000000001") +@pytest.mark.parametrize("memory", ["8", "8GB", "8gb", "8192MB", "8GiB", "0.0078125TB"]) +def test_compute_memory_uses_skypilot_binary_units(memory: str) -> None: + assert Request.parse("1", memory + "+").memory_bytes == 8 * GIB + assert Request.parse("1", memory + "+").min_memory + assert Resources.model_validate({"cpus": 1, "memory": memory}).memory_bytes == 8 * GIB + + +def test_recipe_and_compute_memory_keep_their_specification_units() -> None: + from lightcone.engine.execution_resources import TaskResources + + assert Request.parse("1", "8GB").memory_bytes == 8 * GIB + assert TaskResources.parse({"memory": "8GB"}).memory_bytes == 8_000_000_000 + + +def test_compute_memory_rejects_unit_suffix_without_a_size_prefix() -> None: + with pytest.raises(ComputeError, match="memory"): + Request.parse("1", "1IB") + + +@pytest.mark.parametrize("value", ["A100:4", {"A100": 4}]) +def test_accelerator_sky_notations_share_one_model(value: object) -> None: + resource = Resources.model_validate({"cpus": 1, "memory": 1, "accelerators": value}) + assert resource.accelerators == Accelerator(name="A100", count=4) + assert resource.as_dict()["accelerators"] == {"A100": 4} + + +@pytest.mark.parametrize("value", [{"A100": 1, "H100": 1}, ["A100:1", "H100:1"], {"A100:1"}]) +def test_accelerator_alternatives_are_not_silently_treated_as_capacity(value: object) -> None: + with pytest.raises(ValidationError): + Resources.model_validate({"cpus": 1, "memory": 1, "accelerators": value}) + + +@pytest.mark.parametrize("value", [{"count": 2}, {"name": 2}, "count:2", "name:2"]) +def test_accelerator_field_names_are_not_misread_as_device_types(value: object) -> None: + with pytest.raises(ValidationError): + Resources.model_validate({"cpus": 1, "memory": 1, "accelerators": value}) + + +@pytest.mark.parametrize("count", [-1, 0, True, 0.5, "1", None]) +def test_accelerator_envelopes_require_positive_integer_counts(count: object) -> None: + with pytest.raises(ValidationError): + Resources.model_validate({"cpus": 1, "memory": 1, "accelerators": {"A100": count}}) + with pytest.raises(ValidationError): + Request.model_validate({"cpus": 1, "memory_bytes": GIB, "accelerators": {"A100": count}}) + + +@pytest.mark.parametrize("gpus", ["-1", "1++", "", "A100:0.5", "A100:2+"]) +def test_cli_gpu_requests_reject_invalid_accelerator_specifications(gpus: str) -> None: + with pytest.raises(ComputeError, match="accelerators"): + Request.parse("1", "1", gpus=gpus) + + +def test_accelerator_types_and_counts_survive_native_and_request_roundtrips() -> None: + resource = Resources.from_bytes(cpus=8, memory_bytes=16 * GIB, gpus=4, accelerator_name="A100") + assert resource.as_dict() == {"cpus": 8, "memory": 16, "accelerators": {"A100": 4}} + assert Resources.model_validate_json(resource.model_dump_json(by_alias=True)) == resource + assert resource.replace(cpus=4).accelerators == Accelerator(name="A100", count=4) + request = Request.parse("8", "16", gpus="A100:2") + assert request.accelerators == Accelerator(name="A100", count=2) + assert request.as_dict()["resources"]["accelerators"] == {"A100": 2} + assert Request.parse("8", "16", gpus="A100").accelerators == Accelerator(name="A100", count=1) + assert Request.parse("8", "16").accelerators is None + + +def test_numeric_native_accelerator_names_remain_types() -> None: + resource = Resources.from_bytes(cpus=1, memory_bytes=GIB, gpus=2, accelerator_name="4090") + request = Request.parse("1", "1", gpus="4090:2") + assert request.accelerators is not None + assert request.accelerators.matches(resource.accelerators) + assert Request.parse("1", "1", gpus="4090").accelerators == Accelerator(name="4090", count=1) + + +def test_accelerator_selection_honors_type_and_exact_count( + catalog: Path, provider: MagicMock, +) -> None: + data = yaml.safe_load(catalog.read_text()) + cpu = data["offers"][0] + data["offers"].insert(0, { + **cpu, "name": "gpu", "resources": {**cpu["resources"], "accelerators": "A100:4"}, + }) + catalog.write_text(yaml.safe_dump(data)) + service = compute.Compute() + assert service.plan(Request.parse("4", "8")).offer.name == "quick" + assert service.plan(Request.parse("4", "8", gpus="a100:4")).offer.name == "gpu" + assert service.plan(Request.parse("4", "8", gpus="GPU:4")).offer.name == "gpu" + for gpus in ("A100", "H100:4"): + with pytest.raises(ComputeError, match="no configured offer"): + service.plan(Request.parse("4", "8", gpus=gpus)) + data["offers"][0]["resources"]["accelerators"] = "GPU:4" + catalog.write_text(yaml.safe_dump(data)) + with pytest.raises(ComputeError, match="no configured offer"): + compute.Compute().plan(Request.parse("4", "8", gpus="A100:4")) + + +def test_builtin_catalog_stays_cpu_only_without_probing_native_gpus( + default_home: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1") + probe = MagicMock(side_effect=AssertionError("catalog loading must not probe GPU hardware")) + monkeypatch.setattr(subprocess, "run", probe) + loaded = Catalog.load() + assert [(offer.name, offer.resources.gpus) for offer in loaded.offers] == [ + ("local", 0), + ] + probe.assert_not_called() + assert not list(default_home.iterdir()) + + +def test_configured_catalogs_do_not_probe_local_gpus( + catalog: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + probe = MagicMock(side_effect=AssertionError("catalog loading must not probe GPU hardware")) + monkeypatch.setattr(subprocess, "run", probe) + assert Catalog.load().offers + probe.assert_not_called() + + +def test_cli_gpu_request_and_resource_output(catalog: Path, provider: MagicMock) -> None: + data = yaml.safe_load(catalog.read_text()) + data["offers"][0]["resources"]["accelerators"] = {"A100": 2} + catalog.write_text(yaml.safe_dump(data)) + runner = CliRunner() + result = runner.invoke(main, [ + "compute", "launch", "--cpus", "4", "--memory", "8", "--gpus", "A100:2", + "--dry-run", "--json", + ]) + assert result.exit_code == 0, result.output + plan = json.loads(result.output)["plan"] + assert plan["request"]["resources"]["accelerators"] == {"A100": 2} + assert plan["resources"]["accelerators"] == {"A100": 2} + resources = runner.invoke(main, ["compute", "resources", "--json"]) + assert json.loads(resources.output)["units"]["accelerators"] == "type and count per node" + rendered = runner.invoke(main, ["compute", "resources"]).output + assert "GPUS" in rendered and "A100:2" in rendered + + +@pytest.mark.parametrize( + ("value", "seconds"), + [("1h30m", 5400), ("45s", 45), ("2d3h4m5s", 183845)], +) +def test_allocation_durations_accept_compound_units(value: str, seconds: int) -> None: + assert duration(value) == seconds + assert TimeLimits(default=value, max="3d").default_seconds == seconds + assert Request.parse("1", "1", time=value).seconds == seconds + + +@pytest.mark.parametrize("value", ["", "0s", "1.5h", "30m1h", "1h30", 60, True]) +def test_allocation_duration_refuses_ambiguous_or_zero_values(value: object) -> None: + with pytest.raises(ValueError): + duration(value) + with pytest.raises(ValidationError): + TimeLimits(default=value, max="3d") + with pytest.raises(ComputeError): + Request.parse("1", "1", time=value) + + @pytest.mark.parametrize("size", [1, GIB // 2, 8 * GIB, 2**80 + 1]) def test_resource_units_survive_construction_serialization_and_updates(size: int) -> None: whole, fraction = divmod(size, GIB) @@ -410,7 +569,9 @@ def test_catalog_uses_the_public_models_and_roundtrips_without_an_adapter(catalo assert type(instance) is model assert "name" not in loaded.connections["test"].model_dump() dumped = loaded.model_dump(by_alias=True) - assert dumped["offers"][0]["resources"] == {"cpus": 4, "memory": Decimal(8)} + assert dumped["offers"][0]["resources"] == { + "cpus": 4, "memory": Decimal(8), "accelerators": None, + } assert dumped["offers"][0]["time"] == {"default": "30m", "max": "2h"} assert Catalog.model_validate(dumped) == loaded assert Catalog.model_validate_json(loaded.model_dump_json(by_alias=True)) == loaded @@ -717,14 +878,14 @@ def test_cli_resources_dry_run_launch_down(catalog: Path, provider: MagicMock) - assert result.exit_code == 0, result.output resources = json.loads(result.output) assert [item["name"] for item in resources["offers"]] == ["quick", "large"] - assert resources["offers"][0]["resources"] == {"cpus": 4, "memory": 8} + assert resources["offers"][0]["resources"] == {"cpus": 4, "memory": 8, "accelerators": None} args = ["compute", "launch", "--cpus", "4", "--memory", "8", "--json"] result = runner.invoke(main, [*args, "--dry-run"]) assert result.exit_code == 0, result.output plan = json.loads(result.output)["plan"] assert plan["offer"] == "quick" assert plan["connection"] == "test" - assert plan["resources"] == {"cpus": 4, "memory": 8} + assert plan["resources"] == {"cpus": 4, "memory": 8, "accelerators": None} assert plan["time_seconds"] == 1800 assert plan["startup"] == "fast" assert "memory_gib" not in result.output @@ -737,6 +898,16 @@ def test_cli_resources_dry_run_launch_down(catalog: Path, provider: MagicMock) - provider.terminate.assert_called_once_with(IDENTITY) +def test_cli_resources_preserves_seconds(catalog: Path) -> None: + data = yaml.safe_load(catalog.read_text()) + data["offers"][0]["time"] = {"default": "45s", "max": "1m30s"} + catalog.write_text(yaml.safe_dump(data)) + result = CliRunner().invoke(main, ["compute", "resources"]) + assert result.exit_code == 0, result.output + assert "45s" in result.output + assert "1m30s" in result.output + + def test_cli_launch_name_output_can_be_captured_without_json( catalog: Path, provider: MagicMock, ) -> None: diff --git a/tests/test_compute_local.py b/tests/test_compute_local.py index 0da9d92a..c50ecf17 100644 --- a/tests/test_compute_local.py +++ b/tests/test_compute_local.py @@ -15,11 +15,14 @@ from pathlib import Path from types import SimpleNamespace from typing import Any +from unittest.mock import MagicMock from uuid import uuid4 import psutil import pytest +from lightcone.engine.compute import Compute, local, local_runtime +from lightcone.engine.compute.catalog import Catalog from lightcone.engine.compute.local import LocalProvider from lightcone.engine.compute.model import ( ComputeError, @@ -223,6 +226,51 @@ def expand(path): assert Compute().discover() == ([], {}) +def test_cpu_allocation_without_gpu_fields_remains_discoverable_and_stoppable( + provider: LocalProvider, +) -> None: + identity = _launch(provider) + try: + _ready(provider, identity) + path = provider.root / identity.token / "identity.json" + record = read_private_json(path) + del record["gpus"], record["accelerator_name"] + write_private_json(path, record) + + snapshot, = provider.discover() + assert snapshot.identity == identity + assert snapshot.resources is not None + assert snapshot.resources.gpus == 0 + assert snapshot.resources.accelerator_name is None + with provider.connect(identity) as client: + assert client.submit(sum, [1, 2]).result(timeout=5) == 3 + finally: + provider.terminate(identity) + _ended(provider, identity) + + +@pytest.mark.parametrize("threads", [None, "2"]) +def test_local_workers_keep_native_dask_thread_defaults_and_launch_overrides( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, threads: str | None, +) -> None: + names = ("OMP_NUM_THREADS", "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS") + for name in names: + if threads is None: + monkeypatch.delenv(name, raising=False) + else: + monkeypatch.setenv(name, threads) + identity = _launch(provider) + try: + _ready(provider, identity) + with provider.connect(identity) as client: + actual = client.submit(lambda: {name: os.environ.get(name) for name in names}).result( + timeout=5, + ) + assert actual == dict.fromkeys(names, threads or "1") + finally: + provider.terminate(identity) + + def test_named_local_allocation_is_discovered_and_name_can_be_reused_after_down( provider: LocalProvider, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -807,3 +855,148 @@ def test_local_plan_does_not_infer_policy_from_login_hostname_or_slurm_environme plan = provider.plan(offer, Request(cpus=1, memory_bytes=512 * 1024**2)) assert plan.resources == offer.resources assert not provider.root.exists() + + +@pytest.mark.parametrize("gpus", [0, 1]) +@pytest.mark.parametrize("order", [None, "PCI_BUS_ID", "FASTEST_FIRST"]) +def test_local_gpu_plan_freezes_native_mask_and_publishes_the_configured_envelope( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, gpus: int, + order: str | None, +) -> None: + monkeypatch.setattr(sys, "platform", "linux") + # Mask interpretation and agreement with the catalog belong to the operator. + mask = "GPU-opaque,3" + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) + if order is None: + monkeypatch.delenv("CUDA_DEVICE_ORDER", raising=False) + else: + monkeypatch.setenv("CUDA_DEVICE_ORDER", order) + probe = MagicMock(side_effect=AssertionError("local GPU planning must not probe hardware")) + monkeypatch.setattr(local.subprocess, "run", probe) + offer = Offer( + name="gpu", connection="workstation", + resources=Resources.from_bytes( + cpus=1, memory_bytes=512 * 1024**2, gpus=gpus, accelerator_name="A100", + ), + max_nodes=1, time=TimeLimits(default="1m", max="1m"), + ) + plan = provider.plan(offer, Request.parse("1", "0.5", gpus=f"GPU:{gpus}" if gpus else "0")) + assert plan.details["cuda_visible_devices"] == (mask if gpus else "") + assert plan.details["cuda_device_order"] == order + assert not provider.root.exists() + monkeypatch.setattr(local, "_boot_identity", lambda: str(uuid4())) + popen = MagicMock(return_value=SimpleNamespace(pid=12345)) + monkeypatch.setattr(local.subprocess, "Popen", popen) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "an-ambient-mask") + monkeypatch.setenv("CUDA_DEVICE_ORDER", "changed-after-planning") + + identity = provider.launch(plan) + + assert popen.call_args.kwargs["env"]["CUDA_VISIBLE_DEVICES"] == (mask if gpus else "") + assert popen.call_args.kwargs["env"].get("CUDA_DEVICE_ORDER") == order + assert os.environ["CUDA_VISIBLE_DEVICES"] == "an-ambient-mask" + assert os.environ["CUDA_DEVICE_ORDER"] == "changed-after-planning" + record = read_private_json(provider.root / identity.token / "identity.json") + assert record["gpus"] == gpus + monkeypatch.setattr(provider, "_process", lambda *_: None) + snapshot = provider.inspect(identity) + assert snapshot.resources is not None and snapshot.resources.gpus == gpus + assert snapshot.resources.accelerator_name == ("A100" if gpus else None) + probe.assert_not_called() + + +@pytest.mark.parametrize("mask", [None, ""]) +def test_local_gpu_offers_require_an_explicit_nonempty_native_mask( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, + mask: str | None, +) -> None: + monkeypatch.setattr(sys, "platform", "linux") + if mask is None: + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + else: + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) + offer = Offer( + name="gpu", connection="workstation", + resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=1), + max_nodes=1, time=TimeLimits(default="1m", max="1m"), + ) + request = Request.parse("1", "0.5", gpus="GPU:1") + with pytest.raises(ComputeError, match="nonempty CUDA_VISIBLE_DEVICES"): + provider.plan(offer, request) + assert not provider.root.exists() + + +def test_unconfigured_local_gpu_mask_does_not_hide_a_later_slurm_offer( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + local_offer = Offer( + name="local-gpu", connection="workstation", + resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=1), + max_nodes=1, time=TimeLimits(default="1m", max="1m"), + ) + service = Compute.__new__(Compute) + service.catalog = Catalog( + version=1, + connections={ + "workstation": provider.connection, + "hpc": Connection(namespace=str(uuid4()), provider="slurm"), + }, + offers=[local_offer, local_offer.replace(name="batch-gpu", connection="hpc")], + ) + plan = service.plan(Request.parse("1", "0.5", gpus="GPU:1")) + assert plan.offer.name == "batch-gpu" + assert not provider.root.exists() + + +def test_local_gpu_offers_are_unavailable_outside_linux( + provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(sys, "platform", "darwin") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") + offer = Offer( + name="gpu", connection="workstation", + resources=Resources.from_bytes( + cpus=1, memory_bytes=512 * 1024**2, gpus=1, + ), + max_nodes=1, time=TimeLimits(default="1m", max="1m"), + ) + with pytest.raises(ComputeError, match="GPU allocations require Linux"): + provider.plan(offer, Request.parse("1", "0.5", gpus="GPU:1")) + + +@pytest.mark.parametrize("gpus", [None, 0, 2]) +def test_local_runtime_advertises_configured_resources_and_preserves_native_gpu_mask( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, gpus: int | None, +) -> None: + import distributed + + directory = private_directory(tmp_path / "allocation", create=True) + scratch = private_directory(tmp_path / "scratch", create=True) + write_private_json(directory / "launch.json", { + "identity": "allocation", "deadline": time.monotonic() + 60, + "task_slots": 1, "scratch": str(scratch), + }) + record = {"cpus": 1, "memory": 1024**3} + if gpus is not None: + record["gpus"] = gpus + write_private_json(directory / "identity.json", record) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "3,1") + monkeypatch.setattr(sys, "argv", ["local_runtime", str(directory)]) + monkeypatch.setattr(os, "getsid", lambda _: os.getpid()) + monkeypatch.setattr(os, "getpgrp", os.getpid) + monkeypatch.setattr(os, "umask", lambda _: 0) + monkeypatch.setattr(local_runtime.atexit, "register", lambda *_: None) + monkeypatch.setattr(signal, "setitimer", lambda *_: None) + monkeypatch.setattr( + signal, "signal", lambda signum, handler: handler(signum, None) + if signum == signal.SIGTERM else None, + ) + monkeypatch.setattr(local_runtime, "create_security", lambda _: None) + cluster = MagicMock() + cluster.return_value.__enter__.return_value.scheduler.id = "Scheduler-gpu" + monkeypatch.setattr(distributed, "LocalCluster", cluster) + + local_runtime.main() + assert cluster.call_args.kwargs["resources"] == {"CPU": 1, "MEMORY": 1024**3, "GPU": gpus or 0} + assert os.environ["CUDA_VISIBLE_DEVICES"] == ("3,1" if gpus else "") diff --git a/tests/test_compute_slurm.py b/tests/test_compute_slurm.py index 7d4b0fa9..d51e40ec 100644 --- a/tests/test_compute_slurm.py +++ b/tests/test_compute_slurm.py @@ -13,7 +13,7 @@ from pathlib import Path from types import SimpleNamespace from typing import Any -from unittest.mock import MagicMock +from unittest.mock import AsyncMock, MagicMock import psutil import pytest @@ -96,6 +96,7 @@ def _control( f"JobId=123 JobName=lc-v1-{name} UserId=alice({owner}) JobState={state}\n" f" Comment={comment} \n" f" NumNodes=2 NumCPUs=512 CPUs/Task=256 MinMemoryNode=480G Restarts={restarts} " + "ReqTRES=cpu=512,mem=960G,node=2 " "SubmitTime=2026-09-27T10:00:00 StartTime=2026-09-27T10:00:05 Reason=None\n" ) @@ -169,6 +170,59 @@ def test_plan_preserves_native_envelope_without_native_queries( assert calls == [] +@pytest.mark.parametrize("submit", ["sbatch", "salloc"]) +def test_gpu_plan_requests_per_node_devices_for_the_allocation_and_step( + provider: slurm.SlurmProvider, offer: Offer, submit: str, +) -> None: + offer = offer.replace( + resources=offer.resources.replace(accelerators={"GPU": 4}), + config={**offer.config, "submit": submit, "constraint": "gpu"}, + ) + plan = provider.plan(offer, Request.parse("256", "480", gpus="GPU:4", num_nodes=2)) + assert "--gres=gpu:4" in plan.details["native_args"] + assert "--constraint=gpu" in plan.details["native_args"] + payload = provider._payload(plan, TOKEN) + assert "--gres=gpu:4" in payload + assert "--ntasks-per-node=1" in payload + assert payload[payload.index("--gpus") + 1] == "4" + + +def test_named_accelerator_uses_explicit_native_gres_mapping( + provider: slurm.SlurmProvider, offer: Offer, +) -> None: + offer = offer.replace(resources=offer.resources.replace(accelerators={"A100": 4})) + request = Request.parse("256", "480", gpus="A100:4", num_nodes=2) + with pytest.raises(ComputeError, match="requires its native gpu_type"): + provider.plan(offer, request) + offer = offer.replace(config={**offer.config, "gpu_type": "a100_80gb"}) + plan = provider.plan(offer, request) + assert "--gres=gpu:a100_80gb:4" in plan.details["native_args"] + assert "--gres=gpu:a100_80gb:4" in provider._payload(plan, TOKEN) + assert plan.resources.accelerator_name == "A100" + + +@pytest.mark.parametrize("gpu_type", ["", "a100:4", "a100,v100", "two types", 4]) +def test_slurm_gpu_type_must_be_one_native_type( + provider: slurm.SlurmProvider, offer: Offer, gpu_type: object, +) -> None: + offer = offer.replace( + resources=offer.resources.replace(accelerators={"A100": 4}), + config={**offer.config, "gpu_type": gpu_type}, + ) + with pytest.raises(ComputeError, match="one native GPU GRES type"): + provider.plan(offer, Request.parse("256", "480", gpus="A100:4")) + + +def test_cpu_offer_cannot_request_a_gpu_type( + provider: slurm.SlurmProvider, offer: Offer, +) -> None: + with pytest.raises(ComputeError, match="requires accelerator resources"): + provider.plan( + offer.replace(config={**offer.config, "gpu_type": "a100"}), + Request.parse("256", "480"), + ) + + def test_default_launch_assumes_a_shared_home_and_node_local_scratch( offer: Offer, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -696,6 +750,43 @@ def test_discovery_preserves_allocations_with_unknown_native_resource_evidence( assert snapshot.evidence == "unknown" +@pytest.mark.parametrize("native,expected,name", [ + ("TresPerNode=gres/gpu:4", 4, "GPU"), + ("TresPerNode=gres:gpu:4", 4, "GPU"), + ("TresPerNode=gpu:4", 4, "GPU"), + ("TresPerNode=gres/gpu:a100:4", 4, "a100"), + ("TresPerNode=gres:gpu:a100:4", 4, "a100"), + ("TresPerNode=gres/gpu:4090:2", 2, "4090"), + ("TresPerNode=gres/gpu:a100:2,gres/gpu:v100:2", 4, "GPU"), + ("Gres=gpu:a100:2", 2, "a100"), + ("Gres=(null)", 0, None), + ("ReqTRES=cpu=512,mem=960G,node=2", 0, None), + ("TRES=cpu=512,mem=960G,node=2", 0, None), + ("ReqTRES=cpu=512,mem=960G,node=2,gres/gpu=8", None, None), + ("TRES=cpu=512,mem=960G,node=2,gres/gpu=8", None, None), + ("TresPerNode=gres/gpu:unknown", None, None), + ("TresPerNode=gres:gpu:unknown TRES=cpu=512,mem=960G,node=2", None, None), + ("TresPerNode=gres/gpu:a100/80gb:2", None, None), + ("TresPerNode=gres/gpu:1.5", None, None), + ("TresPerNode=gres/gpu:4,gres/gpu:a100:4", None, None), + ("", None, None), +]) +def test_gpu_discovery_reports_only_native_per_node_evidence( + provider: slurm.SlurmProvider, monkeypatch: pytest.MonkeyPatch, + native: str, expected: int | None, name: str | None, +) -> None: + control = _control().replace("ReqTRES=cpu=512,mem=960G,node=2", native) + _native(monkeypatch, {"squeue": _live(), "scontrol": control}) + snapshot, = provider.discover() + if expected is None: + assert snapshot.resources is None + assert snapshot.evidence == "unknown" + else: + assert snapshot.resources is not None and snapshot.resources.gpus == expected + assert snapshot.resources.accelerator_name == name + assert snapshot.evidence == "requested" + + def test_native_query_failure_is_not_an_empty_list( provider: slurm.SlurmProvider, monkeypatch: pytest.MonkeyPatch ) -> None: @@ -909,6 +1000,7 @@ def _bootstrap_args(tmp_path: Path) -> argparse.Namespace: num_nodes=2, cpus=1, memory_bytes=64 * 1024**2, + gpus=0, task_slots=1, interface=None, ) @@ -916,6 +1008,7 @@ def _bootstrap_args(tmp_path: Path) -> argparse.Namespace: def _bootstrap_env(rank: int) -> dict[str, str]: return { + "CUDA_VISIBLE_DEVICES": "", "SLURM_JOB_ID": "123", "SLURM_PROCID": str(rank), "SLURM_NTASKS": "2", @@ -936,14 +1029,98 @@ def test_bootstrap_refuses_mismatched_native_envelope( slurm_bootstrap._allocation(_bootstrap_args(tmp_path)) +@pytest.mark.parametrize("native,valid", [ + ("", False), ("unknown", False), ("1", False), ("2", True), ("4", True), +]) +def test_gpu_bootstrap_requires_native_capacity_without_probing_or_rewriting_the_mask( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, + native: str, valid: bool, +) -> None: + for key, value in _bootstrap_env(0).items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("SLURM_GPUS_ON_NODE", native) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "1,3") + monkeypatch.setenv("CUDA_DEVICE_ORDER", "FASTEST_FIRST") + + probe = MagicMock(side_effect=AssertionError("Slurm bootstrap must not probe CUDA")) + monkeypatch.setattr(subprocess, "run", probe) + args = _bootstrap_args(tmp_path) + args.gpus = 2 + if valid: + _, _, rank = slurm_bootstrap._allocation(args) + assert rank == 0 + else: + with pytest.raises(ComputeError, match="GPUs"): + slurm_bootstrap._allocation(args) + assert os.environ["CUDA_VISIBLE_DEVICES"] == "1,3" + assert os.environ["CUDA_DEVICE_ORDER"] == "FASTEST_FIRST" + probe.assert_not_called() + + +@pytest.mark.parametrize("mask", [None, ""]) +def test_gpu_bootstrap_requires_a_nonempty_native_mask( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, mask: str | None, +) -> None: + for key, value in _bootstrap_env(0).items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("SLURM_GPUS_ON_NODE", "2") + if mask is None: + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + else: + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) + args = _bootstrap_args(tmp_path) + args.gpus = 2 + with pytest.raises(ComputeError, match="native CUDA_VISIBLE_DEVICES mask"): + slurm_bootstrap._allocation(args) + + +def test_gpu_worker_advertises_native_capacity_with_the_native_mask( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + import distributed + + args = _bootstrap_args(tmp_path) + args.gpus = 2 + for key, value in _bootstrap_env(1).items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("SLURM_GPUS_ON_NODE", "2") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "1,3") + monkeypatch.setenv("CUDA_DEVICE_ORDER", "FASTEST_FIRST") + connection = Connection( + namespace=NAMESPACE, provider="slurm", launch={"connection_root": args.connection_root}, + ) + directory = private_directory(slurm.attempt_directory(connection, IDENTITY, 0), create=True) + write_private_json(directory / "identity.json", { + "namespace": NAMESPACE, "native_id": "123", "token": TOKEN, "uid": os.getuid(), + "restarts": 0, "num_nodes": 2, "cpus": 1, "memory_bytes": args.memory_bytes, + "gpus": 2, "task_slots": 1, + }) + monkeypatch.setattr(slurm_bootstrap, "load_security", lambda _: None) + worker = MagicMock() + worker.__aenter__ = AsyncMock(return_value=worker) + worker.finished = AsyncMock() + factory = MagicMock(return_value=worker) + monkeypatch.setattr(distributed, "Worker", factory) + + asyncio.run(slurm_bootstrap.run(args)) + + assert factory.call_args.kwargs["resources"] == { + "CPU": 1, "MEMORY": args.memory_bytes, "GPU": 2, + } + assert os.environ["CUDA_VISIBLE_DEVICES"] == "1,3" + assert os.environ["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" + + def test_worker_rendezvous_has_a_finite_deadline( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: for key, value in _bootstrap_env(1).items(): monkeypatch.setenv(key, value) monkeypatch.setattr(slurm_bootstrap, "_STARTUP_TIMEOUT", 0) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "an-ambient-GPU") with pytest.raises(ComputeError, match="timed out"): asyncio.run(slurm_bootstrap.run(_bootstrap_args(tmp_path))) + assert os.environ["CUDA_VISIBLE_DEVICES"] == "" def test_bootstrap_defaults_scratch_to_the_node_temporary_directory( @@ -997,7 +1174,8 @@ def test_standard_bootstrap_starts_scheduler_and_worker_on_rank_zero_and_worker_ ) argv = [sys.executable, "-P", "-m", "lightcone.engine.compute.slurm_bootstrap"] for key, value in vars(args).items(): - if value is not None: + # CPU-only submissions can omit the optional GPU count. + if value is not None and key != "gpus": argv += ["--" + key.replace("_", "-"), str(value)] connection = Connection( namespace=NAMESPACE, provider="slurm", launch={"connection_root": args.connection_root} @@ -1032,6 +1210,7 @@ def test_standard_bootstrap_starts_scheduler_and_worker_on_rank_zero_and_worker_ workers = client.scheduler_info()["workers"] assert {worker["name"] for worker in workers.values()} == {"lightcone-0", "lightcone-1"} assert all(worker["nthreads"] == 1 for worker in workers.values()) + assert all(worker["resources"]["GPU"] == 0 for worker in workers.values()) assert client.submit(sum, [2, 3]).result(timeout=5) == 5 assert client.scheduler_info()["address"].startswith("tls://127.0.0.1:") assert all(address.startswith("tls://127.0.0.1:") for address in workers) diff --git a/tests/test_container.py b/tests/test_container.py index dc7a8f6a..fd28ed6a 100644 --- a/tests/test_container.py +++ b/tests/test_container.py @@ -193,6 +193,28 @@ def test_a_missing_archive_refuses_unless_the_caller_may_build( assert container.runtime_for_run(root, build=False).image_id == runtime.image_id +@pytest.mark.parametrize("runtime", ["docker", "podman"]) +@pytest.mark.parametrize("build", [False, True]) +def test_unsupported_gpu_runtime_refuses_before_any_image_preparation( + root: Path, fake: list[list[str]], monkeypatch: pytest.MonkeyPatch, + runtime: str, build: bool, +) -> None: + monkeypatch.setattr(container, "runtime_name", lambda _: runtime) + with pytest.raises(ProjectError, match="GPU containers require podman-hpc"): + container.runtime_for_run(root, build=build, use_gpus=True) + assert fake == [] + assert not (root / ".datalad").exists() + + +def test_supported_gpu_runtime_can_prepare_the_image_without_a_driver_gpu_mask( + root: Path, hpc: list[list[str]], monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + runtime = container.runtime_for_run(root, build=True, use_gpus=True) + assert runtime.supports_gpus + assert _argvs(hpc, "podman-hpc", "build") + + def test_unfetched_archive_content_is_fetched_by_lc_itself( root: Path, fake: list[list[str]], monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/tests/test_execution_resources.py b/tests/test_execution_resources.py new file mode 100644 index 00000000..3bc1feeb --- /dev/null +++ b/tests/test_execution_resources.py @@ -0,0 +1,203 @@ +"""Parse recipe requests and refuse impossible execution before submitting work.""" + +from __future__ import annotations + +from typing import Any + +import pytest +from pydantic import ValidationError + +from lightcone.engine.execution_resources import TaskResources, worker_capacities +from lightcone.engine.project import ProjectError + +GIB = 1024**3 + + +def _workers(*capacities: tuple[int, int]) -> dict[str, Any]: + return { + f"worker-{index}": {"resources": {"CPU": cpus, "MEMORY": memory}} + for index, (cpus, memory) in enumerate(capacities) + } + + +@pytest.mark.parametrize( + ("memory", "expected"), + [("512Mi", 512 * 1024**2), ("1.5GiB", 3 * GIB // 2), ("8GB", 8_000_000_000), + ("1B", 1), ("2 Ti", 2 * 1024**4), ("1000kB", 1_000_000)], +) +def test_memory_units_have_explicit_decimal_or_binary_meaning(memory: str, expected: int) -> None: + assert TaskResources.parse({"memory": memory}).memory_bytes == expected + + +@pytest.mark.parametrize("duration", ["1h30m", "30m", "45s", None]) +def test_recipe_time_limit_is_explicitly_refused(duration: object) -> None: + with pytest.raises(ProjectError, match="recipe time_limit is not supported"): + TaskResources.parse({"time_limit": duration}) + + +def test_integral_astra_float_cpu_count_is_accepted_without_rounding() -> None: + assert TaskResources.parse({"cpus": 4.0}).cpus == 4 + with pytest.raises(ProjectError, match="fractional CPUs"): + TaskResources.parse({"cpus": 0.5}) + + +@pytest.mark.parametrize( + "declaration", + [ + {"cpus": 0}, {"cpus": True}, {"cpus": "4"}, {"cpus": None}, + {"memory": "0Gi"}, {"memory": "0.1B"}, {"memory": "16"}, {"memory": 16}, + {"memory": "400m"}, {"memory": None}, + {"time_limit": "0m"}, {"time_limit": ""}, {"time_limit": "5m2h"}, + {"time_limit": "unlimited"}, {"disk": "10Gi"}, {"ram": "1Gi"}, + {"gpus": -1}, {"gpus": True}, {"gpus": 0.5}, {"gpus": "1"}, {"gpus": None}, + ], +) +def test_invalid_or_unhonored_declarations_are_not_silently_ignored( + declaration: dict[str, Any], +) -> None: + with pytest.raises(ProjectError): + TaskResources.parse(declaration) + + +def test_internal_resource_models_remain_validated() -> None: + with pytest.raises(ValidationError): + TaskResources(memory_bytes=-1) + with pytest.raises(ValidationError): + TaskResources(gpus=-1) + + +def test_recipe_memory_uses_exact_bytes_without_decimal_context_rounding() -> None: + one_byte = "0.000000000931322574615478515625" + assert TaskResources.parse({"memory": f"{one_byte}Gi"}).memory_bytes == 1 + with pytest.raises(ProjectError, match="exactly representable"): + TaskResources.parse({"memory": f"{one_byte}00000000000000001Gi"}) + + +def test_declared_requests_reserve_exact_cpu_and_memory_budgets() -> None: + task = TaskResources.parse({"cpus": 4, "memory": "6Gi"}) + assert task.requirements(worker_capacities(_workers((8, 16 * GIB)))) == { + "CPU": 4, "MEMORY": 6 * GIB, + } + + +def test_missing_memory_does_not_reserve_a_budget() -> None: + capacities = worker_capacities(_workers((8, 16 * GIB), (8, 16 * GIB))) + assert TaskResources().requirements(capacities) == {"CPU": 1} + + +def test_probe_reserves_an_entire_worker() -> None: + capacities = worker_capacities(_workers((8, 16 * GIB))) + assert TaskResources().requirements(capacities, whole_worker=True) == { + "CPU": 8, "MEMORY": 16 * GIB, + } + + +def test_cpu_count_is_independent_of_dask_execution_threads() -> None: + workers = _workers((8, 16 * GIB)) + workers["worker-0"]["nthreads"] = 1 + assert TaskResources(cpus=8).requirements(worker_capacities(workers))["CPU"] == 8 + + +def test_a_task_must_fit_one_worker_not_the_sum_of_the_cluster() -> None: + with pytest.raises(ProjectError, match="on one worker"): + TaskResources(cpus=8, memory_bytes=20 * GIB).requirements( + worker_capacities(_workers((4, 16 * GIB), (4, 16 * GIB))) + ) + + +def test_cpu_and_memory_must_fit_on_the_same_worker() -> None: + with pytest.raises(ProjectError, match="no worker"): + TaskResources(cpus=8, memory_bytes=16 * GIB).requirements( + worker_capacities(_workers((8, 4 * GIB), (4, 16 * GIB))) + ) + + +def test_explicit_requests_can_select_a_fitting_worker() -> None: + assert TaskResources(cpus=8, memory_bytes=8 * GIB).requirements( + worker_capacities(_workers((4, 4 * GIB), (8, 16 * GIB))) + ) == {"CPU": 8, "MEMORY": 8 * GIB} + + +def test_probes_require_identical_worker_budgets() -> None: + with pytest.raises(ProjectError, match="identical"): + TaskResources().requirements( + worker_capacities(_workers((4, 4 * GIB), (8, 16 * GIB))), whole_worker=True + ) + + +@pytest.mark.parametrize( + "workers", + [{}, {"worker": {}}, {"worker": {"resources": {"CPU": 1}}}, + {"worker": {"resources": {"CPU": True, "MEMORY": GIB}}}, + {"worker": {"resources": {"CPU": 1, "MEMORY": float("nan")}}}], +) +def test_absent_or_unknown_worker_capacity_refuses_execution(workers: dict[str, Any]) -> None: + with pytest.raises(ProjectError): + TaskResources().requirements(worker_capacities(workers)) + + +def test_gpu_recipe_reserves_the_whole_worker_gpu_budget() -> None: + workers = _workers((8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 4 + task = TaskResources.parse({"cpus": 2, "memory": "4Gi", "gpus": 1}) + assert task.requirements(worker_capacities(workers)) == {"CPU": 2, "MEMORY": 4 * GIB, "GPU": 4} + + +def test_gpu_request_must_fit_one_worker() -> None: + workers = _workers((8, 16 * GIB), (8, 16 * GIB)) + for worker in workers.values(): + worker["resources"]["GPU"] = 1 + with pytest.raises(ProjectError, match="2 GPUs on one worker"): + TaskResources(gpus=2).requirements(worker_capacities(workers)) + + +def test_cpu_workers_cannot_satisfy_gpu_requests() -> None: + with pytest.raises(ProjectError, match="1 GPUs on one worker"): + TaskResources(gpus=1).requirements(worker_capacities(_workers((8, 16 * GIB)))) + + +def test_cpu_recipe_does_not_reserve_gpu_capacity() -> None: + workers = _workers((8, 16 * GIB), (8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 4 + assert TaskResources().requirements(worker_capacities(workers)) == {"CPU": 1} + + +def test_probe_reserves_cpu_memory_and_all_gpus() -> None: + workers = _workers((8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 4 + assert TaskResources().requirements(worker_capacities(workers), whole_worker=True) == { + "CPU": 8, "MEMORY": 16 * GIB, "GPU": 4, + } + + +def test_gpu_requirements_ignore_workers_that_cannot_fit_the_recipe() -> None: + workers = _workers((2, 2 * GIB), (8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 1 + workers["worker-1"]["resources"]["GPU"] = 4 + task = TaskResources(cpus=4, memory_bytes=8 * GIB, gpus=1) + assert task.requirements(worker_capacities(workers)) == { + "CPU": 4, "MEMORY": 8 * GIB, "GPU": 4, + } + + +@pytest.mark.parametrize("whole_worker", [False, True]) +def test_whole_gpu_reservations_refuse_ambiguous_worker_budgets(whole_worker: bool) -> None: + workers = _workers((8, 16 * GIB), (8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = 1 + workers["worker-1"]["resources"]["GPU"] = 4 + with pytest.raises(ProjectError, match="identical GPU budgets"): + TaskResources(gpus=1).requirements(worker_capacities(workers), whole_worker=whole_worker) + + +@pytest.mark.parametrize("capacity", [-1, 0.5, True, "1", None, float("nan"), float("inf")]) +def test_malformed_gpu_capacity_is_never_ignored(capacity: object) -> None: + workers = _workers((8, 16 * GIB)) + workers["worker-0"]["resources"]["GPU"] = capacity + with pytest.raises(ProjectError, match="GPU count"): + TaskResources().requirements(worker_capacities(workers)) + + +def test_undeclared_memory_accepts_heterogeneous_workers() -> None: + assert TaskResources(cpus=4).requirements( + worker_capacities(_workers((4, 4 * GIB), (8, 16 * GIB))) + ) == {"CPU": 4} diff --git a/tests/test_gpu_execution.py b/tests/test_gpu_execution.py new file mode 100644 index 00000000..4ecf70b3 --- /dev/null +++ b/tests/test_gpu_execution.py @@ -0,0 +1,195 @@ +"""GPU commands inherit the allocation mask without probing or assigning devices.""" + +from __future__ import annotations + +import os +from collections.abc import Callable, Iterator +from contextlib import contextmanager +from dataclasses import replace +from pathlib import Path +from typing import Any +from unittest.mock import Mock + +import pytest + +from lightcone.engine import assets, container, identity, plan, worker +from lightcone.engine import run as engine_run +from lightcone.engine.project import ProjectError +from lightcone.engine.sandbox import policy as policy_module + +_SPEC = """ +version: "0.0.13" +name: analysis +inputs: [] +outputs: + - id: visibility + type: metric + format: txt + recipe: + command: printf '%s' "$CUDA_VISIBLE_DEVICES" > {output} +""" + + +@pytest.fixture +def project(analysis: Callable[..., Path]) -> tuple[Path, plan.Task, worker.RunContext]: + root = analysis(_SPEC) + task = plan.build(root).tasks[("baseline", "visibility")] + context = worker.RunContext( + env_version=identity.env_version(root), + head=("0123456789abcdef", "https://example/analysis.git"), + versions=assets.Versions(), + runtime=container.runtime_for_run(root, build=False), + uv_version="0.0.0-test", + ) + return root, task, context + + +@pytest.mark.parametrize("requested", [0, 1, 2]) +def test_recipe_inherits_the_whole_mask_without_changing_the_worker_environment( + project: tuple[Path, plan.Task, worker.RunContext], + monkeypatch: pytest.MonkeyPatch, requested: int, +) -> None: + root, task, context = project + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + monkeypatch.setattr(policy_module, "_gpu_device_paths", lambda: ()) + result = worker.execute(root, replace(task, resources={"gpus": requested}), {}, context) + assert result.status == "ok", result.reason + assert task.output_path.read_text() == ("2,0" if requested else "") + assert os.environ["CUDA_VISIBLE_DEVICES"] == "2,0" + + +@pytest.mark.parametrize("mask", [None, ""]) +def test_missing_worker_gpu_mask_preserves_existing_output_and_manifest( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + mask: str | None, +) -> None: + root, task, context = project + task.output_path.parent.mkdir(parents=True, exist_ok=True) + task.output_path.write_text("previous output") + task.manifest_path.write_text("previous manifest") + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + if mask is not None: + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) + with pytest.raises(ProjectError, match="nonempty CUDA_VISIBLE_DEVICES"): + worker.execute(root, replace(task, resources={"gpus": 1}), {}, context) + assert task.output_path.read_text() == "previous output" + assert task.manifest_path.read_text() == "previous manifest" + + +@pytest.mark.parametrize("runtime", ["docker", "podman"]) +def test_unsupported_gpu_container_preserves_existing_output_and_manifest( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + runtime: str, +) -> None: + root, task, context = project + task.output_path.parent.mkdir(parents=True, exist_ok=True) + task.output_path.write_text("previous output") + task.manifest_path.write_text("previous manifest") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + context = replace(context, runtime=replace( + context.runtime, mode="containerized", runtime=runtime, + )) + with pytest.raises(ProjectError, match="GPU containers require podman-hpc"): + worker.execute(root, replace(task, resources={"gpus": 1}), {}, context) + assert task.output_path.read_text() == "previous output" + assert task.manifest_path.read_text() == "previous manifest" + + +@pytest.mark.parametrize("use_gpus", [False, True]) +def test_probe_inherits_the_native_allocation_mask_or_hides_gpus( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + use_gpus: bool, +) -> None: + _, _, context = project + monkeypatch.setattr(policy_module, "_gpu_device_paths", lambda: ()) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + received: list[bytes] = [] + outcome = engine_run._probe( + context.runtime, [], ("sh", "-c", "printf '%s' \"$CUDA_VISIBLE_DEVICES\""), + use_gpus, + output=lambda stream, data: received.append(data) if stream == "stdout" else None, + ) + assert outcome.returncode == 0 + assert b"".join(received).decode() == ("2,0" if use_gpus else "") + assert os.environ["CUDA_VISIBLE_DEVICES"] == "2,0" + + +def test_probe_missing_gpu_mask_refuses_before_running_the_command( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, +) -> None: + _, _, context = project + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + execute = Mock() + monkeypatch.setattr(engine_run.sandbox, "run", execute) + with pytest.raises(ProjectError, match="nonempty CUDA_VISIBLE_DEVICES"): + engine_run._probe(context.runtime, [], ("true",), True, output=lambda *_: None) + execute.assert_not_called() + + +@pytest.mark.parametrize("runtime_name", ["docker", "podman", "podman-hpc"]) +def test_container_probe_on_gpu_cluster_exposes_only_supported_gpus_and_reports_cpu_mode( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + runtime_name: str, +) -> None: + from distributed import Client, LocalCluster + + from lightcone.engine import compute, sandbox + + root, _, context = project + runtime = replace(context.runtime, mode="containerized", runtime=runtime_name, image_id="test") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + monkeypatch.setattr(container, "runtime_for_run", lambda *args, **kwargs: runtime) + monkeypatch.setattr(container, "converge", lambda _: []) + commands = [] + reservations = [] + + def execute(backend: Any, policy: Any, argv: Any, **kwargs: Any) -> sandbox.Outcome: + commands.append(backend.wrap(policy, argv)) + return sandbox.Outcome(0, backend.attest(policy)) + + submit = Client.submit + + def record(client: Any, *args: Any, **kwargs: Any) -> Any: + reservations.append(kwargs["resources"]) + return submit(client, *args, **kwargs) + + @contextmanager + def connect(cluster_id: str) -> Iterator[Client]: + with LocalCluster( + n_workers=1, threads_per_worker=1, processes=False, dashboard_address=None, + resources={"CPU": 2, "MEMORY": 1024**3, "GPU": 2}, + ) as cluster, Client(cluster) as client: + yield client + + monkeypatch.setattr(sandbox, "run", execute) + monkeypatch.setattr(compute, "connect", connect) + monkeypatch.setattr(Client, "submit", record) + outcome = engine_run.probe(root, ["true"], cluster_id="gpu-cluster") + assert outcome.returncode == 0 + assert reservations == [{"CPU": 2, "MEMORY": 1024**3, "GPU": 2}] + assert len(commands) == 1 + if runtime_name == "podman-hpc": + assert "--gpu" in commands[0] + assert "--env=CUDA_VISIBLE_DEVICES=2,0" in commands[0] + assert not any("without GPUs" in note for note in outcome.notes) + else: + assert "--env=CUDA_VISIBLE_DEVICES=" in commands[0] + assert "--env=NVIDIA_VISIBLE_DEVICES=void" in commands[0] + assert "--gpu" not in commands[0] + assert any("this probe runs without GPUs" in note for note in outcome.notes) + + +def test_gpu_rerun_missing_mask_explains_how_to_select_local_devices( + project: tuple[Path, plan.Task, worker.RunContext], monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture[str], +) -> None: + root, _, _ = project + (root / "astra.yaml").write_text(_SPEC.replace( + "command:", "resources: {gpus: 1}\n command:", + )) + monkeypatch.chdir(root) + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + assert worker.main(["baseline/visibility"]) == 2 + error = capsys.readouterr().err + assert "CUDA_VISIBLE_DEVICES=0 datalad rerun" in error + assert "Slurm sets it" in error diff --git a/tests/test_materialize.py b/tests/test_materialize.py index c9b9a29b..44b323af 100644 --- a/tests/test_materialize.py +++ b/tests/test_materialize.py @@ -471,7 +471,9 @@ def digest(path: Path) -> str: return real(path) class _Copied(_Inline): - def submit(self, fn: Callable[..., object], *args: object, key: str) -> object: + def submit( + self, fn: Callable[..., object], *args: object, key: str, resources: dict[str, float], + ) -> object: return fn(*pickle.loads(pickle.dumps(args))) monkeypatch.setattr(assets, "data_version", digest) @@ -1060,6 +1062,200 @@ def test_the_recorded_command_holds_on_a_fresh_clone( # ---- the scheduler seam ---------------------------------------------------- +def _resource_cluster( + monkeypatch: pytest.MonkeyPatch, *, workers: int = 1, gpus: int = 0, +) -> None: + from distributed import Client, LocalCluster + + from lightcone.engine import compute + + @contextmanager + def connect(cluster_id: str) -> Iterator[Any]: + with LocalCluster( + n_workers=workers, threads_per_worker=4, processes=False, + dashboard_address=None, resources={"CPU": 4, "MEMORY": 2 * 1024**3, "GPU": gpus}, + ) as cluster, Client(cluster, set_as_default=False) as client: + yield client + + monkeypatch.setattr(compute, "connect", connect) + + +@pytest.mark.parametrize("resource_spec", ["cpus: 5", "memory: 3Gi"]) +def test_resource_refusal_precedes_preparation_and_all_recipes( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, resource_spec: str, +) -> None: + spec = _SPEC.replace( + "command: cat", f"resources: {{{resource_spec}}}\n command: cat" + ) + root = analysis(spec, universes={"baseline": _UNIVERSE}) + before = dataset.head(root) + _resource_cluster(monkeypatch) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("preparation began before all resource requests were validated") + + monkeypatch.setattr(engine, "_fetch_inputs", unexpected) + monkeypatch.setattr(engine.container, "runtime_for_run", unexpected) + with pytest.raises(ProjectError, match="baseline/second:.*no worker"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert dataset.head(root) == before + assert not dataset.status(root) + assert not (root / "results/baseline/first.txt").exists() + + +@pytest.mark.parametrize("resource_spec", [ + "gpus: 1", "disk: 1Gi", "cpus: 0.5", "cpus: 0", "memory: null", "time_limit: 1s", +]) +def test_execution_only_resources_do_not_block_read_only_commands( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, resource_spec: str, +) -> None: + spec = _SPEC.replace( + "command: cat", f"resources: {{{resource_spec}}}\n command: cat", + ) + root = analysis(spec, universes={"baseline": _UNIVERSE}) + assert len(engine.status(root).outputs) == 2 + assert set(engine.check(root, []).planned) == {"baseline/first", "baseline/second"} + + _resource_cluster(monkeypatch) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("preparation began before execution requirements were validated") + + monkeypatch.setattr(engine, "_fetch_inputs", unexpected) + with pytest.raises(ProjectError, match="baseline/second"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert not dataset.status(root) + + + +@pytest.mark.parametrize("runtime_name", ["docker", "podman"]) +def test_gpu_runtime_refusal_precedes_image_build_and_all_recipes( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, runtime_name: str, +) -> None: + spec = _SPEC.replace("command: cat", "resources: {gpus: 1}\n command: cat") + root = analysis(spec, universes={"baseline": _UNIVERSE}) + _resource_cluster(monkeypatch, gpus=1) + pyproject = root / "pyproject.toml" + pyproject.write_text(pyproject.read_text() + "\n[tool.lightcone.image]\napt-install = []\n") + dataset.save(root, [pyproject], "declare a container image") + monkeypatch.setattr(engine.container, "runtime_name", lambda _: runtime_name) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("image preparation began for an unsupported GPU runtime") + + monkeypatch.setattr(engine.container.image, "tag", unexpected) + before = dataset.head(root) + with pytest.raises(ProjectError, match="GPU containers require podman-hpc"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert dataset.head(root) == before + assert not dataset.status(root) + assert not (root / "results/baseline/first.txt").exists() + +def test_empty_cluster_refuses_before_project_preparation( + root: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + _resource_cluster(monkeypatch, workers=0) + + def unexpected(*args: object, **kwargs: object) -> None: + pytest.fail("preparation began without an available worker") + + monkeypatch.setattr(engine, "_fetch_inputs", unexpected) + with pytest.raises(ProjectError, match="no workers"): + engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert not dataset.status(root) + + +@pytest.mark.parametrize( + ("resource_spec", "gpus", "expected_parallelism"), + [ + ("", 0, 4), + ("cpus: 3, memory: 256Mi", 0, 1), + ("cpus: 1, memory: 1Gi", 0, 2), + ("cpus: 1, memory: 256Mi, gpus: 1", 2, 1), + ], +) +def test_real_dask_respects_recipe_resource_reservations( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, + resource_spec: str, gpus: int, expected_parallelism: int, +) -> None: + # Four Dask threads would run all four subprocesses together without + # resource reservations. Each independent output records its live interval. + spec = 'version: "0.0.13"\nname: analysis\ninputs: []\noutputs:\n' + "".join( + f" - id: task{index}\n" + " type: metric\n" + " format: json\n" + " recipe:\n" + f" resources: {{{resource_spec}}}\n" + " command: python src/work.py {output}\n" + for index in range(4) + ) + root = analysis(spec, files={"src/work.py": """ + import json + import os + import sys + import time + from pathlib import Path + start = time.monotonic() + time.sleep(0.5) + Path(sys.argv[1]).write_text(json.dumps([ + start, time.monotonic(), os.environ.get("CUDA_VISIBLE_DEVICES"), + ])) + """}) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + _resource_cluster(monkeypatch, gpus=gpus) + + report = engine.materialize(root, [], cluster_id=CLUSTER_ID) + + assert report.ok and len(report.made) == 4 + events = [] + for path in (root / "results/baseline").glob("task*.json"): + start, finish, visible = json.loads(path.read_text()) + assert visible == ("2,0" if gpus else "") + events.extend([(start, 1), (finish, -1)]) + live = peak = 0 + for _, change in sorted(events): + live += change + peak = max(peak, live) + assert peak == expected_parallelism + assert not dataset.status(root) + + + +def test_current_gpu_output_needs_no_gpu_to_build_a_cpu_dependent( + analysis: Callable[..., Path], monkeypatch: pytest.MonkeyPatch, +) -> None: + spec = _SPEC.replace("command: echo", "resources: {gpus: 1}\n command: echo") + root = analysis(spec, universes={"baseline": _UNIVERSE}) + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") + _resource_cluster(monkeypatch, gpus=1) + assert engine.materialize(root, ["first"], cluster_id=CLUSTER_ID).made == ["baseline/first"] + original = (root / "results/baseline/.first.manifest.json").read_bytes() + + _resource_cluster(monkeypatch) + report = engine.materialize(root, [], cluster_id=CLUSTER_ID) + assert report.ok + assert report.current == ["baseline/first"] + assert report.made == ["baseline/second"] + assert (root / "results/baseline/.first.manifest.json").read_bytes() == original + assert not dataset.status(root) + + +def test_current_outputs_are_not_submitted_to_dask( + root: Path, inline: None, monkeypatch: pytest.MonkeyPatch, +) -> None: + engine.materialize(root, [], cluster_id=CLUSTER_ID) + + class NoExecution(_Inline): + def validate(self, tasks: Any) -> dict[Any, Any]: + assert not list(tasks) + return {} + + def submit(self, *args: Any, **kwargs: Any) -> Any: + pytest.fail("a current output was submitted to Dask") + + _cluster(monkeypatch, NoExecution()) + assert len(engine.materialize(root, [], cluster_id=CLUSTER_ID).current) == 2 + def test_a_real_cluster_still_fits_through_the_seam(root: Path, cluster_id: str) -> None: """The one test that starts Dask. The seam is only worth having if the thing it abstracts still goes through it.""" @@ -1084,7 +1280,8 @@ def test_a_processes_cluster_fits_through_the_seam( @contextmanager def processes(cluster_id: str) -> Iterator[Any]: with LocalCluster( - n_workers=2, threads_per_worker=1, processes=True, dashboard_address=None + n_workers=2, threads_per_worker=1, processes=True, dashboard_address=None, + resources={"CPU": 1, "MEMORY": 1024**3}, ) as cluster: with Client(cluster, set_as_default=False) as client: yield client diff --git a/tests/test_plan.py b/tests/test_plan.py index 48a994b0..9f069d38 100644 --- a/tests/test_plan.py +++ b/tests/test_plan.py @@ -122,6 +122,39 @@ def test_an_output_addresses_its_own_file(tmp_path: Path) -> None: assert "results/baseline/fit.json" in task.recipe +def test_recipe_resources_survive_graph_resolution(tmp_path: Path) -> None: + spec = _SPEC.replace( + "command: python src/fit.py", + "resources: {cpus: 4, memory: 6Gi, time_limit: 1h30m}\n" + " command: python src/fit.py", + ) + graph = _build(_project(tmp_path, spec)) + task = graph.tasks[("baseline", "fit")] + assert task.resources == {"cpus": 4, "memory": "6Gi", "time_limit": "1h30m"} + assert graph.tasks[("baseline", "report")].resources == {} + + +@pytest.mark.parametrize( + ("declaration", "expected"), + [ + ("gpus: 1", {"gpus": 1}), + ("disk: 10Gi", {"disk": "10Gi"}), + ("cpus: 0.5", {"cpus": 0.5}), + ("cpus: 0", {"cpus": 0}), + ("memory: null", {"memory": None}), + ], +) +def test_execution_support_does_not_limit_graph_construction( + tmp_path: Path, declaration: str, expected: dict[str, object], +) -> None: + spec = _SPEC.replace( + "command: python src/fit.py", + f"resources: {{{declaration}}}\n command: python src/fit.py", + ) + task = _build(_project(tmp_path, spec)).tasks[("baseline", "fit")] + assert task.resources == expected + + def test_a_declared_input_resolves_to_its_source(tmp_path: Path) -> None: task = _build(_project(tmp_path)).tasks[("baseline", "fit")] assert task.inputs == {"catalog": tmp_path / "data" / "catalog.fits"} @@ -280,5 +313,3 @@ def test_an_output_without_a_format_is_refused_by_name(tmp_path: Path) -> None: - - diff --git a/tests/test_sandbox_oci.py b/tests/test_sandbox_oci.py index cf8cdccd..de6efac3 100644 --- a/tests/test_sandbox_oci.py +++ b/tests/test_sandbox_oci.py @@ -8,11 +8,13 @@ from __future__ import annotations import subprocess +from dataclasses import replace from pathlib import Path from typing import Any import pytest +from lightcone.engine.project import ProjectError from lightcone.engine.sandbox import boundary, exec_policy from lightcone.engine.sandbox.boundary import Unavailable from lightcone.engine.sandbox.model import Policy @@ -185,6 +187,43 @@ def test_runtimes_differ_only_in_their_spellings(root: Path, policy: Policy) -> assert p == d == h +def test_podman_hpc_preserves_the_allocation_mask_and_enables_native_gpu_support( + root: Path, policy: Policy, +) -> None: + devices = "2,0" + selected = replace(policy, env={ + **policy.env, "CUDA_VISIBLE_DEVICES": devices, "CUDA_DEVICE_ORDER": "PCI_BUS_ID", + }) + backend = _backend(root, "podman-hpc") + argv = backend.wrap(selected, ["true"]) + assert argv == backend.wrap(selected, ["true"]) + assert f"--env=CUDA_VISIBLE_DEVICES={devices}" in argv + assert "--env=CUDA_DEVICE_ORDER=PCI_BUS_ID" in argv + assert "--gpu" in argv + assert not any(arg.startswith("--device=") for arg in argv) + + +@pytest.mark.parametrize("runtime", ["podman", "docker"]) +def test_generic_gpu_containers_are_explicitly_refused( + root: Path, policy: Policy, runtime: str, +) -> None: + selected = replace(policy, env={**policy.env, "CUDA_VISIBLE_DEVICES": "2,0"}) + with pytest.raises(ProjectError, match="GPU containers require podman-hpc"): + _backend(root, runtime).wrap(selected, ["true"]) + + +@pytest.mark.parametrize("runtime", ["podman", "docker", "podman-hpc"]) +def test_cpu_containers_do_not_request_gpu_access(root: Path, policy: Policy, runtime: str) -> None: + policy = replace(policy, env={**policy.env, "NVIDIA_VISIBLE_DEVICES": "all"}) + argv = _backend(root, runtime).wrap(policy, ["true"]) + assert "--env=CUDA_VISIBLE_DEVICES=" in argv + assert "--env=NVIDIA_VISIBLE_DEVICES=void" in argv + assert "--env=NVIDIA_VISIBLE_DEVICES=all" not in argv + assert policy.env["NVIDIA_VISIBLE_DEVICES"] == "all" + assert "--gpu" not in argv and "--gpus" not in argv + assert not any(arg.startswith("--device=") for arg in argv) + + def test_the_environment_is_an_allowlist_never_ambient( root: Path, policy: Policy, monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/tests/test_sandbox_policy.py b/tests/test_sandbox_policy.py index 0cfc2605..66978d7b 100644 --- a/tests/test_sandbox_policy.py +++ b/tests/test_sandbox_policy.py @@ -14,6 +14,7 @@ import pytest +from lightcone.engine.project import ProjectError from lightcone.engine.sandbox import policy as policy_module from lightcone.engine.sandbox.boundary import scope from lightcone.engine.sandbox.model import EXEC_ALLOWLIST_VERSION @@ -238,6 +239,50 @@ def test_the_entropy_sources_stay_read_only(built: policy_module.Policy) -> None assert device not in built.write, node +@pytest.mark.parametrize("containerized", [False, True]) +@pytest.mark.parametrize("use_gpus", [False, True]) +def test_gpu_policy_sets_command_visibility_without_changing_the_host( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, + containerized: bool, use_gpus: bool, +) -> None: + project = tmp_path / "project" + project.mkdir() + device = tmp_path / "nvidia0" + device.touch() + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "2,0") + monkeypatch.setenv("CUDA_DEVICE_ORDER", "PCI_BUS_ID") + monkeypatch.setattr(policy_module, "_gpu_device_paths", lambda: (device,)) + with scope(policy_module.exec_policy( + project, containerized=containerized, use_gpus=use_gpus, + )) as built: + assert built.env["CUDA_VISIBLE_DEVICES"] == ("2,0" if use_gpus else "") + if use_gpus: + assert built.env["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" + assert (device in built.write) == (use_gpus and not containerized) + assert Path("/dev") not in built.write + assert os.environ["CUDA_VISIBLE_DEVICES"] == "2,0" + assert os.environ["CUDA_DEVICE_ORDER"] == "PCI_BUS_ID" + + +def test_gpu_policy_refuses_a_missing_mask_before_creating_private_state( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) + with pytest.raises(ProjectError, match="nonempty CUDA_VISIBLE_DEVICES"): + policy_module.exec_policy(tmp_path, containerized=True, use_gpus=True) + assert not (tmp_path / ".lightcone").exists() + + +def test_only_nvidia_character_nodes_are_granted( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + for name in ("nvidia0", "nvidiactl", "nvidia-user-file", "unrelated"): + (tmp_path / name).touch() + monkeypatch.setattr(policy_module, "_DEVICE_ROOT", tmp_path) + monkeypatch.setattr(Path, "is_char_device", lambda p: p.name in {"nvidia0", "nvidiactl"}) + assert policy_module._gpu_device_paths() == (tmp_path / "nvidia0", tmp_path / "nvidiactl") + + def test_proc_and_sys_are_not_restricted(built: policy_module.Policy) -> None: """Real tools write them — /proc/self/oom_score_adj, coredump_filter, MPI and CUDA runtimes poking /sys — and none of it is a channel