diff --git a/CLAUDE.md b/CLAUDE.md index cf7933d9..f68206c6 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -181,7 +181,7 @@ src/lightcone/ # namespace — NO __init__.py ├── run.py # what `lc run` is: the probe + the uv hop ├── compute/ # explicit allocations and borrowed Dask clients │ ├── __init__.py # Compute: catalog, resolve, launch, status, down; connect() - │ ├── model.py # the shared Pydantic models and the Provider protocol + │ ├── model.py # the shared Pydantic models, the Provider protocol, PROVIDERS │ ├── catalog.py # compute.yaml with local defaults and policy │ ├── runtime.py # private files, TLS material, the scheduler config │ ├── local.py # local provider: validated OS process identities @@ -1704,8 +1704,8 @@ refusal point. the ambient venue ladder. `lc compute resources/launch/status/down` manages local and Slurm allocations through `engine.compute.Provider`. `run` and `materialize` require a cluster name or full ID as their first positional argument; neither creates compute. -`materialize --check` remains cluster-free. Catalog offers expose resource shapes; -connections supply stable native namespaces. Native jobs and validated local OS +`materialize --check` remains cluster-free. Catalog offers expose resource shapes +and name their provider; each provider reaches one native authority. Native jobs and validated local OS identities are the allocation authority; standard Dask supplies execution state. No Lightcone server, lifecycle database, custom Dask worker, or implicit allocation. @@ -1717,14 +1717,22 @@ Resolve names through fresh discovery and refuse missing, ambiguous, or incomple observations. Check existing names before submission, but do not claim atomic global reservation across native backends. A name can be reused after termination; use the full ID to address an exact incarnation or bypass unrelated discovery failures. -Slurm uses `JobName=lc-v1-` and -`Comment=lightcone:v1:kind=dask:token=<32hex>`. Verify the owner and both native +Slurm uses `JobName=lc-` and +`Comment=lightcone:kind=dask:token=<32hex>`. Verify the owner and both native fields before attachment or cancellation. Missing live comments make discovery -incomplete; missing historical comments leave identity unknown. Historical +incomplete; missing historical comments leave identity unknown, so a user's own +job named `lc-…` blocks name checks until it is renamed or ends — loudly, +naming the job, never by adopting it. Historical comment retention requires Slurm's `AccountingStoreFlags` to include `job_comment`. Resolve the Slurm command user's UID through `id -u` on the same command runner, and use it for every native ownership check and filter. +**Compute formats carry no version (2026-10).** Cluster IDs, native labels, +allocation records, the catalog and compute JSON output are unversioned: +allocations are ephemeral, so a format change strands only allocations that +end on their own, and new ones are made under the new format. Manifests are +different — committed, they keep `schema_version`. + **Local compute needs no setup.** The built-in local offer provides detected usable logical CPUs and RAM, one node, fast startup, no walltime and a 30-minute idle timeout. Loading the catalog writes no catalog and starts no cluster. @@ -1732,22 +1740,27 @@ timeout. Loading the catalog writes no catalog and starts no cluster. the name to `local`. `--wait` returns when the accepted allocation is ready; timeout or startup failure retains its ID without resubmitting or terminating it. Configured remote offers precede the built-in local offer in selection order. -Explicit local connections supply their own offers instead. `local.resources` -and `local.time` override the built-in CPU/RAM budget and time limits and cannot -accompany explicit local connections. -`local.enabled: false` blocks local launch and execution while preserving inspection -and termination. Recognized NERSC login nodes disable local compute automatically; +A catalog's own local offers replace it, which is the one way to change the +local CPU/RAM budget or time limits: there is no separate local settings block. +`allow_local: false`, the catalog's only local key, blocks local launch and +execution while preserving inspection and termination. Recognized NERSC login nodes disable local compute automatically; other sites can disable it in their catalogs. Native permissions remain the enforcement boundary. -GPU offers require an explicit catalog and, for local launches, a nonempty -`CUDA_VISIBLE_DEVICES` mask on Linux. No GPU auto-discovery. Local GPU capacity and -model labels are configured, not hardware-verified; allocations do not reserve -devices exclusively against other host programs or allocations. +The built-in local offer takes its GPUs from the `CUDA_VISIBLE_DEVICES` mask on +Linux, counted as CUDA reads it (up to the first entry that is neither an index nor +a device UUID, so `-1` is none), labelled generic `GPU`; without a mask it is +CPU-only. Remote GPU offers still require an explicit catalog, and an explicit +local GPU offer plans only when the mask exposes at least its count. No GPU +auto-discovery: the mask is the allocation, never probed against hardware, so local +GPU capacity and model labels are configured, not hardware-verified; allocations do +not reserve devices exclusively against other host programs or allocations. The +local shortcut takes an offer whole, GPUs included; `--gpus 0` takes it without +them, since local GPUs are never reserved. Missing explicit paths and invalid files are errors. Execution still requires an explicitly launched cluster's name or ID. -**A Slurm connection needs no launch settings (2026-09).** Every `launch` key -defaults, and the defaults assume a home directory shared by login and compute +**A Slurm offer needs no launch settings (2026-09).** Every launch key in an +offer's `config` defaults, and the defaults assume a home directory shared by login and compute nodes, which is also what SkyPilot's Slurm backend assumes. Workers run `sys.executable`, the driver's own installation, so client, scheduler and workers match exactly with no resolution and no network on compute nodes. Launching the @@ -1755,7 +1768,7 @@ bootstrap through `uv run --with` at job time was considered and rejected: it re-resolves the Dask closure away from the driver's, and it needs package-index access from compute nodes. SkyPilot installs its runtime per node only because its client is off-cluster; lc submits from the login node, where its installation -already is. `connection_root` defaults to `~/.lightcone/compute` +already is. The catalog's top-level `connection_root` defaults to `~/.lightcone/compute` (`runtime.DEFAULT_CONNECTION_ROOT`, shared with local). An unset `scratch_root` is chosen by the bootstrap on each node (`tempfile.gettempdir()`), never frozen from the driver's temporary directory. @@ -1774,13 +1787,13 @@ still require all expected workers. Serial per-output Git/annex commits remain the driver's responsibility; changing that persistence/provenance model is deferred. **Compute uses one shared Pydantic model family.** `Catalog` loads directly into -the `Connection`, `Offer`, `Resources`, `TimeLimits`, and `Startup` objects used by +the `Offer`, `Resources`, `TimeLimits`, and `Startup` objects used by providers; do not introduce parallel configuration classes. Memory units are explicit: `Resources.memory_gib` / `memory_bytes`, `Resources.from_bytes(...)`, and `Request.memory_bytes`. Duration strings expose derived seconds through `TimeLimits`. -Connection names live only in the catalog's mapping keys. Use validated `replace` +Use validated `replace` for updates and the explicit `as_dict` allowlists for public output. Preserve -duplicate-key rejection in YAML; providers validate their own `launch` and `config`. +duplicate-key rejection in YAML; providers validate their own offers' `config`. **Allocation syntax follows SkyPilot without depending on SkyPilot.** Compute CPU/memory requests support exact quantities or `+` minimums. Compute memory @@ -1794,7 +1807,9 @@ Named Slurm offers must map their public label to the site's GRES type through evidence in observations. **Configured compute roots may be filesystem aliases.** Resolve connection and -scratch roots before appending managed namespace, submission, or attempt paths. +scratch roots before appending managed provider, submission, or attempt paths. +The catalog resolves `connection_root` once, at load, so an unusable root is a +catalog error naming the field; providers append to it and never resolve it again. Keep symlink rejection within those managed paths and enforce private directory and credential permissions. Do not resolve Python executables: virtualenv paths must retain their environment identity. Never change existing ancestor permissions. @@ -1883,7 +1898,7 @@ instead of holding it for Dask's default hour. **Block local compute on recognized NERSC login nodes (2026-09).** A nonempty `NERSC_HOST` plus a short hostname matching `login[0-9]+` disables local launch -and execution, including explicit local offers and `local.enabled: true`. +and execution, including explicit local offers and `allow_local: true`. Interactive compute nodes remain eligible; inherited `SLURM_JOB_ID` never exempts a login node. Apply this policy at runtime without writing a configuration file. Keep inspection, termination, and Slurm execution available. Both execution @@ -2127,6 +2142,33 @@ unlinks before writing; a new tampering test should too. ### Recorded decisions +- **Offers name their provider; there are no connections (2026-10).** A + catalog used to be `connections` — a UUID `namespace`, a `provider`, a + `context`, `launch` settings — plus offers referencing one by name. lc runs + inside one Slurm environment, so `context` (`--clusters`) went, and with it + every provider reaches exactly one native authority: this host, or the + cluster the Slurm client reaches by default. That left the connection a pure + indirection. Offers carry `provider`; worker launch settings joined each + offer's `config` (`task_slots_per_node` was already validated per offer); + `connection_root`, the one setting read outside `plan`, is the catalog's one + top-level key; and the cluster ID encodes the provider instead of the + namespace, so a full ID routes with no catalog lookup. The UUID was worse + than redundant: Slurm discovery never filtered on it, so two catalogs with + different UUIDs gave one job two IDs. What a full ID does *not* free from + the catalog is connection material, which lives under the launching + catalog's `connection_root`, for Slurm as for local: native inspection and + `down` work from any catalog, connecting does not. Launch creates the + submission's directory (`/slurm/`, its log and each + attempt's TLS material) before submitting, so a running job without one + is refused as launched under another root rather than told to wait. + `PROVIDERS` lives beside the `Provider` protocol in `model.py`, and an + offer naming anything else is a catalog error — a typo once loaded and + then failed every discovery, blocking every launch. Discovery queries `Catalog.providers` — + those the offers name, plus local always — so removing a provider's last + offer hides its jobs from name lookup and the listing, never from a full + ID. Re-add a per-authority concept only with a venue that genuinely has two + authorities of one provider. + - **Guard NERSC login nodes by default (2026-09).** This reverses the earlier no-login-node-guard decision: an unconfigured first launch must not allocate a whole shared login node. Detect the documented NERSC environment marker and @@ -2166,9 +2208,11 @@ unlinks before writing; a new tampering test should too. - **Local compute accompanies remote catalogs (2026-09).** The built-in offer uses the host's usable CPU/RAM capacity and follows configured offers, replacing the previous one-CPU/1-GiB fallback that disappeared when a catalog existed. - `local.resources` sets a smaller budget; explicit local connections use their - own offers. Login-node catalogs disable local launch and execution through - `local.enabled: false`. Inspection and termination stay available. One local + A catalog's own local offers replace it, including to set a smaller budget + (a `local.resources`/`local.time` block once did that too, a second way to + say what an offer already says, so it went). Login-node catalogs disable + local launch and execution through `allow_local: false`. Inspection and + termination stay available. One local allocation per user per machine is enforced across catalogs and connection roots by scanning the process table for a live owner before launch (`local._running_owners`: a session leader of this user running diff --git a/docs/api/compute.md b/docs/api/compute.md index 76704d08..7ccc4749 100644 --- a/docs/api/compute.md +++ b/docs/api/compute.md @@ -7,10 +7,10 @@ It owns no service, registry, or saved current-cluster selection. | Symbol | Contract | |---|---| | `Request.parse(...)` | Exact/minimum CPU and memory requests, exact accelerator type/count, node count, walltime, startup class. | -| `Catalog.load(path)` | Ordered fixed shapes and stable connection namespaces; apply local defaults or disable policy alongside configured offers. | +| `Catalog.load(path)` | Ordered fixed shapes, each naming its provider; apply local defaults or disable policy alongside configured offers. | | `Compute.plan(request, *, name=None)` | Select an eligible offer and freeze its native launch settings and optional name without allocation. | | `Compute.launch(plan)` | Check names across native authorities, generate one if omitted, submit once, and return a self-contained `Identity`. | -| `Compute.discover()` | Snapshots and per-connection errors, querying each authority once. | +| `Compute.discover()` | Snapshots and per-provider errors, querying each provider in `Catalog.providers` once. | | `Compute.status(cluster_id, wait=False, timeout=300)` | Resolve a name or full ID; return native allocation state plus authenticated Dask readiness. Waiting backs off from one to 30 seconds between native queries. | | `Compute.down(cluster_id)` | Resolve a name or full ID, request native termination independent of scheduler health, and return the canonical `Identity`. | | `connect(cluster_id, timeout=10)` | Resolve a name or full ID; borrow a standard Dask client, closing the client but never the allocation. Submits one no-op task, so a caller's preparation restarts the idle countdown. | @@ -20,10 +20,10 @@ It owns no service, registry, or saved current-cluster selection. offer uses detected usable CPUs and RAM, one node, fast startup, and no walltime: it ends after 30 minutes without task activity. `local.resources` overrides its CPU/RAM budget and `local.time` its time limits; -`local.enabled: false` blocks local launch and execution while retaining connections -for inspection and termination. Remote catalogs retain the implicit local offer -unless disabled. Explicit local connections supply their own offers instead and -cannot be combined with `local.resources` or `local.time`. GPU offers require explicit configuration. +`local.enabled: false` blocks local launch and execution while local discovery +continues for inspection and termination. Remote catalogs retain the implicit local +offer unless disabled. Explicit local offers replace it and cannot be combined +with `local.resources` or `local.time`. GPU offers require explicit configuration. Loading creates no configuration file or allocation. The effective local policy also disables local offers on recognized NERSC login nodes: nonempty `NERSC_HOST` and a short hostname matching `login[0-9]+`. @@ -31,20 +31,22 @@ Explicit enablement and Slurm job environment variables do not override this guard; interactive compute nodes remain eligible. Local planning, launch, and execution check the same policy, while status and termination remain available. Missing paths selected through an argument or `LC_COMPUTE_CONFIG`, unreadable -files, and invalid catalogs remain errors. Stable connection namespaces let -separate invocations discover and attach to the same local allocations. +files, and invalid catalogs remain errors. `Catalog.providers` names the native +authorities to query: those the offers use, and always `local`. Each provider +keeps its allocations in one directory under the catalog's `connection_root`, so +separate invocations discover and attach to the same allocations. `Compute.plan_local()` selects only local offers and defaults the name to `local`. CLI `launch --wait` waits through `Compute.status` using the accepted immutable ID; errors retain that ID without resubmission or termination. -`model.py` defines the shared Pydantic models: `Connection`, `Offer`, `Resources`, `Accelerator`, +`model.py` defines the shared Pydantic models: `Offer`, `Resources`, `Accelerator`, `TimeLimits`, `Startup`, `Request`, `Identity`, `LaunchPlan`, and `Snapshot`. `Catalog` validates YAML directly into these objects, which providers also use. Unknown common fields are rejected; schema errors identify paths such as `offers.0.resources.cpus` without echoing input values. The YAML loader rejects -duplicate and non-string mapping keys before model validation. Provider-specific -`launch` and `config` mappings remain the provider's responsibility. +duplicate and non-string mapping keys before model validation. Each offer's +provider-specific `config` mapping remains the provider's responsibility. Units are explicit. `Resources.memory_gib` stores exact decimal GiB (the YAML key is `memory`), and `memory_bytes` derives an exact integer. Native observations use @@ -55,7 +57,8 @@ but requiring a `default` or an `idle`, and exposes `default_seconds`, resolved hard walltime as `seconds` and derives `idle_seconds` from its offer; either may be `None` for a local plan, while Slurm plans always have `seconds` and never `idle_seconds`. `Startup.class_` corresponds to YAML `class`. -Connection names exist only as catalog mapping keys, referenced by `Offer.connection`. +`Offer.provider` names a factory in `PROVIDERS`, which receives the catalog's +`connection_root`. Compute memory accepts bare GiB quantities and SkyPilot-style binary units: `32`, `32GB`, and `32GiB` agree. CPU and memory requests accept a trailing `+`. @@ -95,7 +98,7 @@ See [Dask's Nanny](https://distributed.dask.org/en/stable/worker.html#nanny), [Dask resilience](https://distributed.dask.org/en/stable/resilience.html), and [Slurm's `srun` options](https://slurm.schedmd.com/srun.html). -Configured connection and scratch roots are resolved before managed paths are +The configured `connection_root` and scratch roots are resolved before managed paths are appended, so filesystem aliases such as a symlinked home directory are supported. Managed directories and credential files retain strict symlink, ownership, and permission checks, including modes `0700` and `0600`, respectively. @@ -106,19 +109,19 @@ not query or freeze the site's time policy. Every launch requests a finite nativ independent Lightcone deadline or guarantee of a finite overrun. `Snapshot` distinguishes native allocation evidence from scheduler observations. -No live allocation size is filled from today's catalog. Connection namespaces -persist independently of offers, and IDs encode native incarnation evidence -without a UUID-to-job lookup database. Exceptions retain known cluster IDs and +No live allocation size is filled from today's catalog. IDs encode their provider +and native incarnation evidence, so a full ID routes without consulting the offers +and without a UUID-to-job lookup database. Exceptions retain known cluster IDs and submission tokens for partial/ambiguous acceptance. `Identity.name` is a human-facing name; `Identity.encode()` is the immutable allocation reference. Generated names use `lc-` plus 12 hexadecimal characters -from a standard-library UUID4. Launch checks every configured connection and +from a standard-library UUID4. Launch checks every provider in `Catalog.providers` and rejects explicit duplicates or incomplete discovery before submission. Name resolution also requires complete discovery and exactly one current match. Concurrent launches can still race; ambiguous names are refused. Names can be reused after termination, while full IDs continue to identify the original -allocation without discovering unrelated connections. Slurm carries the name +allocation without discovering other providers. Slurm carries the name in `JobName=lc-v1-` and the submission token in `Comment=lightcone:v1:kind=dask:token=<32hex>`; local private locators carry the encoded identity. Slurm discovery and lifecycle checks verify both native fields @@ -227,7 +230,7 @@ command. A driver that exits before every task reports says so with materialization sends recipe output to stderr to leave stdout for its report. Local allocations are limited to one per user on each machine, independent of -connection roots and namespaces. Before spawning, the launcher scans the process +connection roots. Before spawning, the launcher scans the process table for a live owner of the same user: a session leader running `-P -m lightcone.engine.compute.local_runtime `, which excludes workers forked from it. The process table spans every catalog and connection root and needs no diff --git a/docs/cli/compute.md b/docs/cli/compute.md index 5486dc4f..e30cc802 100644 --- a/docs/cli/compute.md +++ b/docs/cli/compute.md @@ -22,8 +22,8 @@ GPUs still require explicit offers and, locally, a `CUDA_VISIBLE_DEVICES` mask. `~/.lightcone/compute.yaml` configures resource offers; `LC_COMPUTE_CONFIG` selects another file for all compute and execution commands. The top-level `local` block can override the built-in CPU/RAM budget or time limits, or disable local compute. -Without an explicit local connection, the built-in local offer is appended after -configured offers. Catalogs with explicit local connections use their own offers instead; +Each offer names its `provider` (`local` or `slurm`). The built-in local offer is +appended after configured offers, unless the catalog lists its own local offers; the shortcut chooses the first eligible local offer. See [local configuration](../user/cluster.md#customize-resource-offers). Missing explicit paths and invalid catalogs are errors. Loading a catalog or @@ -41,7 +41,7 @@ See [local allocations](../user/cluster.md#local-allocations) for detection deta | `launch` | Resolve one resource request and submit exactly once; print only the cluster name to stdout on acceptance. | | `launch --wait` | Submit once, then wait for all expected workers. `--timeout` sets the readiness deadline (default 300 seconds). | | `launch --dry-run` | Show the resolved shape and native launch parameters without allocation. | -| `status` | List one `name: status` line per current allocation across every configured connection; report the connections that could not be queried. | +| `status` | List one `name: status` line per current allocation across every provider the catalog's offers use, and always local; report the providers that could not be queried. | | `status CLUSTER` | Resolve a name or full ID, inspect native state, and probe Dask readiness separately. | | `status CLUSTER --wait` | Wait for readiness: the allocation is active and every node's worker is connected. The default deadline is 300 seconds, and queries grow less frequent as the wait goes on (up to every 30 seconds). Exits 1 on timeout, or at once if the allocation is ending or has ended; the allocation is left unchanged. | | `down CLUSTER` | Request native termination even if the scheduler is unavailable. An allocation that has already ended is a successful no-op when addressed by full ID; its name no longer resolves. A Slurm job that has left the queue is refused unless accounting confirms it ended. | @@ -57,14 +57,15 @@ CLUSTER=$(lc compute launch --wait) lc compute down "$CLUSTER" ``` -Launch checks current allocations across all configured connections. An explicit +Launch checks current allocations across all of the catalog's providers. An explicit name already in use is rejected; generated collisions are retried before the single submission. This check is not an atomic reservation: concurrent launches can race. Name lookup refuses ambiguous or incomplete discovery instead of choosing a cluster. A name can be reused after its allocation ends; it is not a durable reference to that allocation. Use the full immutable `id` from launch or -status JSON to address one allocation directly, including when unrelated -connections are unavailable. No name registry is maintained. +status JSON to address one allocation directly, including when other providers +are unavailable or no offer uses its provider any more. No name registry is +maintained. Supply both `--cpus` and `--memory`, or omit both for the local shortcut. The shortcut never selects a remote offer, including when local compute is disabled. @@ -122,8 +123,8 @@ name only once ready; JSON adds `ready: true`. A timeout or startup failure exit 1 and includes the accepted immutable ID in the error. Waiting never resubmits or terminates the accepted allocation. -Only one local cluster may run per user on a machine, across names, namespaces, -and configured roots. A launch that finds one of your local clusters running in +Only one local cluster may run per user on a machine, across names and +configured roots. A launch that finds one of your local clusters running in the process table fails until that cluster ends; launches that overlap can both succeed. @@ -134,7 +135,7 @@ scheduler credentials: |---|---| | `launch` | `plan`, `id`, `name`, `accepted` (`ready: true` after `--wait`; only `plan` with `--dry-run`) | | `status CLUSTER` | `id`, `name`, `phase`, `allocation`, `dask`, `reason`, `native_state` | -| `status` | `clusters` (a list of the above) and `errors` (by connection name) | +| `status` | `clusters` (a list of the above) and `errors` (by provider name) | | `down` | `id`, `name`, `termination_requested` | | any failure | `error`, `id`, `submission_token`, on stdout, with exit 1 | diff --git a/docs/user/cluster.md b/docs/user/cluster.md index 32e8e7ea..1856b3a1 100644 --- a/docs/user/cluster.md +++ b/docs/user/cluster.md @@ -47,12 +47,12 @@ Use `lc compute status NAME` for resource details and Dask readiness. Local resources are cooperative limits, not an exclusive CPU/RAM reservation. Only one local cluster can run per user on each machine. A launch checks the process table for a running local cluster of yours and refuses if it finds one, -including one launched through a different name, catalog, namespace, or -connection root. End the existing cluster before launching another; once its +including one launched through a different name, catalog, or connection root. +End the existing cluster before launching another; once its owner process exits, including on failure, idle expiry, or walltime expiry, a new launch proceeds. A refusal identifies the running cluster and its original catalog and connection root. Use that catalog to inspect or stop the cluster if the current -catalog no longer includes its connection. If the cluster's record is missing or +catalog sets a different `connection_root`. If the cluster's record is missing or damaged, the refusal names its process ID instead, to stop with `kill`. Launches that overlap can both succeed, and the check sees only the processes visible where `lc` runs: a launch inside a container does not see a cluster @@ -84,11 +84,9 @@ sessions. A `SLURM_JOB_ID` variable does not exempt a login node. See NERSC's [environment conventions](https://docs.nersc.gov/environment/) and [interactive sessions](https://docs.nersc.gov/connect/vscode/). -A local connection's optional `launch` settings are `connection_root` (default -`~/.lightcone/compute`), `scratch_root` (default: the temporary directory), -`python` (default: the interpreter running `lc`), and `task_slots_per_node` -(default: all of the offer's CPUs). A local connection's `context`, when set, is -the hostname it belongs to. Local offers take no `config`. +A local offer's optional `config` settings are `scratch_root` (default: the +temporary directory), `python` (default: the interpreter running `lc`), and +`task_slots_per_node` (default: all of the offer's CPUs). ## Cluster names @@ -105,16 +103,16 @@ letter and ending with a letter or digit. Launch writes only the name to stdout; readiness guidance goes to stderr. All execution and lifecycle commands accept either that name or the full immutable ID available in launch and status JSON. -Names are checked against current allocations across all configured connections -before launch. An explicit duplicate is refused, and an autogenerated collision +Names are checked against current allocations across every provider the catalog's +offers use, and always local, before launch. An explicit duplicate is refused, and an autogenerated collision is regenerated before submission. Discovery failures prevent this check from succeeding. Concurrent launches can still choose the same name, so lookup also refuses ambiguous names or incomplete discovery. Use a full ID to select a known -allocation directly when another connection cannot be queried. +allocation directly when another provider cannot be queried. -This includes local connections when `local.enabled: false`: existing local -allocations remain visible and can still have conflicting names. Repair a -connection's discovery error before launching another cluster or resolving names. +This includes local allocations when `local.enabled: false`: they remain visible +and can still have conflicting names. Repair a provider's discovery error before +launching another cluster or resolving names. A name may be reused once its allocation has ended. Keep the full ID when you need a durable reference to one allocation; a later cluster with the same name @@ -152,7 +150,7 @@ compute on every node using the catalog, set: version: 1 local: enabled: false -# Add Slurm connections and offers as shown below. +# Add Slurm offers as shown below. ``` This blocks local launches, including explicit local offers, and new execution @@ -165,54 +163,45 @@ Leave `local.enabled` at its default to use local compute inside a NERSC interactive compute-node session. The login-node guard still applies, and it creates no configuration file. -By default, a built-in `local` connection and offer accompany remote offers, with -configured offers taking selection priority. If the catalog already defines local -connections, those offers replace the implicit local offer; omit `local.resources` -and `local.time`, and size and time those offers directly. The no-resource shortcut selects the first eligible -local offer and defaults its cluster name to `local`. +By default, a built-in `local` offer follows configured offers, which take +selection priority. If the catalog lists its own local offers, those replace the +built-in offer; omit `local.resources` and `local.time`, and size and time those +offers directly. The no-resource shortcut selects the first eligible local offer +and defaults its cluster name to `local`. Resource requests can select the built-in local offer when no earlier remote offer is eligible. Set `local.enabled: false` for catalogs that must use only remote -compute. Without an explicit local connection, the connection name `local` is -reserved for the built-in backend; its offer name is also reserved while enabled. +compute. While the built-in offer is enabled, the offer name `local` is reserved. -The namespace is a stable UUID identifying a connection; keep it unchanged while -that connection's clusters exist. - -For example, this catalog explicitly defines a local allocation: +For example, this catalog defines its own local offer: ```yaml version: 1 -connections: - workstation: - namespace: 22c84e48-2f0a-4cd2-90a2-30ce2e909bd1 - provider: local offers: - name: workstation - connection: workstation + provider: local resources: {cpus: 4, memory: 8} max_nodes: 1 time: {idle: 30m} - startup: {class: fast} + startup: fast ``` +Allocations launched from the built-in offer stay visible and can still be +stopped after this file exists: both are the same local provider. + Set `LC_COMPUTE_CONFIG` to choose another file for all commands, including `lc run` and `lc materialize`, which find clusters through the same catalog. A missing explicit path or an invalid catalog is an error. An absent implicit default -file uses the default local policy. Stop existing built-in allocations before replacing -their connection namespace with your own. - -This example keeps the built-in connection's namespace, so allocations launched -from the built-in offer stay visible and can still be stopped after the file -exists. +file uses the default local policy. -A catalog has `version: 1`, an optional `local` policy, a `connections` mapping, -and an ordered `offers` list. Connections and offers default to empty: +A catalog has `version: 1`, an optional `connection_root`, an optional `local` +policy, and an ordered `offers` list, empty by default: -- A connection has a `namespace` (a UUID), a `provider` (`local` or `slurm`), an - optional `context`, and optional provider `launch` settings. Namespaces must - be unique, and so must each provider/`context` pair. -- An offer has a unique `name`, the `connection` it uses, per-node `resources` +- `connection_root` (default `~/.lightcone/compute`) holds every allocation's + private scheduler files and credentials, one directory per provider. Stop + running allocations before changing it: allocations under the old root are no + longer found. +- An offer has a unique `name`, its `provider` (`local` or `slurm`), per-node `resources` (`cpus`, `memory`, and optional `accelerators`), `max_nodes`, and `time`. `time` holds an optional walltime `default`, no longer than an optional `max`, and an optional `idle` timeout; it needs a `default` or an `idle`, so every allocation @@ -221,6 +210,9 @@ and an ordered `offers` list. Connections and offers default to empty: `unknown`), written either as a bare class or as `{class: …, source: …}`. `config` holds provider-specific settings. +Each provider reaches one native authority: `local` is this machine, and `slurm` +is the cluster the Slurm client on this machine reaches by default. + Catalog errors identify the invalid field, for example `offers.0.resources.cpus`. Unknown common fields and duplicate YAML keys are rejected. CPU and node counts must be positive integers. Compute memory follows SkyPilot's binary-unit @@ -233,8 +225,8 @@ day/hour/minute/second units, such as Selection takes the first offer in catalog order that matches the request. An offer this host cannot provide is skipped: a local offer with more nodes, CPUs or -memory than the host has, or whose `context` names another host. When nothing -matches, the error lists why each skipped offer was unavailable: +memory than the host has. When nothing matches, the error lists why each skipped +offer was unavailable: ```text Error: no configured offer matches this resource request; see lc compute resources; huge: the local offer exceeds this host's CPU or RAM capacity @@ -244,8 +236,7 @@ Error: no configured offer matches this resource request; see lc compute resourc The CLI runs the native `sbatch`, `salloc`, `squeue`, `sacct`, `scontrol`, and `scancel` commands as the current user. It needs a compatible Slurm client -installation and access to the selected service. `context` is the native Slurm -cluster name; omit it to use the current service. +installation, and it uses the cluster that installation reaches by default. Those commands run without inherited request settings: every `SBATCH_*`, `SALLOC_*`, `SRUN_*`, `SQUEUE_*`, `SACCT_*`, `SCANCEL_*`, and `SLURM_*` variable @@ -260,30 +251,24 @@ resource sizing. It has not been validated by submitting a job at NERSC: version: 1 local: enabled: false -connections: - perlmutter: - namespace: 9d0c0fc5-9be8-407a-a3ec-f17c4110b162 - provider: slurm - context: perlmutter - offers: - name: quick - connection: perlmutter + provider: slurm resources: {cpus: 256, memory: 480} max_nodes: 2 time: {default: 1h, max: 4h} - startup: {class: fast} + startup: fast config: submit: salloc account: myproject constraint: cpu qos: interactive - name: batch - connection: perlmutter + provider: slurm resources: {cpus: 256, memory: 480} max_nodes: 16 time: {default: 1h, max: 12h} - startup: {class: batch} + startup: batch config: submit: sbatch account: myproject @@ -291,23 +276,21 @@ offers: qos: regular ``` -An offer's `config` accepts `submit` (`sbatch`, the default, or `salloc`), -`account`, `partition`, `qos`, `constraint`, `reservation`, and `gpu_type`. +A Slurm offer's `config` accepts `submit` (`sbatch`, the default, or `salloc`), +`account`, `partition`, `qos`, `constraint`, `reservation`, `gpu_type`, and the +launch settings below. For a named accelerator offer, `gpu_type` maps the public type to the site's native Slurm GRES name; it is required even when the spellings happen to match. A generic `GPU` offer may omit it. Slurm offers must state memory as a whole number of MiB. -Every setting under a Slurm connection's `launch` mapping is optional. The -defaults assume a home directory that the login and compute nodes share: +Every launch setting is optional. The defaults assume a home directory that the +login and compute nodes share: - `python`: the interpreter that ran `lc compute launch`, so workers use the driver's own Lightcone installation. `uv tool install lightcone-cli` places it under `$HOME`. Avoid launching through `uvx`, whose environment lives in uv's cache and can be pruned while the allocation runs. -- `connection_root`: `~/.lightcone/compute`. The scheduler's connection files, - TLS credentials, and batch logs live in private directories there, which the - driver and every node must reach. - `scratch_root`: each node's own temporary directory (`$TMPDIR`, usually `/tmp`), which holds the Dask workers' files. - `cwd`: your home directory, as the job's working directory. @@ -317,6 +300,10 @@ defaults assume a home directory that the login and compute nodes share: interface instead if nodes cannot reach each other by hostname; Perlmutter's high-speed network is `hsn0`. +The catalog's top-level `connection_root` (default `~/.lightcone/compute`) must +be reachable from the driver and every node too: the scheduler's connection +files, TLS credentials, and batch logs live in private directories there. + The offered CPU and memory shape is per node. Bare resource quantities request an exact match; a trailing `+` permits a larger offered shape. Selection takes the first eligible offer in catalog order. `--startup fast` filters to that service @@ -385,9 +372,10 @@ Slurm displays `lc-v1-` as the job name, for example `lc-v1-analysis`. Its native comment carries the random submission token as `lightcone:v1:kind=dask:token=<32hex>`. Lightcone verifies the name, token, and owner before attaching or cancelling; a job name alone does not establish identity. -The opaque cluster ID encodes the connection namespace, native job ID, token, -and name. There is no job registry to reconcile. Removing an offer prevents new launches without -hiding existing jobs; retain its connection to inspect and terminate them. +The opaque cluster ID encodes the provider, native job ID, token, and name. +There is no job registry to reconcile. Removing an offer prevents new launches; +`status` and `down` with a full cluster ID still reach the job, while name lookup +and the `status` listing query only the providers the catalog's offers use. Native job state and live Dask readiness are separate observations. A worker loss can leave a job active but not ready. Unknown native state is reported as unknown. @@ -410,7 +398,7 @@ waiting at most ten seconds. Everything lives under the connection root: - Submission logs: `submissions//.out` for `sbatch`, or `submissions//salloc.log`. - Scheduler connection files and TLS credentials: - `/-/attempt-/`. + `slurm/-/attempt-/`. Each worker's files go under `//attempt-/`. @@ -437,7 +425,7 @@ For local GPUs, add an offer to the [workstation catalog above](#customize-resou ```yaml - name: workstation-gpu - connection: workstation + provider: local resources: {cpus: 4, memory: 8GB, accelerators: 'GPU:1'} max_nodes: 1 time: {idle: 30m} @@ -463,7 +451,7 @@ type. For example, add this offer with settings adjusted to your site: ```yaml - name: gpu-batch - connection: perlmutter + provider: slurm resources: {cpus: 32, memory: 128GB, accelerators: 'A100:4'} max_nodes: 2 time: {default: 1h, max: 4h} @@ -577,8 +565,8 @@ Direct recipes inherit the allocation workers' environment, not variables added to the invoking CLI after launch. Remote execution does not forward stdin. The catalog contains policy, not credentials or live state. Scheduler connection -material is private and uses standard Dask TLS and scheduler files. Configured -connection and scratch roots can contain symlinks, including a symlinked home +material is private and uses standard Dask TLS and scheduler files. The configured +`connection_root` and scratch roots can contain symlinks, including a symlinked home directory: Lightcone resolves the root before appending managed paths. Allocation directories and credential files still reject symlinks, retain ownership and ancestor-permission checks, and require modes `0700` and `0600`, respectively. diff --git a/evals/prompt.md b/evals/prompt.md index 7675453d..87dd91d4 100644 --- a/evals/prompt.md +++ b/evals/prompt.md @@ -58,11 +58,11 @@ it alive, and `--time` adds a hard lifetime that ends it even mid-run. After it ends, launch again; the name `local` can be reused, but the immutable ID changes. Compute configuration is `~/.lightcone/compute.yaml`, or the file selected by -`LC_COMPUTE_CONFIG`. Its `local.resources` mapping can override the built-in CPU/RAM -budget (for example, `{cpus: 4, memory: 8GiB}`). Explicit local connections use -their own offers instead. Remote offers otherwise coexist with the default local -offer. If `local.enabled: false` is configured, respect that policy: local launch -and execution are disabled. Inspect `lc compute resources` and supply both +`LC_COMPUTE_CONFIG`. A catalog that lists its own local offers (`provider: local`) +uses those instead of the built-in one, which is how to set a smaller CPU/RAM +budget. Remote offers otherwise coexist with the default local offer. If +`allow_local: false` is configured, respect that policy: local launch and execution +are disabled. Inspect `lc compute resources` and supply both `--cpus` and `--memory` to select a configured remote allocation; `--wait` works there too. The no-resource shortcut never selects remote compute automatically. Recognized NERSC login nodes refuse local compute automatically, even without a diff --git a/src/lightcone/cli/compute.py b/src/lightcone/cli/compute.py index 729b5440..5113b66e 100644 --- a/src/lightcone/cli/compute.py +++ b/src/lightcone/cli/compute.py @@ -30,7 +30,6 @@ def _errors(as_json: bool) -> Iterator[None]: click.echo( json.dumps( { - "schema_version": 1, "error": str(exc), "id": exc.cluster_id, "submission_token": exc.submission_token, @@ -63,7 +62,8 @@ def compute() -> None: Read LC_COMPUTE_CONFIG or ~/.lightcone/compute.yaml. A built-in local offer follows configured offers unless local compute is disabled or - explicit local connections supply their own offers; by default it ends + the catalog supplies its own local offers. It offers this host's CPUs and + memory, plus the GPUs CUDA_VISIBLE_DEVICES lists; by default it ends after 30 minutes without task activity rather than at a fixed age. Local compute is automatically disabled on recognized NERSC login nodes. """ @@ -110,8 +110,9 @@ def resources(as_json: bool) -> None: @click.option("--cpus", help="Logical CPUs per node; suffix + requests a minimum.") @click.option("--memory", help="Memory per node, e.g. 16 or 16GB; suffix + requests a minimum.") -@click.option("--gpus", default="0", show_default=True, - help="Accelerator NAME[:COUNT] per node, e.g. A100:4 or GPU:1; 0 requests CPU only.") +@click.option("--gpus", + help="Accelerator NAME[:COUNT] per node, e.g. A100:4 or GPU:1; 0 requests CPU " + "only, the default except for local compute, which takes the offer's GPUs.") @click.option("--num-nodes", default=1, type=click.IntRange(min=1), show_default=True) @click.option( "--time", "walltime", @@ -129,7 +130,7 @@ def launch( name: str | None, cpus: str | None, memory: str | None, - gpus: str, + gpus: str | None, num_nodes: int, walltime: str | None, startup: str | None, @@ -163,7 +164,7 @@ def launch( ), name=name, ) - data: dict[str, Any] = {"schema_version": 1, "plan": plan.as_dict()} + data: dict[str, Any] = {"plan": plan.as_dict()} if dry_run: if as_json: click.echo(json.dumps(data)) @@ -248,7 +249,6 @@ def status( click.echo( json.dumps( { - "schema_version": 1, "clusters": [item.as_dict() for item in snapshots], "errors": errors, } @@ -277,7 +277,6 @@ def down(cluster_id: str, as_json: bool) -> None: if as_json: click.echo( json.dumps({ - "schema_version": 1, "id": identity.encode(), "name": identity.name, "termination_requested": True, diff --git a/src/lightcone/engine/compute/__init__.py b/src/lightcone/engine/compute/__init__.py index 353b8f93..77f7f686 100644 --- a/src/lightcone/engine/compute/__init__.py +++ b/src/lightcone/engine/compute/__init__.py @@ -6,40 +6,24 @@ import time from collections.abc import Iterator from contextlib import contextmanager +from pathlib import Path from typing import Any from uuid import uuid4 from .catalog import Catalog, local_disabled_reason from .model import ( + PROVIDERS, ComputeError, - Connection, Identity, LaunchPlan, Offer, Provider, - ProviderFactory, Request, Snapshot, UnavailableOfferError, validate_name, ) - -def _local(connection: Connection) -> Provider: - from .local import LocalProvider - - return LocalProvider(connection) - - -def _slurm(connection: Connection) -> Provider: - from .slurm import SlurmProvider - - return SlurmProvider(connection) - - -# The lifecycle seam is intentionally small: execution never dispatches on a provider. -PROVIDERS: dict[str, ProviderFactory] = {"local": _local, "slurm": _slurm} - #: What a driver leaving early must say: closing a client cannot prove that a #: remote subprocess has stopped. UNSTOPPED = ( @@ -61,13 +45,13 @@ class Compute: def __init__(self) -> None: self.catalog = Catalog.load() + self._providers: dict[str, Provider] = {} - def provider(self, connection: Connection) -> Provider: - """Construct an adapter for an explicitly configured native authority.""" - factory = PROVIDERS.get(connection.provider) - if factory is None: - raise ComputeError(f"unsupported compute provider: {connection.provider}") - return factory(connection) + def provider(self, name: str) -> Provider: + """Construct the adapter for one native authority, once per command.""" + if name not in self._providers: + self._providers[name] = PROVIDERS[name](Path(self.catalog.connection_root)) + return self._providers[name] def resolve(self, cluster_id: str) -> tuple[Provider, Identity]: """Route an immutable ID, or resolve one unambiguous name from native state.""" @@ -97,12 +81,11 @@ def resolve(self, cluster_id: str) -> tuple[Provider, Identity]: f"cluster name {cluster_id!r} is ambiguous; use a full cluster ID: {ids}" ) identity = matches.pop() - return self.provider(self.catalog.connection_for(identity.namespace)), identity + return self.provider(identity.provider), identity def resources(self) -> dict[str, Any]: """Describe configured policy, without inventing live free capacity.""" return { - "schema_version": 1, "units": { "cpus": "logical CPUs per node", "memory": "GiB per node", "accelerators": "type and count per node", @@ -144,7 +127,6 @@ def _plan_offer( self, offer: Offer, request: Request, name: str | None, ) -> LaunchPlan | None: """Match one shape and validate its provider without allocating anything.""" - connection = self.catalog.connections[offer.connection] if request.num_nodes > offer.max_nodes: return None if request.startup is not None and request.startup != offer.startup.class_: @@ -168,26 +150,33 @@ def _plan_offer( matches = request.accelerators.matches(offer.resources.accelerators) if not matches: return None - return self.provider(connection).plan(offer, request).replace(name=name) + return self.provider(offer.provider).plan(offer, request).replace(name=name) def plan_local( self, *, name: str | None = None, time: str | None = None, - gpus: str = "0", num_nodes: int = 1, startup: str | None = None, + gpus: str | None = None, num_nodes: int = 1, startup: str | None = None, ) -> LaunchPlan: - """Plan the first usable local offer, without considering remote backends.""" - if reason := local_disabled_reason(self.catalog.local.enabled): + """Plan the first usable local offer, without considering remote backends. + + Without ``gpus``, each offer is taken whole, GPUs included; ``"0"`` takes + it without them, since a local allocation never reserves its GPUs. + """ + if reason := local_disabled_reason(self.catalog.allow_local): raise ComputeError(reason) name = "local" if name is None else name validate_name(name) unavailable: list[str] = [] for offer in self.catalog.offers: - connection = self.catalog.connections[offer.connection] - if connection.provider != "local": + if offer.provider != "local": continue + if gpus == "0": + offer = offer.replace(resources=offer.resources.replace(accelerators=None)) request = Request.parse( str(offer.resources.cpus), f"{offer.resources.memory_bytes}B", gpus=gpus, num_nodes=num_nodes, time=time, startup=startup, ) + if gpus is None: + request = request.replace(accelerators=offer.resources.accelerators) try: plan = self._plan_offer(offer, request, name) if plan is not None: @@ -202,8 +191,8 @@ def plan_local( def launch(self, plan: LaunchPlan) -> Identity: """Choose an unused name from native observations, then submit exactly once.""" - if plan.connection.provider == "local" and ( - reason := local_disabled_reason(self.catalog.local.enabled) + if plan.offer.provider == "local" and ( + reason := local_disabled_reason(self.catalog.allow_local) ): raise ComputeError(reason) if plan.name is not None: @@ -225,15 +214,15 @@ def launch(self, plan: LaunchPlan) -> Identity: raise ComputeError("could not generate an unused cluster name; no allocation made") elif name in names: raise ComputeError(f"cluster name {name!r} is already in use; choose another name") - return self.provider(plan.connection).launch(plan.replace(name=name)) + return self.provider(plan.offer.provider).launch(plan.replace(name=name)) def discover(self) -> tuple[list[Snapshot], dict[str, str]]: """Query each native authority once, retaining partial discovery failures.""" snapshots: list[Snapshot] = [] errors: dict[str, str] = {} - for name, connection in self.catalog.connections.items(): + for name in self.catalog.providers: try: - snapshots.extend(self.provider(connection).discover()) + snapshots.extend(self.provider(name).discover()) except ComputeError as exc: errors[name] = str(exc) return snapshots, errors @@ -247,18 +236,11 @@ def status(self, cluster_id: str, *, wait: bool = False, timeout: float = 300) - # Native schedulers ask users not to poll in a tight loop: back off # from one second, so a long queue wait costs a few dozen queries. delay = 1.0 + unreachable = "" while True: snapshot = provider.inspect(identity) - if snapshot.phase == "active": - remaining = deadline - time.monotonic() - if remaining <= 0: - if not wait: - return snapshot - raise ComputeError( - f"cluster did not become ready within {timeout:g}s; " - "allocation is unchanged", - cluster_id=identity.encode(), - ) + remaining = deadline - time.monotonic() + if snapshot.phase == "active" and remaining > 0: try: with provider.connect(identity, timeout=min(10, remaining)) as client: info = client.scheduler_info() @@ -267,16 +249,18 @@ def status(self, cluster_id: str, *, wait: bool = False, timeout: float = 300) - snapshot.ready = ( snapshot.num_nodes is not None and snapshot.workers >= snapshot.num_nodes ) + unreachable = "" except ComputeError as exc: snapshot.observation = "unreachable" snapshot.ready = False - snapshot.reason = str(exc) + snapshot.reason = unreachable = str(exc) + remaining = deadline - time.monotonic() if not wait or snapshot.ready or snapshot.phase in ("ended", "stopping"): return snapshot - remaining = deadline - time.monotonic() if remaining <= 0: raise ComputeError( - f"cluster did not become ready within {timeout:g}s; allocation is unchanged", + f"cluster did not become ready within {timeout:g}s; allocation is unchanged" + + (f"; last connection attempt: {unreachable}" if unreachable else ""), cluster_id=identity.encode(), ) time.sleep(min(delay, remaining)) @@ -296,9 +280,8 @@ def connect(cluster_id: str, *, timeout: float = 10) -> Iterator[Any]: raise ComputeError("timeout must be finite and positive") service = Compute() provider, identity = service.resolve(cluster_id) - connection = service.catalog.connection_for(identity.namespace) - if connection.provider == "local" and ( - reason := local_disabled_reason(service.catalog.local.enabled) + if identity.provider == "local" and ( + reason := local_disabled_reason(service.catalog.allow_local) ): raise ComputeError(reason, cluster_id=identity.encode()) snapshot = provider.inspect(identity) diff --git a/src/lightcone/engine/compute/catalog.py b/src/lightcone/engine/compute/catalog.py index 19b50995..c53d8caf 100644 --- a/src/lightcone/engine/compute/catalog.py +++ b/src/lightcone/engine/compute/catalog.py @@ -1,33 +1,31 @@ -"""Ordered resource offers and stable native service namespaces.""" +"""Ordered resource offers, each naming the provider that supplies it.""" from __future__ import annotations import os import re import socket +import sys from pathlib import Path from typing import Annotated, Any, Self import yaml -from pydantic import Field, ValidationError, model_validator +from pydantic import AfterValidator, Field, ValidationError, model_validator from .model import ( ComputeError, ComputeModel, - Connection, - Name, Offer, Resources, Startup, + Text, TimeLimits, validation_message, ) +from .runtime import DEFAULT_CONNECTION_ROOT, configured_directory -# Native boot/session evidence identifies the host, independently of its hostname. -_LOCAL_NAMESPACE = "22c84e48-2f0a-4cd2-90a2-30ce2e909bd1" - -def local_disabled_reason(enabled: bool = True) -> str | None: +def local_disabled_reason(allowed: bool = True) -> str | None: """Explain a local-compute refusal while retaining inspection and termination.""" if os.environ.get("NERSC_HOST") and re.fullmatch( r"login[0-9]+", socket.gethostname().split(".", 1)[0].lower(), @@ -36,13 +34,36 @@ def local_disabled_reason(enabled: bool = True) -> str | None: "local compute is disabled on NERSC login nodes; use an interactive compute node " "or configure a Slurm offer and launch with --cpus and --memory" ) - if not enabled: - return "local compute is disabled by the compute configuration" + if not allowed: + return "local compute is disabled by allow_local: false in the compute catalog" return None +def cuda_device_count() -> int: + """Count the GPUs CUDA_VISIBLE_DEVICES exposes on Linux, reading it as CUDA does. + + The mask is the allocation, so nothing probes hardware. CUDA stops at the + first entry that is neither an index nor a device UUID, so ``-1`` exposes none. + """ + if sys.platform != "linux": + return 0 + count = 0 + for entry in os.environ.get("CUDA_VISIBLE_DEVICES", "").split(","): + if not re.fullmatch(r"[0-9]+|(?:GPU|MIG)-\S+", entry.strip()): + break + count += 1 + return count + + +def _resolved_root(value: str) -> str: + try: + return str(configured_directory(Path(value))) + except ComputeError as exc: + raise ValueError(str(exc)) from exc + + class _UniqueLoader(yaml.SafeLoader): - """Do not silently replace a connection or limit through duplicate YAML keys.""" + """Do not silently replace a setting or limit through duplicate YAML keys.""" def _unique_mapping(loader: yaml.SafeLoader, node: yaml.MappingNode) -> dict[str, Any]: @@ -58,51 +79,36 @@ def _unique_mapping(loader: yaml.SafeLoader, node: yaml.MappingNode) -> dict[str _UniqueLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, _unique_mapping) -class LocalSettings(ComputeModel): - """Policy for local launches, including the implicit workstation offer.""" - - enabled: bool = True - resources: Resources | None = None - time: TimeLimits | None = None - - @model_validator(mode="after") - def cpu_only(self) -> Self: - if self.resources is not None and self.resources.gpus: - raise ValueError( - "local.resources supports CPUs and memory; configure GPU offers explicitly" - ) - return self - - class Catalog(ComputeModel): """Configuration for new requests, never a registry of live clusters.""" - version: Annotated[int, Field(ge=1, le=1)] - connections: dict[Name, Connection] = Field(default_factory=dict) + #: Resolved once here, so providers append managed paths to a physical root. + connection_root: Annotated[Text, AfterValidator(_resolved_root)] = Field( + DEFAULT_CONNECTION_ROOT, validate_default=True, + ) offers: list[Offer] = Field(default_factory=list) - local: LocalSettings = Field(default_factory=LocalSettings) + #: False blocks local launch and execution; inspection and termination remain. + allow_local: bool = True @model_validator(mode="after") - def relationships(self) -> Self: - namespaces: set[str] = set() - contexts: set[tuple[str, str]] = set() - for connection in self.connections.values(): - context = (connection.provider, connection.context) - if connection.namespace in namespaces or context in contexts: - raise ValueError("connections must have unique namespaces and native contexts") - namespaces.add(connection.namespace) - contexts.add(context) + def unique_offers(self) -> Self: names: set[str] = set() for offer in self.offers: if offer.name in names: raise ValueError(f"duplicate offer name: {offer.name}") names.add(offer.name) - if offer.connection not in self.connections: - raise ValueError( - f"offer {offer.name} references unknown connection {offer.connection}" - ) return self + @property + def providers(self) -> list[str]: + """Name each native authority to query, local always among them. + + A provider reaches the one authority where lc runs: this host, or + this Slurm environment. Local stays even when disabled, so its + allocations can still be inspected and stopped. + """ + return list(dict.fromkeys([*(offer.provider for offer in self.offers), "local"])) + @classmethod def load(cls, path: Path | None = None) -> Catalog: """Load configured offers and apply the local-compute policy. @@ -114,14 +120,16 @@ def load(cls, path: Path | None = None) -> Catalog: path = path if path is not None else Path( os.environ.get("LC_COMPUTE_CONFIG", "~/.lightcone/compute.yaml") ) - path = path.expanduser() try: + path = path.expanduser() raw = yaml.load(path.read_text(), Loader=_UniqueLoader) + # With no required keys, an empty document is the empty catalog. + raw = {} if raw is None else raw except FileNotFoundError as exc: if configured or path.is_symlink(): raise ComputeError(f"cannot read compute catalog {path}: {exc}") from exc - raw = {"version": 1} - except (OSError, UnicodeError, yaml.YAMLError) as exc: + raw = {} + except (OSError, RuntimeError, UnicodeError, yaml.YAMLError) as exc: raise ComputeError(f"cannot read compute catalog {path}: {exc}") from exc try: catalog = cls.model_validate(raw) @@ -133,53 +141,25 @@ def load(cls, path: Path | None = None) -> Catalog: ) from exc def _with_local(self) -> Catalog: - """Keep explicit local connections, or add the stable built-in connection.""" - connections = dict(self.connections) + """Keep explicit local offers, or add the built-in one with the mask's GPUs.""" offers = list(self.offers) - local = self.local - if local_disabled_reason(local.enabled) is not None: - local = local.replace(enabled=False) - explicit = any(connection.provider == "local" for connection in connections.values()) - if explicit and (local.resources is not None or local.time is not None): - raise ComputeError( - "local.resources and local.time cannot be combined with explicit local " - "connections; set their offers' resources and time instead" - ) - if not explicit: - if "local" in connections: + if local_disabled_reason(self.allow_local): + offers = [offer for offer in offers if offer.provider != "local"] + elif not any(offer.provider == "local" for offer in offers): + if any(offer.name == "local" for offer in offers): raise ComputeError( - "connection name 'local' is reserved for the built-in local backend; " - "rename the configured connection and its offer references" - ) - # Retain this authority even when disabled so existing allocations can be stopped. - connections["local"] = Connection(namespace=_LOCAL_NAMESPACE, provider="local") - if local.enabled: - if any(offer.name == "local" for offer in offers): - raise ComputeError( - "offer name 'local' is reserved for the built-in local backend; " - "rename the configured offer" - ) - from dask.system import CPU_COUNT - from distributed.system import MEMORY_LIMIT - - resources = local.resources or Resources.from_bytes( - cpus=CPU_COUNT, memory_bytes=MEMORY_LIMIT, + "offer name 'local' is reserved for the built-in local backend; " + "rename the configured offer" ) - # Interactive work comes and goes: end when idle, not at a fixed age. - offers.append(Offer( - name="local", connection="local", resources=resources, - max_nodes=1, time=local.time or TimeLimits(idle="30m"), - startup=Startup(class_="fast"), - )) - if not local.enabled: - offers = [ - offer for offer in offers if connections[offer.connection].provider != "local" - ] - return self.replace(connections=connections, offers=offers, local=local) - - def connection_for(self, namespace: str) -> Connection: - """Find the configured authority without relying on current offers.""" - for connection in self.connections.values(): - if connection.namespace == namespace: - return connection - raise ComputeError("cluster's connection namespace is absent from the compute catalog") + from dask.system import CPU_COUNT + from distributed.system import MEMORY_LIMIT + + # Interactive work comes and goes: end when idle, not at a fixed age. + offers.append(Offer( + name="local", provider="local", + resources=Resources.from_bytes( + cpus=CPU_COUNT, memory_bytes=MEMORY_LIMIT, gpus=cuda_device_count(), + ), + max_nodes=1, time=TimeLimits(idle="30m"), startup=Startup(class_="fast"), + )) + return self.replace(offers=offers) diff --git a/src/lightcone/engine/compute/local.py b/src/lightcone/engine/compute/local.py index 32c5ad2c..d71a44ca 100644 --- a/src/lightcone/engine/compute/local.py +++ b/src/lightcone/engine/compute/local.py @@ -21,10 +21,9 @@ import psutil -from lightcone.engine.compute.catalog import local_disabled_reason +from lightcone.engine.compute.catalog import cuda_device_count, local_disabled_reason from lightcone.engine.compute.model import ( ComputeError, - Connection, Identity, LaunchPlan, Offer, @@ -32,11 +31,11 @@ Resources, Snapshot, UnavailableOfferError, + config_text, positive_int, validate_name, ) from lightcone.engine.compute.runtime import ( - DEFAULT_CONNECTION_ROOT, NOT_STARTED, configured_directory, open_client, @@ -134,9 +133,8 @@ def _refuse_a_second_allocation() -> None: else "env -u LC_COMPUTE_CONFIG" ) + " lc compute down " + shlex.quote(cluster_id) raise ComputeError( - f"{message}; allocation {cluster_id} uses local namespace {directory.parent.name!r} " - f"with connection_root {str(directory.parent.parent)!r} " - f"(locator {str(directory.parent)!r}); using that launch catalog, stop it with " + f"{message}; allocation {cluster_id} uses connection_root " + f"{str(directory.parent.parent)!r}; using that launch catalog, stop it with " f"`{command}`" ) @@ -160,12 +158,10 @@ def _boot_identity() -> str: class LocalProvider: """Allocate one cooperative Dask execution node on the current host.""" - def __init__(self, connection: Connection) -> None: - self.connection = connection - root = connection.launch.get("connection_root", DEFAULT_CONNECTION_ROOT) - if not isinstance(root, str) or not root or any(ord(c) < 32 for c in root): - raise ComputeError("local connection_root must be a nonempty path string") - self.root = configured_directory(Path(root)) / connection.namespace + def __init__(self, root: Path) -> None: + # The catalog resolves the connection root; allocations live under local/. + self.root = root + self.allocations = root / "local" def plan(self, offer: Offer, request: Request) -> LaunchPlan: """Validate a one-node local offer without creating allocation files.""" @@ -175,23 +171,17 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: raise ComputeError("local allocations require POSIX process sessions and signals") if request.num_nodes != 1: raise UnavailableOfferError("a local allocation provides exactly one execution node") - if self.connection.context not in ("", socket.gethostname()): - raise UnavailableOfferError("this local connection belongs to a different host") - allowed = {"connection_root", "scratch_root", "python", "task_slots_per_node"} - unknown = self.connection.launch.keys() - allowed - if unknown or offer.config: + config = offer.config + if config.keys() - {"scratch_root", "python", "task_slots_per_node"}: raise ComputeError( - "local offers support only connection_root, scratch_root, " - "python, and task_slots_per_node" + "local offer config supports only scratch_root, python, and task_slots_per_node" ) - for name in ("python", "scratch_root"): - if name in self.connection.launch: - value = self.connection.launch[name] - if not isinstance(value, str) or not value or any(ord(c) < 32 for c in value): - raise ComputeError(f"local {name} must be a nonempty path string") + python_text = config_text(config.get("python", sys.executable), "local python") + scratch_text = config_text( + config.get("scratch_root", tempfile.gettempdir()), "local scratch_root", + ) slots = positive_int( - self.connection.launch.get("task_slots_per_node", offer.resources.cpus), - "task_slots_per_node", + config.get("task_slots_per_node", offer.resources.cpus), "task_slots_per_node", ) if slots > offer.resources.cpus: raise ComputeError("task_slots_per_node exceeds the offered CPU envelope") @@ -204,23 +194,24 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: if offer.resources.gpus: if sys.platform != "linux": raise UnavailableOfferError("local GPU allocations require Linux") - mask = os.environ.get("CUDA_VISIBLE_DEVICES", "") - if not mask: + if (visible := cuda_device_count()) < offer.resources.gpus: raise UnavailableOfferError( - "local GPU offers require an explicit nonempty CUDA_VISIBLE_DEVICES mask" + f"the local offer has {offer.resources.gpus} GPUs, but " + f"CUDA_VISIBLE_DEVICES exposes {visible} on this host" ) + mask = os.environ["CUDA_VISIBLE_DEVICES"] seconds = request.seconds if request.seconds is not None else offer.time.default_seconds limit = offer.time.max_seconds if seconds is not None and limit is not None and seconds > limit: raise ComputeError("the requested time exceeds the local offer's maximum") - python = Path(self.connection.launch.get("python", sys.executable)).expanduser() - scratch = configured_directory( - Path(self.connection.launch.get("scratch_root", tempfile.gettempdir())) - ) + try: + python = Path(python_text).expanduser() + except RuntimeError as exc: + raise ComputeError(f"cannot expand the configured local Python: {exc}") from exc + scratch = configured_directory(Path(scratch_text)) if not python.is_absolute() or not python.is_file() or not os.access(python, os.X_OK): raise ComputeError("the configured local Python must be an executable absolute path") return LaunchPlan( - connection=self.connection, offer=offer, request=request, seconds=seconds, @@ -240,8 +231,6 @@ def launch(self, plan: LaunchPlan) -> Identity: """Start a detached allocation owner and retain its immutable OS identity.""" if reason := local_disabled_reason(): raise ComputeError(reason) - if plan.connection != self.connection or plan.num_nodes != 1: - raise ComputeError("local launch plan belongs to a different connection or node count") if plan.name is not None: validate_name(plan.name) _refuse_a_second_allocation() @@ -261,7 +250,7 @@ def launch(self, plan: LaunchPlan) -> Identity: else: environment["CUDA_DEVICE_ORDER"] = plan.details["cuda_device_order"] try: - directory = private_directory(self.root / token, create=True) + directory = private_directory(self.allocations / token, create=True) scratch = private_directory( Path(plan.details["scratch_root"]) / f"lc-{token}", create=True, ) @@ -292,7 +281,7 @@ def launch(self, plan: LaunchPlan) -> Identity: os.close(startup_read) startup_read = None identity = Identity( - namespace=self.connection.namespace, native_id=str(process.pid), token=token, + provider="local", native_id=str(process.pid), token=token, host=socket.gethostname(), name=plan.name or "", ) @@ -348,12 +337,12 @@ def launch(self, plan: LaunchPlan) -> Identity: def _directory(self, identity: Identity) -> Path: if ( - identity.namespace != self.connection.namespace + identity.provider != "local" or not re.fullmatch(r"[0-9a-f]{32}", identity.token) or not re.fullmatch(r"[1-9][0-9]*", identity.native_id) ): - raise ComputeError("this local allocation does not belong to this connection") - return private_directory(self.root / identity.token) + raise ComputeError("this cluster ID does not identify a local allocation") + return private_directory(self.allocations / identity.token) def _record(self, identity: Identity) -> tuple[Path, dict[str, Any]]: directory = self._directory(identity) @@ -427,12 +416,12 @@ def _process( def discover(self) -> Sequence[Snapshot]: """Find active local owners by checking their private locators against the OS.""" - if not self.root.exists(): + if not self.allocations.exists(): return [] - private_directory(self.root) + private_directory(self.allocations) boot = _boot_identity() snapshots = [] - for directory in sorted(self.root.iterdir()): + for directory in sorted(self.allocations.iterdir()): if not re.fullmatch(r"[0-9a-f]{32}", directory.name): continue if (directory / _RETIRED).exists(): diff --git a/src/lightcone/engine/compute/model.py b/src/lightcone/engine/compute/model.py index 3c89a0cb..29455db7 100644 --- a/src/lightcone/engine/compute/model.py +++ b/src/lightcone/engine/compute/model.py @@ -8,8 +8,8 @@ from collections.abc import Callable, Sequence from contextlib import AbstractContextManager from decimal import Decimal, localcontext +from pathlib import Path from typing import Annotated, Any, Literal, Protocol, Self -from uuid import UUID from pydantic import ( AfterValidator, @@ -86,9 +86,20 @@ def positive_int(value: object, name: str) -> int: return int(str(value)) +def config_text(value: object, name: str) -> str: + """Accept a nonempty offer setting without control characters.""" + if not isinstance(value, str) or not value or any(ord(char) < 32 for char in value): + raise ComputeError(f"{name} must be a nonempty string without control characters") + return value + + +#: A cluster name: portable across native providers and safe in commands. +NAME_PATTERN = r"[a-z](?:[a-z0-9-]{0,61}[a-z0-9])?" + + def validate_name(value: str) -> None: """Keep cluster names portable across native providers and safe in commands.""" - if not re.fullmatch(r"[a-z](?:[a-z0-9-]{0,61}[a-z0-9])?", value): + if not re.fullmatch(NAME_PATTERN, value): raise ComputeError( "cluster names must be 1–63 lowercase letters, digits, or hyphens; " "start with a letter and end with a letter or digit" @@ -109,15 +120,6 @@ def _name(value: str) -> str: return value -def _namespace(value: str) -> str: - try: - if str(UUID(value)) == value: - return value - except ValueError: - pass - raise ValueError("must be a canonical UUID") - - def _count(value: object) -> int: try: return positive_int(value, "count") @@ -141,9 +143,15 @@ def _duration(value: str) -> str: return value +def _provider(value: str) -> str: + if value not in PROVIDERS: + raise ValueError(f"must name a supported provider: {', '.join(PROVIDERS)}") + return value + + Text = Annotated[str, Field(pattern=r"^[^\x00-\x1f]*$")] Name = Annotated[Text, AfterValidator(_name)] -Namespace = Annotated[str, AfterValidator(_namespace)] +ProviderName = Annotated[str, AfterValidator(_provider)] PositiveInt = Annotated[int, Field(gt=0, strict=True)] Count = Annotated[int, BeforeValidator(_count, json_schema_input_type=int | str)] GiB = Annotated[ @@ -261,17 +269,17 @@ def parse( cpus: str, memory: str, *, - gpus: str = "0", + gpus: str | None = None, num_nodes: int = 1, time: str | None = None, startup: str | None = None, ) -> Request: - """Parse the CLI's exact or minimum per-node quantities.""" + """Parse the CLI's exact or minimum per-node quantities; no ``gpus`` is CPU only.""" try: return cls.model_validate({ "cpus": positive_int(cpus.removesuffix("+"), "cpus"), "memory_bytes": memory_bytes(memory.removesuffix("+")), - "accelerators": None if gpus == "0" else gpus, + "accelerators": None if gpus in (None, "0") else gpus, "num_nodes": positive_int(num_nodes, "num_nodes"), "min_cpus": cpus.endswith("+"), "min_memory": memory.endswith("+"), @@ -297,15 +305,6 @@ def as_dict(self) -> dict[str, Any]: } -class Connection(ComputeModel): - """One native authority, independent of the offers that reference it.""" - - namespace: Namespace - provider: Name - context: Text = "" - launch: dict[str, Any] = Field(default_factory=dict) - - class TimeLimits(ComputeModel): """How an allocation ends: a hard walltime, an idle timeout, or both. @@ -358,7 +357,7 @@ class Offer(ComputeModel): """A fixed resource shape with user-supplied limits and native bindings.""" name: Name - connection: Name + provider: ProviderName resources: Resources max_nodes: Count time: TimeLimits @@ -369,7 +368,7 @@ class Offer(ComputeModel): class Identity(ComputeModel): """A self-contained native identity; its encoding is not a credential.""" - namespace: Namespace + provider: ProviderName native_id: Name token: Name host: Text = "" @@ -384,7 +383,7 @@ def generated_name(self) -> Self: def encode(self) -> str: """Encode identity without a local UUID-to-allocation database.""" payload = json.dumps( - [1, self.namespace, self.native_id, self.token, self.host, self.name], + [self.provider, self.native_id, self.token, self.host, self.name], separators=(",", ":"), ).encode() return "clu_" + base64.urlsafe_b64encode(payload).decode().rstrip("=") @@ -397,18 +396,13 @@ def decode(cls, value: str) -> Identity: raise ValueError raw = value[4:] data = json.loads(base64.b64decode(raw + "=" * (-len(raw) % 4), altchars=b"-_")) - if not isinstance(data, list) or len(data) != 6 or type(data[0]) is not int: - raise ValueError - version, namespace, native_id, token, host, name = data - if version != 1 or not all(isinstance(item, str) for item in data[1:]): - raise ValueError - if str(UUID(namespace)) != namespace or not native_id or not token: - raise ValueError - if any(ord(char) < 32 for item in data[1:] for char in item): + if not isinstance(data, list) or len(data) != 5: raise ValueError + provider, native_id, token, host, name = data validate_name(name) + # The fields' own strict types refuse the rest; ValidationError is a ValueError. identity = cls( - namespace=namespace, native_id=native_id, token=token, host=host, name=name + provider=provider, native_id=native_id, token=token, host=host, name=name ) if identity.encode() != value: raise ValueError @@ -422,7 +416,6 @@ def decode(cls, value: str) -> Identity: class LaunchPlan(ComputeModel): """Resolved immutable sizing and nonsecret provider launch parameters.""" - connection: Connection offer: Offer request: Request seconds: PositiveInt | None @@ -444,11 +437,10 @@ def num_nodes(self) -> int: def as_dict(self) -> dict[str, Any]: """Allowlist the plan's public contract; details must contain no credentials.""" return { - "schema_version": 1, "name": self.name, "request": self.request.as_dict(), "offer": self.offer.name, - "connection": self.offer.connection, + "provider": self.offer.provider, "num_nodes": self.num_nodes, "resources": self.resources.as_dict(), "time_seconds": self.seconds, @@ -478,7 +470,6 @@ class Snapshot(ComputeModel): def as_dict(self) -> dict[str, Any]: """Keep endpoint credentials and native response objects private.""" return { - "schema_version": 1, "id": self.identity.encode(), "name": self.identity.name, "phase": self.phase, @@ -506,4 +497,21 @@ def connect( def terminate(self, identity: Identity) -> None: ... -ProviderFactory = Callable[[Connection], Provider] +#: Builds a provider from the catalog's resolved connection root. +ProviderFactory = Callable[[Path], Provider] + + +def _local(root: Path) -> Provider: + from .local import LocalProvider + + return LocalProvider(root) + + +def _slurm(root: Path) -> Provider: + from .slurm import SlurmProvider + + return SlurmProvider(root) + + +# The lifecycle seam is intentionally small: execution never dispatches on a provider. +PROVIDERS: dict[str, ProviderFactory] = {"local": _local, "slurm": _slurm} diff --git a/src/lightcone/engine/compute/runtime.py b/src/lightcone/engine/compute/runtime.py index 32c0a339..368763c9 100644 --- a/src/lightcone/engine/compute/runtime.py +++ b/src/lightcone/engine/compute/runtime.py @@ -43,13 +43,16 @@ def configured_directory(path: Path) -> Path: Raises: ComputeError: If the root is relative, contains ``..``, or cannot resolve. """ - path = path.expanduser() - if not path.is_absolute() or ".." in path.parts: - raise ComputeError(f"compute root must be an absolute path without '..': {path}") try: - return path.resolve() - except (OSError, RuntimeError) as exc: + expanded = path.expanduser() + except RuntimeError as exc: raise ComputeError(f"cannot resolve compute root {path}: {exc}") from exc + if not expanded.is_absolute() or ".." in expanded.parts: + raise ComputeError(f"compute root must be an absolute path without '..': {expanded}") + try: + return expanded.resolve() + except (OSError, RuntimeError) as exc: + raise ComputeError(f"cannot resolve compute root {expanded}: {exc}") from exc def private_directory(path: Path, *, create: bool = False) -> Path: diff --git a/src/lightcone/engine/compute/slurm.py b/src/lightcone/engine/compute/slurm.py index d5a13bd8..a00ed1aa 100644 --- a/src/lightcone/engine/compute/slurm.py +++ b/src/lightcone/engine/compute/slurm.py @@ -16,19 +16,19 @@ from typing import Any from lightcone.engine.compute.model import ( + NAME_PATTERN, ComputeError, - Connection, Identity, LaunchPlan, Offer, Request, Resources, Snapshot, + config_text, positive_int, validate_name, ) from lightcone.engine.compute.runtime import ( - DEFAULT_CONNECTION_ROOT, NOT_STARTED, configured_directory, open_client, @@ -36,10 +36,10 @@ read_private_json, ) -_PREFIX = "lc-v1-" -_COMMENT_PREFIX = "lightcone:v1:kind=dask:token=" +_PREFIX = "lc-" +_COMMENT_PREFIX = "lightcone:kind=dask:token=" _TOKEN = re.compile(r"[0-9a-f]{32}") -_JOB_NAME = re.compile(r"lc-v1-([a-z](?:[a-z0-9-]{0,61}[a-z0-9])?)") +_JOB_NAME = re.compile(f"{re.escape(_PREFIX)}({NAME_PATTERN})") _QUERY_TIMEOUT = 10.0 _SUBMIT_TIMEOUT = 60.0 _ACCEPT_TIMEOUT = 10.0 @@ -70,17 +70,19 @@ def native_environment() -> dict[str, str]: } -def attempt_directory(connection: Connection, identity: Identity, restarts: int) -> Path: - """Locate connection material for one native allocation incarnation and attempt.""" - root = configured_directory( - Path(str(connection.launch.get("connection_root", DEFAULT_CONNECTION_ROOT))) - ) - return ( - root - / connection.namespace - / f"{identity.native_id}-{identity.token}" - / f"attempt-{restarts}" - ) +def allocation_directory(root: Path, token: str) -> Path: + """Locate one submission's private directory under a resolved connection root. + + It holds the submission's log and each attempt's connection material. Launch + creates it before submitting, so a running job without one was launched + under another connection root. + """ + return root / "slurm" / token + + +def attempt_directory(root: Path, token: str, restarts: int) -> Path: + """Locate one native attempt's connection material; the bootstrap writes, connect reads.""" + return allocation_directory(root, token) / f"attempt-{restarts}" def _phase(state: str) -> str: @@ -94,12 +96,6 @@ def _phase(state: str) -> str: ) -def _value(value: object, name: str) -> str: - if not isinstance(value, str) or not value or any(ord(char) < 32 for char in value): - raise ComputeError(f"Slurm {name} must be a nonempty string without control characters") - return value - - def _native_gpus(row: Mapping[str, str]) -> tuple[str, int] | None: """Read a per-node GPU count, never divide an aggregate into invented grants.""" for field in ("TresPerNode", "Gres"): @@ -141,10 +137,11 @@ def _native_gpus(row: Mapping[str, str]) -> tuple[str, int] | None: class SlurmProvider: - """Submit, observe, and cancel allocations using the selected Slurm authority.""" + """Submit, observe, and cancel allocations in the Slurm environment lc runs in.""" - def __init__(self, connection: Connection) -> None: - self.connection = connection + def __init__(self, root: Path) -> None: + # The catalog resolves the connection root; allocations live under slurm/. + self.root = root @cached_property def _uid(self) -> int: @@ -154,14 +151,6 @@ def _uid(self) -> int: raise ComputeError("id -u did not return a numeric Slurm command user ID") return int(value) - def _scope(self) -> list[str]: - context = self.connection.context - if not context: - return [] - if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]*", context) or context == "all": - raise ComputeError("Slurm context must name one native cluster") - return [f"--clusters={context}"] - def _command( self, argv: list[str], *, payload: str | None = None, timeout: float = _QUERY_TIMEOUT ) -> subprocess.CompletedProcess[str]: @@ -185,38 +174,6 @@ def _command( def plan(self, offer: Offer, request: Request) -> LaunchPlan: """Freeze one fixed, homogeneous allocation and its standard Dask launcher.""" - launch = self.connection.launch - allowed = { - "python", - "connection_root", - "scratch_root", - "task_slots_per_node", - "interface", - "cwd", - } - if extra := launch.keys() - allowed: - raise ComputeError(f"unknown Slurm launch settings: {', '.join(sorted(extra))}") - # The defaults assume a home directory shared by login and compute - # nodes: workers run the driver's own installation, so they match it - # exactly, and rendezvous under its home. Scratch left unset is chosen - # by each node, whose temporary directory may not be the driver's. - defaults = { - "python": sys.executable, - "connection_root": DEFAULT_CONNECTION_ROOT, - "cwd": str(Path.home()), - } - paths: dict[str, str | None] = {} - for name in ("python", "connection_root", "scratch_root", "cwd"): - if name not in launch and name not in defaults: - paths[name] = None - continue - value = launch.get(name, defaults.get(name)) - path = Path(_value(value, name)) - if name in {"connection_root", "scratch_root"}: - path = configured_directory(path) - elif not path.is_absolute() or ".." in path.parts: - raise ComputeError(f"Slurm {name} must be an absolute path without '..'") - paths[name] = str(path) config = offer.config if extra := config.keys() - { "submit", @@ -226,8 +183,26 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: "constraint", "reservation", "gpu_type", + "python", + "scratch_root", + "task_slots_per_node", + "interface", + "cwd", }: raise ComputeError(f"unknown Slurm offer settings: {', '.join(sorted(extra))}") + # The defaults assume a home directory shared by login and compute + # nodes: workers run the driver's own installation, so they match it + # exactly, and rendezvous under its home. Scratch left unset is chosen + # by each node, whose temporary directory may not be the driver's. + paths: dict[str, str | None] = {"connection_root": str(self.root), "scratch_root": None} + if "scratch_root" in config: + scratch = config_text(config["scratch_root"], "Slurm scratch_root") + paths["scratch_root"] = str(configured_directory(Path(scratch))) + for name, default in (("python", sys.executable), ("cwd", str(Path.home()))): + path = Path(config_text(config.get(name, default), f"Slurm {name}")) + if not path.is_absolute() or ".." in path.parts: + raise ComputeError(f"Slurm {name} must be an absolute path without '..'") + paths[name] = str(path) submit = config.get("submit", "sbatch") if submit not in {"sbatch", "salloc"}: raise ComputeError("Slurm submit must be sbatch or salloc") @@ -249,15 +224,15 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: if memory % _MIB: raise ComputeError("Slurm offer memory must be an exact whole number of MiB") slots = positive_int( - launch.get("task_slots_per_node", max(1, cpus - 1)), "task_slots_per_node" + config.get("task_slots_per_node", max(1, cpus - 1)), "task_slots_per_node" ) if slots > cpus: raise ComputeError( "Slurm task_slots_per_node cannot exceed the allocation CPU envelope" ) - interface = launch.get("interface") + interface = config.get("interface") if interface is not None: - interface = _value(interface, "interface") + interface = config_text(interface, "Slurm interface") seconds = request.seconds or offer.time.default_seconds if seconds is None or offer.time.idle is not None: raise ComputeError( @@ -266,12 +241,12 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: ) hours, remainder = divmod(seconds, 3600) minutes, seconds_part = divmod(remainder, 60) - args = self._scope() + args: list[str] = [] for name in ("account", "qos", "constraint", "reservation"): if name in config: - args.append(f"--{name}={_value(config[name], name)}") + args.append(f"--{name}={config_text(config[name], f'Slurm {name}')}") if "partition" in config: - partition = _value(config["partition"], "partition") + partition = config_text(config["partition"], "Slurm partition") if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]*", partition): raise ComputeError("Slurm partition must name one native partition") args.append(f"--partition={partition}") @@ -286,7 +261,6 @@ def plan(self, offer: Offer, request: Request) -> LaunchPlan: if offer.resources.gpus: args.append(f"--gres={gres}") return LaunchPlan( - connection=self.connection, offer=offer, request=request, seconds=seconds, @@ -319,10 +293,8 @@ def _payload(self, plan: LaunchPlan, token: str) -> list[str]: "lightcone.engine.compute.slurm_bootstrap", "--submission", token, - "--namespace", - self.connection.namespace, "--connection-root", - details["connection_root"], + str(self.root), "--num-nodes", str(plan.num_nodes), "--cpus", @@ -342,16 +314,12 @@ def _payload(self, plan: LaunchPlan, token: str) -> list[str]: def launch(self, plan: LaunchPlan) -> Identity: """Submit once; preserve the nonce when native acceptance is uncertain.""" - if plan.connection != self.connection: - raise ComputeError("Slurm launch plan belongs to another connection") token = uuid.uuid4().hex name = plan.name if plan.name is not None else f"lc-{token[:12]}" validate_name(name) job_name = f"{_PREFIX}{name}" details = plan.details - logs = private_directory( - Path(details["connection_root"]) / "submissions" / token, create=True - ) + logs = private_directory(allocation_directory(self.root, token), create=True) common = [ *details["native_args"], f"--job-name={job_name}", @@ -363,15 +331,12 @@ def launch(self, plan: LaunchPlan) -> Identity: script = "#!/bin/bash\nset -euo pipefail\numask 077\nexec " + shlex.join(payload) + "\n" try: result = self._command(argv, payload=script, timeout=_SUBMIT_TIMEOUT) - match = re.fullmatch(r"([0-9]+)(?:;([^;\s]+))?", result.stdout.strip()) - if match and ( - not self.connection.context or match[2] in (None, self.connection.context) - ): + native_id = result.stdout.strip() + if re.fullmatch(r"[0-9]+", native_id): return Identity( - namespace=self.connection.namespace, native_id=match[1], token=token, - name=name, + provider="slurm", native_id=native_id, token=token, name=name, ) - reason = "sbatch did not return an unambiguous allocation ID" + reason = f"sbatch printed {native_id!r}, not one numeric allocation ID" except ComputeError as exc: reason = str(exc) return self._recover_or_raise(token, name, reason) @@ -427,14 +392,13 @@ def _recover_or_raise(self, token: str, name: str, reason: str) -> Identity: def _live(self, native_id: str | None = None) -> list[dict[str, str]]: argv = [ "squeue", - *self._scope(), "--noheader", f"--user={self._uid}", "--format=%i|%128j|%U|%T|%128k", ] rows = [] for line in self._command(argv).stdout.splitlines(): - if not line.strip() or line.startswith("CLUSTER:"): + if not line.strip(): continue fields = [value.strip() for value in line.split("|", 4)] if len(fields) != 5: @@ -455,7 +419,6 @@ def _history( ) -> list[dict[str, str]]: argv = [ "sacct", - *self._scope(), "--noheader", "--parsable2", "--allocations", @@ -491,10 +454,7 @@ def _find_token(self, token: str, name: str, *, history: bool = False) -> list[I rows = self._live() job_name = f"{_PREFIX}{name}" matches = { - Identity( - namespace=self.connection.namespace, native_id=row["JobId"], token=token, - name=name, - ) + Identity(provider="slurm", native_id=row["JobId"], token=token, name=name) for row in rows if row["JobName"] == job_name and row["UID"] == str(self._uid) and row["Comment"] == _COMMENT_PREFIX + token @@ -503,10 +463,7 @@ def _find_token(self, token: str, name: str, *, history: bool = False) -> list[I return list(matches) return list( { - Identity( - namespace=self.connection.namespace, native_id=row["JobId"], token=token, - name=name, - ) + Identity(provider="slurm", native_id=row["JobId"], token=token, name=name) for row in self._history(job_name=job_name) if row["JobName"] == job_name and row["UID"] == str(self._uid) and row["Comment"] == _COMMENT_PREFIX + token @@ -515,14 +472,14 @@ def _find_token(self, token: str, name: str, *, history: bool = False) -> list[I def _validate_identity(self, identity: Identity) -> None: if ( - identity.namespace != self.connection.namespace + identity.provider != "slurm" or identity.host or not re.fullmatch(r"[0-9]+", identity.native_id) or not _TOKEN.fullmatch(identity.token) or not _JOB_NAME.fullmatch(_PREFIX + identity.name) ): raise ComputeError( - "cluster ID does not identify an allocation on this Slurm connection" + "cluster ID does not identify a Slurm allocation" ) def _validate_row(self, identity: Identity, row: Mapping[str, str]) -> None: @@ -538,9 +495,7 @@ def _validate_row(self, identity: Identity, row: Mapping[str, str]) -> None: ) def _control(self, identity: Identity) -> dict[str, str]: - result = self._command( - ["scontrol", *self._scope(), "show", "job", identity.native_id] - ) + result = self._command(["scontrol", "show", "job", identity.native_id]) rows = [line.strip() for line in result.stdout.splitlines() if line.strip()] if sum(row.startswith("JobId=") for row in rows) != 1: raise ComputeError("Slurm did not return exactly one allocation record") @@ -599,10 +554,7 @@ def discover(self) -> Sequence[Snapshot]: f"Slurm job {row['JobId']} ({row['JobName']}) has a missing or malformed " "submission token in Comment; its allocation identity is unknown" ) - identity = Identity( - namespace=self.connection.namespace, native_id=row["JobId"], token=token, - name=name, - ) + identity = Identity(provider="slurm", native_id=row["JobId"], token=token, name=name) self._validate_row(identity, row) snapshots.append(self._snapshot(identity, self._control(identity))) return snapshots @@ -658,12 +610,17 @@ def connect(self, identity: Identity, *, timeout: float = 10) -> Iterator[Any]: if not restart_text.isdigit(): raise ComputeError("Slurm did not identify the current allocation attempt") restarts = int(restart_text) - directory = attempt_directory(self.connection, identity, restarts) + directory = attempt_directory(self.root, identity.token, restarts) + if not directory.parent.is_dir(): + raise ComputeError( + f"this allocation has no directory under connection_root {self.root}: it was " + "launched with another connection_root, or its directory was removed", + cluster_id=identity.encode(), + ) if not (directory / "identity.json").exists(): raise ComputeError(NOT_STARTED, cluster_id=identity.encode()) metadata = read_private_json(directory / "identity.json") expected = { - "namespace": identity.namespace, "native_id": identity.native_id, "token": identity.token, "uid": self._uid, @@ -712,7 +669,6 @@ def terminate(self, identity: Identity) -> None: [ "scancel", "--ctld", - *self._scope(), f"--user={self._uid}", f"--name={_PREFIX}{identity.name}", identity.native_id, diff --git a/src/lightcone/engine/compute/slurm_bootstrap.py b/src/lightcone/engine/compute/slurm_bootstrap.py index 904b63ef..1b179808 100644 --- a/src/lightcone/engine/compute/slurm_bootstrap.py +++ b/src/lightcone/engine/compute/slurm_bootstrap.py @@ -12,9 +12,8 @@ import time from pathlib import Path from typing import Any -from uuid import UUID -from lightcone.engine.compute.model import ComputeError, Connection, Identity +from lightcone.engine.compute.model import ComputeError, Identity from lightcone.engine.compute.runtime import ( SCHEDULER_CONFIG, configured_directory, @@ -33,8 +32,6 @@ def _allocation(args: argparse.Namespace) -> tuple[Identity, int, int]: """Reject missing or conflicting native placement before creating any endpoint.""" if not re.fullmatch(r"[0-9a-f]{32}", args.submission): raise ComputeError("invalid submission token") - if str(UUID(args.namespace)) != args.namespace: - raise ComputeError("invalid connection namespace") native_id = os.environ.get("SLURM_JOB_ID", "") if not re.fullmatch(r"[0-9]+", native_id): raise ComputeError("Dask bootstrap requires a native Slurm allocation") @@ -72,7 +69,7 @@ def _allocation(args: argparse.Namespace) -> tuple[Identity, int, int]: if not restarts.isdigit(): raise ComputeError("invalid native Slurm restart count") return ( - Identity(namespace=args.namespace, native_id=native_id, token=args.submission), + Identity(provider="slurm", native_id=native_id, token=args.submission), int(restarts), values["SLURM_PROCID"], ) @@ -88,17 +85,14 @@ async def run(args: argparse.Namespace) -> None: os.environ["CUDA_DEVICE_ORDER"] = "PCI_BUS_ID" else: os.environ["CUDA_VISIBLE_DEVICES"] = "" - connection = Connection( - namespace=identity.namespace, provider="slurm", - launch={"connection_root": args.connection_root}, + directory = attempt_directory( + configured_directory(Path(args.connection_root)), identity.token, restarts, ) - directory = attempt_directory(connection, identity, restarts) scratch = configured_directory(Path(args.scratch_root or tempfile.gettempdir())) scratch = private_directory( scratch / identity.token / f"attempt-{restarts}" / str(rank), create=True ) identity_values: dict[str, Any] = { - "namespace": identity.namespace, "native_id": identity.native_id, "token": identity.token, "uid": os.getuid(), @@ -172,7 +166,7 @@ async def run(args: argparse.Namespace) -> None: def main() -> None: """Read frozen launcher arguments; a failure becomes a nonzero native task exit.""" parser = argparse.ArgumentParser(description=__doc__) - for name in ("submission", "namespace", "connection-root"): + for name in ("submission", "connection-root"): parser.add_argument(f"--{name}", required=True) for name in ("num-nodes", "cpus", "memory-bytes", "task-slots"): parser.add_argument(f"--{name}", required=True, type=int) diff --git a/tests/conftest.py b/tests/conftest.py index c6541b4f..2dad8b87 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -18,9 +18,7 @@ from lightcone.engine.plan import Key, Task from lightcone.engine.project import _run as _real_run -CLUSTER_ID = Identity( - namespace="00000000-0000-0000-0000-000000000001", native_id="test", token="test" -).encode() +CLUSTER_ID = Identity(provider="local", native_id="test", token="test").encode() @pytest.fixture @@ -45,6 +43,8 @@ def local_allocation_scope( local, "_running_owners", lambda: [o for o in owners() if o[1].is_relative_to(root)], ) monkeypatch.delenv("NERSC_HOST", raising=False) + # The built-in local offer takes its GPUs from this mask. + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) return root diff --git a/tests/test_compute.py b/tests/test_compute.py index 471c57fe..5ce93df1 100644 --- a/tests/test_compute.py +++ b/tests/test_compute.py @@ -16,12 +16,11 @@ from lightcone.cli.commands import main from lightcone.engine import compute -from lightcone.engine.compute.catalog import Catalog +from lightcone.engine.compute.catalog import Catalog, local_disabled_reason from lightcone.engine.compute.model import ( GIB, Accelerator, ComputeError, - Connection, Identity, LaunchPlan, Offer, @@ -36,8 +35,7 @@ validate_name, ) -NAMESPACE = "5a9d058c-7c6e-4e2a-919b-786f1148536c" -IDENTITY = Identity(namespace=NAMESPACE, native_id="1234", token="abc") +IDENTITY = Identity(provider="slurm", native_id="1234", token="abc") @pytest.fixture @@ -54,22 +52,24 @@ def expand(path: Path) -> Path: monkeypatch.setattr(Path, "expanduser", expand) monkeypatch.delenv("LC_COMPUTE_CONFIG", raising=False) monkeypatch.delenv("NERSC_HOST", raising=False) + monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) return tmp_path @pytest.fixture -def catalog(default_home: Path, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: +def catalog( + default_home: Path, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: MagicMock, +) -> Path: + # Its offers name slurm, which the stub `provider` replaces: nothing here runs Slurm. path = tmp_path / "compute.yaml" path.write_text( yaml.safe_dump( { - "version": 1, - "local": {"enabled": False}, - "connections": {"test": {"namespace": NAMESPACE, "provider": "fake"}}, + "allow_local": False, "offers": [ { "name": name, - "connection": "test", + "provider": "slurm", "resources": {"cpus": cpus, "memory": memory}, "max_nodes": nodes, "time": {"default": "30m", "max": "2h"}, @@ -91,7 +91,6 @@ def catalog(default_home: Path, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) def provider(monkeypatch: pytest.MonkeyPatch) -> MagicMock: adapter = MagicMock() adapter.plan.side_effect = lambda offer, request: LaunchPlan( - connection=compute.Compute().catalog.connections["test"], offer=offer, request=request, seconds=request.seconds or offer.time.default_seconds, @@ -102,7 +101,7 @@ def provider(monkeypatch: pytest.MonkeyPatch) -> MagicMock: client = MagicMock() client.scheduler_info.return_value = {"workers": {"one": {}}} adapter.connect.return_value.__enter__.return_value = client - monkeypatch.setitem(compute.PROVIDERS, "fake", lambda connection: adapter) + monkeypatch.setitem(compute.PROVIDERS, "slurm", lambda root: adapter) return adapter @@ -257,17 +256,10 @@ def test_missing_default_catalog_exposes_stable_local_resources_without_writing_ monkeypatch.setattr("distributed.system.MEMORY_LIMIT", GIB) first, second = Catalog.load(), Catalog.load() assert first == second - assert set(first.connections) == {"local"} - connection = first.connections["local"] - assert connection.provider == "local" - assert str(UUID(connection.namespace)) == connection.namespace - with monkeypatch.context() as patch: - patch.setattr("socket.gethostname", lambda: "other-host") - assert Catalog.load().connections["local"].namespace == connection.namespace - assert Catalog.load().connections["local"].namespace == connection.namespace + assert first.providers == ["local"] assert len(first.offers) == 1 offer = first.offers[0] - assert (offer.name, offer.connection) == ("local", "local") + assert (offer.name, offer.provider) == ("local", "local") assert (offer.resources.cpus, offer.resources.memory_bytes, offer.max_nodes) == (1, GIB, 1) # No hard lifetime: the built-in offer ends after 30 minutes without task activity. assert (offer.time.default_seconds, offer.time.max_seconds, offer.time.idle_seconds) == ( @@ -308,11 +300,12 @@ def test_configured_catalogs_can_disable_local_and_obey_path_precedence( default.parent.mkdir() default.write_text(catalog.read_text()) configured = Catalog.load() - assert set(configured.connections) == {"test", "local"} + assert configured.providers == ["slurm", "local"] assert [offer.name for offer in configured.offers] == ["quick", "large"] - # Disabled local connections remain available for inspection and termination. - default.write_text("version: 1\nlocal: {enabled: false}\nconnections: {}\noffers: []\n") + # Disabled local compute remains available for inspection and termination. + default.write_text("allow_local: false\noffers: []\n") assert Catalog.load().offers == [] + assert Catalog.load().providers == ["local"] monkeypatch.setenv("LC_COMPUTE_CONFIG", str(catalog)) assert Catalog.load() == configured assert Catalog.load(default).offers == [] @@ -335,10 +328,10 @@ def test_only_an_absent_implicit_catalog_uses_the_builtin( with pytest.raises(ComputeError, match="cannot read compute catalog"): Catalog.load() default.unlink() - default.write_text("version: [\n") + default.write_text("offers: [\n") with pytest.raises(ComputeError, match="cannot read compute catalog"): Catalog.load() - default.write_text("version: 1\nconnections: {}\noffers: []\n") + default.write_text("offers: []\n") def unreadable(_path: Path, *args: object, **kwargs: object) -> str: raise PermissionError("catalog is not readable") @@ -469,20 +462,52 @@ def test_accelerator_selection_honors_type_and_exact_count( compute.Compute().plan(Request.parse("4", "8", gpus="A100:4")) -def test_builtin_catalog_stays_cpu_only_without_probing_native_gpus( - default_home: Path, monkeypatch: pytest.MonkeyPatch, +@pytest.mark.parametrize("mask, gpus", [ + (None, 0), ("", 0), ("-1", 0), ("0,1", 2), ("GPU-8932f937,MIG-1c2d", 2), ("3,-1,0", 1), +]) +def test_builtin_offer_takes_the_gpus_its_mask_exposes_without_probing_hardware( + default_home: Path, monkeypatch: pytest.MonkeyPatch, mask: str | None, gpus: int, ) -> None: - monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1") + monkeypatch.setattr("sys.platform", "linux") + if mask is not None: + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) probe = MagicMock(side_effect=AssertionError("catalog loading must not probe GPU hardware")) monkeypatch.setattr(subprocess, "run", probe) - loaded = Catalog.load() - assert [(offer.name, offer.resources.gpus) for offer in loaded.offers] == [ - ("local", 0), - ] + (offer,) = Catalog.load().offers + assert (offer.name, offer.resources.gpus) == ("local", gpus) + assert offer.resources.accelerator_name == ("GPU" if gpus else None) probe.assert_not_called() assert not list(default_home.iterdir()) +def test_builtin_offer_has_no_gpus_outside_linux( + default_home: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr("sys.platform", "darwin") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") + assert Catalog.load().offers[0].resources.gpus == 0 + + +def test_local_shortcut_takes_the_offer_whole_unless_cpu_only_is_asked( + default_home: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr("sys.platform", "linux") + monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1") + service = compute.Compute() + whole = service.plan_local() + assert whole.resources.gpus == 2 + assert whole.details["cuda_visible_devices"] == "0,1" + assert service.plan_local(gpus="GPU:2").resources.gpus == 2 + cpu = service.plan_local(gpus="0") + assert cpu.resources.gpus == 0 + assert cpu.details["cuda_visible_devices"] == "" + with pytest.raises(ComputeError, match="no local offer"): + service.plan_local(gpus="GPU:1") + result = CliRunner().invoke(main, ["compute", "launch", "--dry-run", "--json"]) + assert result.exit_code == 0, result.output + assert json.loads(result.stdout)["plan"]["resources"]["accelerators"] == {"GPU": 2} + + def test_local_shortcut_uses_detected_capacity_without_writing_files( default_home: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -511,8 +536,8 @@ def test_nersc_login_nodes_block_first_launch_without_writing_a_catalog( monkeypatch.setenv("SLURM_JOB_ID", "123") monkeypatch.setattr("socket.gethostname", lambda: "login07.nersc.gov") service = compute.Compute() - assert not service.catalog.local.enabled - assert service.catalog.connections["local"] == plan.connection + assert local_disabled_reason(service.catalog.allow_local) + assert service.catalog.providers == ["local"] assert service.resources()["offers"] == [] for flags in ([], ["--dry-run"]): result = CliRunner().invoke(main, ["compute", "launch", *flags, "--json"]) @@ -543,14 +568,10 @@ def test_local_compute_remains_available_outside_identified_nersc_login_nodes( def test_nersc_login_guard_keeps_remote_compute_and_local_inspection_available( catalog: Path, provider: MagicMock, monkeypatch: pytest.MonkeyPatch, ) -> None: - connection = compute.Compute().catalog.connections["local"] - identity = IDENTITY.replace(namespace=connection.namespace, native_id="5678") + identity = IDENTITY.replace(provider="local", native_id="5678") data = yaml.safe_load(catalog.read_text()) - data["local"] = {"enabled": True} - data["connections"]["workstation"] = connection.model_dump() - data["offers"].insert(0, { - **data["offers"][0], "name": "workstation", "connection": "workstation", - }) + data["allow_local"] = True + data["offers"].insert(0, {**data["offers"][0], "name": "workstation", "provider": "local"}) catalog.write_text(yaml.safe_dump(data)) local = MagicMock() local.discover.return_value = [] @@ -558,11 +579,11 @@ def test_nersc_login_guard_keeps_remote_compute_and_local_inspection_available( local.connect.return_value.__enter__.return_value.scheduler_info.return_value = { "workers": {"one": {}}, } - monkeypatch.setitem(compute.PROVIDERS, "local", lambda connection: local) + monkeypatch.setitem(compute.PROVIDERS, "local", lambda root: local) monkeypatch.setenv("NERSC_HOST", "perlmutter") monkeypatch.setattr("socket.gethostname", lambda: "login07") service = compute.Compute() - assert not service.catalog.local.enabled + assert local_disabled_reason(service.catalog.allow_local) assert [offer.name for offer in service.catalog.offers] == ["quick", "large"] plan = service.plan(Request.parse("4", "8")) assert service.launch(plan) == IDENTITY @@ -580,23 +601,25 @@ def test_nersc_login_guard_keeps_remote_compute_and_local_inspection_available( local.launch.assert_not_called() -def test_local_config_overrides_default_and_survives_disabling( +def test_an_explicit_local_offer_replaces_the_builtin_and_allow_local_disables_both( default_home: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: monkeypatch.setattr("dask.system.CPU_COUNT", 8) monkeypatch.setattr("distributed.system.MEMORY_LIMIT", 16 * GIB) path = default_home / "compute.yaml" - path.write_text("version: 1\nlocal:\n resources: {cpus: 2, memory: 3}\n") + offer = "{name: local, provider: local, resources: {cpus: 2, memory: 3}, max_nodes: 1, " + offer += "time: {idle: 30m}}" + path.write_text(f"offers:\n - {offer}\n") monkeypatch.setenv("LC_COMPUTE_CONFIG", str(path)) service = compute.Compute() + assert [offer.name for offer in service.catalog.offers] == ["local"] plan = service.plan_local(name="sandbox", time="1h") assert (plan.name, plan.resources.cpus, plan.resources.memory_bytes) == ("sandbox", 2, 3 * GIB) assert plan.seconds == 3600 - connection = plan.connection - path.write_text("version: 1\nlocal: {enabled: false}\n") + path.write_text(f"allow_local: false\noffers:\n - {offer}\n") disabled = compute.Compute() assert not disabled.resources()["offers"] - assert disabled.catalog.connection_for(connection.namespace) == connection + assert disabled.catalog.providers == ["local"] with pytest.raises(ComputeError, match="disabled"): disabled.plan_local() with pytest.raises(ComputeError, match="disabled"): @@ -609,48 +632,41 @@ def test_catalog_adds_local_after_remote_offers_and_shortcut_never_selects_remot catalog: Path, provider: MagicMock, monkeypatch: pytest.MonkeyPatch, ) -> None: data = yaml.safe_load(catalog.read_text()) - data.pop("local") + data.pop("allow_local") catalog.write_text(yaml.safe_dump(data)) monkeypatch.setattr("dask.system.CPU_COUNT", 4) monkeypatch.setattr("distributed.system.MEMORY_LIMIT", 8 * GIB) service = compute.Compute() assert [offer.name for offer in service.catalog.offers] == ["quick", "large", "local"] assert service.plan(Request.parse("4", "8")).offer.name == "quick" - assert service.plan_local().connection.provider == "local" + assert service.plan_local().offer.provider == "local" with pytest.raises(ComputeError, match="no local offer"): service.plan_local(num_nodes=2) provider.launch.assert_not_called() -@pytest.mark.parametrize("kind, enabled", [ - ("connection", True), ("connection", False), ("offer", True), -]) -def test_builtin_name_conflicts_identify_the_catalog_and_remedy( - catalog: Path, kind: str, enabled: bool, -) -> None: +def test_builtin_name_conflict_identifies_the_catalog_and_remedy(catalog: Path) -> None: data = yaml.safe_load(catalog.read_text()) - data["local"] = {"enabled": enabled} - if kind == "connection": - data["connections"]["local"] = data["connections"].pop("test") - for offer in data["offers"]: - offer["connection"] = "local" - else: - data["offers"][0]["name"] = "local" + data["allow_local"] = True + data["offers"][0]["name"] = "local" catalog.write_text(yaml.safe_dump(data)) with pytest.raises(ComputeError) as error: Catalog.load() assert f"invalid compute catalog {catalog}" in str(error.value) - assert f"{kind} name 'local' is reserved for the built-in local backend" in str(error.value) - assert f"rename the configured {kind}" in str(error.value) + assert "offer name 'local' is reserved for the built-in local backend" in str(error.value) + assert "rename the configured offer" in str(error.value) -def test_local_time_replaces_the_builtin_idle_timeout( +def test_an_explicit_local_offer_sets_its_own_time_limits( default_home: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: monkeypatch.setattr("dask.system.CPU_COUNT", 8) monkeypatch.setattr("distributed.system.MEMORY_LIMIT", 16 * GIB) path = default_home / "compute.yaml" - path.write_text("version: 1\nlocal:\n time: {idle: 1h, max: 8h}\n") + path.write_text( + "offers:\n - {name: local, provider: local, resources: {cpus: 8, memory: 16}, " + "max_nodes: 1, time: {idle: 1h, max: 8h}}\n" + ) monkeypatch.setenv("LC_COMPUTE_CONFIG", str(path)) service = compute.Compute() plan = service.plan_local() @@ -670,36 +686,35 @@ def test_disabled_builtin_does_not_reserve_remote_offer_names(catalog: Path) -> catalog.write_text(yaml.safe_dump(data)) loaded = Catalog.load() assert [offer.name for offer in loaded.offers] == ["local", "large"] - assert loaded.connections[loaded.offers[0].connection].provider == "fake" + assert loaded.offers[0].provider == "slurm" def test_explicit_local_offers_keep_their_sizes_and_replace_the_implicit_offer( catalog: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: data = yaml.safe_load(catalog.read_text()) - data["local"] = {"enabled": True} - data["connections"]["test"]["provider"] = "local" + data["allow_local"] = True + for offer in data["offers"]: + offer["provider"] = "local" catalog.write_text(yaml.safe_dump(data)) monkeypatch.setattr("dask.system.CPU_COUNT", 8) monkeypatch.setattr("distributed.system.MEMORY_LIMIT", 16 * GIB) service = compute.Compute() - assert set(service.catalog.connections) == {"test"} + assert [offer.name for offer in service.catalog.offers] == ["quick", "large"] plan = service.plan_local() assert (plan.offer.name, plan.name, plan.resources.cpus) == ("quick", "local", 4) assert plan.resources.memory_bytes == 8 * GIB - for setting, value in ( - ("resources", {"cpus": 2, "memory": 2}), ("time", {"idle": "1h"}), - ): - catalog.write_text(yaml.safe_dump({**data, "local": {setting: value}})) - with pytest.raises(ComputeError, match="cannot be combined with explicit local"): - Catalog.load() def test_configured_local_budget_still_must_fit_host_capacity( catalog: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: data = yaml.safe_load(catalog.read_text()) - data["local"] = {"resources": {"cpus": 8, "memory": 16}} + data["allow_local"] = True + data["offers"] = [{ + "name": "local", "provider": "local", "resources": {"cpus": 8, "memory": 16}, + "max_nodes": 1, "time": {"idle": "30m"}, + }] catalog.write_text(yaml.safe_dump(data)) monkeypatch.setattr("dask.system.CPU_COUNT", 4) monkeypatch.setattr("distributed.system.MEMORY_LIMIT", 8 * GIB) @@ -708,16 +723,12 @@ def test_configured_local_budget_still_must_fit_host_capacity( assert "exceeds this host's CPU or RAM capacity" in json.loads(result.stdout)["error"] -@pytest.mark.parametrize("settings, message", [ - ({"enabled": "false"}, "local.enabled"), - ({"resources": {"cpus": 0, "memory": 1}}, "local.resources.cpus"), - ({"resources": {"cpus": 1, "memory": 1, "accelerators": "GPU:1"}}, "GPU offers"), -]) -def test_local_policy_validation(catalog: Path, settings: object, message: str) -> None: +@pytest.mark.parametrize("value", ["false", 0, None]) +def test_allow_local_must_be_a_boolean(catalog: Path, value: object) -> None: data = yaml.safe_load(catalog.read_text()) - data["local"] = settings + data["allow_local"] = value catalog.write_text(yaml.safe_dump(data)) - with pytest.raises(ComputeError, match=message): + with pytest.raises(ComputeError, match="allow_local"): Catalog.load() @@ -725,18 +736,20 @@ def test_disabled_policy_blocks_local_execution_but_allows_status_and_down( catalog: Path, provider: MagicMock, monkeypatch: pytest.MonkeyPatch, ) -> None: data = yaml.safe_load(catalog.read_text()) - data["connections"]["test"]["provider"] = "local" + for offer in data["offers"]: + offer["provider"] = "local" catalog.write_text(yaml.safe_dump(data)) - monkeypatch.setitem(compute.PROVIDERS, "local", lambda connection: provider) + monkeypatch.setitem(compute.PROVIDERS, "local", lambda root: provider) + identity = IDENTITY.replace(provider="local") service = compute.Compute() assert service.catalog.offers == [] with pytest.raises(ComputeError, match="disabled"): - with compute.connect(IDENTITY.encode()): + with compute.connect(identity.encode()): pytest.fail("borrowed disabled local compute") provider.connect.assert_not_called() - assert service.status(IDENTITY.encode()).ready - service.down(IDENTITY.encode()) - provider.terminate.assert_called_once_with(IDENTITY) + assert service.status(identity.encode()).ready + service.down(identity.encode()) + provider.terminate.assert_called_once_with(identity) def test_configured_catalogs_do_not_probe_local_gpus( @@ -823,7 +836,6 @@ def test_catalog_uses_the_public_models_and_roundtrips_without_an_adapter(catalo offer = loaded.offers[0] for instance, model in ( (loaded, Catalog), - (loaded.connections["test"], Connection), (offer, Offer), (offer.resources, Resources), (offer.time, TimeLimits), @@ -831,7 +843,6 @@ def test_catalog_uses_the_public_models_and_roundtrips_without_an_adapter(catalo ): assert isinstance(instance, BaseModel) assert type(instance) is model - assert "name" not in loaded.connections["test"].model_dump() dumped = loaded.model_dump(by_alias=True) assert dumped["offers"][0]["resources"] == { "cpus": 4, "memory": Decimal(8), "accelerators": None, @@ -853,7 +864,7 @@ def test_model_updates_revalidate_fields_and_catalog_relationships(catalog: Path with pytest.raises(ValidationError): offer.time.replace(default="3h") with pytest.raises(ValidationError): - loaded.replace(offers=[offer.replace(connection="missing")]) + loaded.replace(offers=[offer, offer]) with pytest.raises(ValidationError): Request(cpus=1, memory_bytes=GIB).replace(memory_bytes=0) @@ -887,8 +898,7 @@ def test_selection_skips_known_ineligibility_but_stops_on_an_unknown_authority( ) -> None: service = compute.Compute() selected = LaunchPlan( - connection=service.catalog.connections["test"], offer=service.catalog.offers[1], - request=Request.parse("1+", "1+"), seconds=1800, + offer=service.catalog.offers[1], request=Request.parse("1+", "1+"), seconds=1800, ) provider.plan.side_effect = [UnavailableOfferError("login host"), selected] assert service.plan(Request.parse("1+", "1+")).offer.name == "large" @@ -913,7 +923,7 @@ def test_discovery_preserves_unknown_authority(catalog: Path, provider: MagicMoc provider.discover.side_effect = ComputeError("native service unavailable") clusters, errors = compute.Compute().discover() assert clusters == [] - assert errors == {"test": "native service unavailable"} + assert errors == {"slurm": "native service unavailable"} def test_live_allocation_is_not_automatically_ready(catalog: Path, provider: MagicMock) -> None: @@ -922,7 +932,8 @@ def test_live_allocation_is_not_automatically_ready(catalog: Path, provider: Mag assert result.phase == "active" assert result.ready is False assert result.observation == "unreachable" - with pytest.raises(ComputeError, match="allocation is unchanged"): + # A wait that runs out names why the scheduler could not be reached. + with pytest.raises(ComputeError, match="last connection attempt: scheduler unavailable"): compute.Compute().status(IDENTITY.encode(), wait=True, timeout=0.01) provider.terminate.assert_not_called() @@ -990,30 +1001,23 @@ def test_borrowed_client_only_detaches(catalog: Path, provider: MagicMock) -> No @pytest.mark.parametrize("mutation", [ - "version_missing", "namespace", "context", "offer", "limits", "unbounded", "reference", - "unknown", "resources_extra", "time_extra", "startup_extra", "connection_extra", + "provider_missing", "provider", "offer", "limits", "unbounded", + "unknown", "resources_extra", "time_extra", "startup_extra", "catalog_extra", ]) def test_invalid_catalog_is_rejected(catalog: Path, mutation: str) -> None: data = yaml.safe_load(catalog.read_text()) - if mutation == "version_missing": - data.pop("version") - elif mutation == "namespace": - data["connections"]["other"] = {"namespace": NAMESPACE, "provider": "other"} - elif mutation == "context": - data["connections"]["other"] = { - "namespace": "8613532d-378c-43f0-bf3a-132895093d6e", - "provider": "fake", - } + if mutation == "provider_missing": + data["offers"][0].pop("provider") + elif mutation == "provider": + data["offers"][0]["provider"] = "Not a provider" elif mutation == "offer": data["offers"][1]["name"] = "quick" elif mutation == "limits": data["offers"][0]["time"]["default"] = "3h" elif mutation == "unbounded": data["offers"][0]["time"] = {"max": "2h"} - elif mutation == "reference": - data["offers"][0]["connection"] = "missing" - elif mutation == "connection_extra": - data["connections"]["test"]["extra"] = 1 + elif mutation == "catalog_extra": + data["connections"] = {} elif mutation.endswith("_extra"): data["offers"][0][mutation.removesuffix("_extra")]["extra"] = 1 else: @@ -1024,10 +1028,7 @@ def test_invalid_catalog_is_rejected(catalog: Path, mutation: str) -> None: @pytest.mark.parametrize("field,value", [ - (("version",), True), - (("version",), 1.0), - (("version",), "1"), - (("connections", "test", "namespace"), NAMESPACE.upper()), + (("connection_root",), 1), (("offers", 0, "resources", "cpus"), True), (("offers", 0, "resources", "cpus"), 4.0), (("offers", 0, "max_nodes"), 1.0), @@ -1076,33 +1077,80 @@ def test_catalog_normalizes_units_and_startup_without_changing_offer_order( def test_catalog_keeps_provider_payloads_opaque(catalog: Path) -> None: data = yaml.safe_load(catalog.read_text()) - launch = {"future-setting": {"nested": [True, 2, None, "value"]}} - config = {"future-option": ["native", {"enabled": False}]} - data["connections"]["test"].update(provider="future-provider", launch=launch) - data["offers"][0]["config"] = config + config = {"future-option": ["native", {"enabled": False, "nested": [True, 2, None]}]} + data["offers"][0].update(config=config) catalog.write_text(yaml.safe_dump(data)) - loaded = Catalog.load(catalog) - assert loaded.connections["test"].provider == "future-provider" - assert loaded.connections["test"].launch == launch - assert loaded.offers[0].config == config + assert Catalog.load(catalog).offers[0].config == config + + +def test_a_misspelled_provider_is_a_catalog_error_not_a_discovery_failure( + catalog: Path, +) -> None: + data = yaml.safe_load(catalog.read_text()) + data["offers"][1]["provider"] = "slurmm" + catalog.write_text(yaml.safe_dump(data)) + result = CliRunner().invoke(main, ["compute", "status", "--json"]) + assert result.exit_code == 1 + error = json.loads(result.output)["error"] + assert "invalid compute catalog" in error + assert "offers.1.provider: Value error, must name a supported provider" in error + + +@pytest.mark.parametrize("root, message", [ + ("relative/compute", "compute root must be an absolute path"), + ("/tmp/../compute", "compute root must be an absolute path"), + ("~nosuchuser42/compute", "cannot resolve compute root"), +]) +def test_an_unusable_connection_root_is_a_catalog_error( + catalog: Path, root: str, message: str, +) -> None: + data = yaml.safe_load(catalog.read_text()) + data["connection_root"] = root + catalog.write_text(yaml.safe_dump(data)) + result = CliRunner().invoke(main, ["compute", "status", "--json"]) + assert result.exit_code == 1 + error = json.loads(result.output)["error"] + assert "invalid compute catalog" in error + assert f"connection_root: Value error, {message}" in error + + +def test_the_connection_root_is_resolved_once_and_handed_to_one_provider_per_name( + catalog: Path, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + physical = tmp_path / "physical" + physical.mkdir() + (tmp_path / "alias").symlink_to(physical, target_is_directory=True) + data = yaml.safe_load(catalog.read_text()) + data["connection_root"] = str(tmp_path / "alias" / "compute") + catalog.write_text(yaml.safe_dump(data)) + roots: list[Path] = [] + factory = compute.PROVIDERS["slurm"] + monkeypatch.setitem( + compute.PROVIDERS, "slurm", lambda root: roots.append(root) or factory(root), + ) + service = compute.Compute() + assert service.catalog.connection_root == str(physical / "compute") + service.discover() + service.launch(service.plan(Request.parse("4", "8"))) + assert roots == [physical / "compute"] def test_catalog_validation_errors_do_not_echo_provider_values(catalog: Path) -> None: data = yaml.safe_load(catalog.read_text()) secret = "private-provider-credential" - data["connections"]["test"]["launch"] = secret + data["offers"][0]["config"] = secret catalog.write_text(yaml.safe_dump(data)) result = CliRunner().invoke(main, ["compute", "resources", "--json"]) assert result.exit_code == 1 error = json.loads(result.output)["error"] - assert "connections.test.launch" in error + assert "offers.0.config" in error assert secret not in error @pytest.mark.parametrize("document", [ - "version: 1\nversion: 1\nconnections: {}\noffers: []\n", - "connections: {test: {launch: {option: 1, option: 2}}}\n", - "connections: {test: {launch: {1: value}}}\n", + "offers: []\noffers: []\n", + "offers: [{config: {option: 1, option: 2}}]\n", + "offers: [{config: {1: value}}]\n", ]) def test_duplicate_or_nonstring_yaml_keys_are_rejected(catalog: Path, document: str) -> None: catalog.write_text(document) @@ -1111,7 +1159,7 @@ def test_duplicate_or_nonstring_yaml_keys_are_rejected(catalog: Path, document: def test_invalid_catalog_encoding_is_a_structured_error(catalog: Path) -> None: - catalog.write_bytes(b"version: 1\n\xff") + catalog.write_bytes(b"offers: []\n\xff") with pytest.raises(ComputeError, match="cannot read compute catalog"): Catalog.load(catalog) result = CliRunner().invoke(main, ["compute", "resources", "--json"]) @@ -1119,18 +1167,19 @@ def test_invalid_catalog_encoding_is_a_structured_error(catalog: Path) -> None: assert "cannot read compute catalog" in json.loads(result.output)["error"] -@pytest.mark.parametrize("setting", ["connection_root", "scratch_root", "python"]) -def test_local_catalog_path_types_fail_without_a_traceback(catalog: Path, setting: str) -> None: +@pytest.mark.parametrize("setting", ["scratch_root", "python"]) +def test_local_offer_path_types_fail_without_a_traceback(catalog: Path, setting: str) -> None: data = yaml.safe_load(catalog.read_text()) - data["local"] = {"enabled": True} - data["connections"]["test"]["provider"] = "local" - data["connections"]["test"]["launch"] = {setting: None} + data["allow_local"] = True + for offer in data["offers"]: + offer["provider"] = "local" + data["offers"][0]["config"] = {setting: None} catalog.write_text(yaml.safe_dump(data)) result = CliRunner().invoke( main, ["compute", "launch", "--cpus", "4", "--memory", "8", "--dry-run", "--json"] ) assert result.exit_code == 1 - assert "path string" in json.loads(result.output)["error"] + assert f"local {setting} must be a nonempty string" in json.loads(result.output)["error"] @pytest.mark.parametrize("timeout", ["nan", "inf"]) @@ -1156,7 +1205,7 @@ def test_cli_resources_dry_run_launch_down(catalog: Path, provider: MagicMock) - assert result.exit_code == 0, result.output plan = json.loads(result.output)["plan"] assert plan["offer"] == "quick" - assert plan["connection"] == "test" + assert plan["provider"] == "slurm" assert plan["resources"] == {"cpus": 4, "memory": 8, "accelerators": None} assert plan["time_seconds"] == 1800 assert plan["startup"] == "fast" @@ -1271,7 +1320,7 @@ def test_cli_partial_failure_and_ambiguous_submit(catalog: Path, provider: Magic runner = CliRunner() result = runner.invoke(main, ["compute", "status", "--json"]) assert result.exit_code == 1 - assert json.loads(result.output)["errors"] == {"test": "unavailable"} + assert json.loads(result.output)["errors"] == {"slurm": "unavailable"} provider.discover.side_effect = None provider.launch.side_effect = ComputeError("uncertain", submission_token="token") result = runner.invoke(main, ["compute", "launch", "--cpus", "4", "--memory", "8", "--json"]) diff --git a/tests/test_compute_local.py b/tests/test_compute_local.py index 138b51c0..80a29c63 100644 --- a/tests/test_compute_local.py +++ b/tests/test_compute_local.py @@ -30,7 +30,6 @@ from lightcone.engine.compute.local import LocalProvider from lightcone.engine.compute.model import ( ComputeError, - Connection, Identity, Offer, Request, @@ -51,26 +50,30 @@ @pytest.fixture def provider(tmp_path: Path) -> LocalProvider: - return LocalProvider( - Connection( - namespace=str(uuid4()), - provider="local", - launch={ - "connection_root": str(tmp_path / "connections"), - "scratch_root": str(tmp_path / "scratch"), - }, - ) + return LocalProvider(tmp_path / "connections") + + +def _scratch(provider: LocalProvider) -> Path: + """Scratch for test launches, beside the provider's configured connection root.""" + return provider.root.with_name("scratch") + + +def _offer( + provider: LocalProvider, *, idle: str | None = None, scratch: Path | None = None, +) -> Offer: + return Offer( + name="small", provider="local", resources=Resources(cpus=1, memory_gib=0.5), + max_nodes=1, + time=TimeLimits(default="1m", max="1m") if idle is None else TimeLimits(idle=idle), + config={"scratch_root": str(scratch or _scratch(provider))}, ) def _launch( provider: LocalProvider, *, seconds: int | None = 60, idle: str | None = None, + scratch: Path | None = None, ) -> Identity: - offer = Offer( - name="small", connection="workstation", resources=Resources(cpus=1, memory_gib=0.5), - max_nodes=1, - time=TimeLimits(default="1m", max="1m") if idle is None else TimeLimits(idle=idle), - ) + offer = _offer(provider, idle=idle, scratch=scratch) return provider.launch( provider.plan(offer, Request(cpus=1, memory_bytes=512 * 1024**2, seconds=seconds)) ) @@ -82,31 +85,29 @@ def test_singleton_survives_cli_exit_and_spans_names_and_connection_roots( # An independent launcher counts only this session's owners, like the fixture. script = """ import sys +from pathlib import Path from lightcone.engine.compute import local -from lightcone.engine.compute.model import Connection, Offer, Request, Resources, TimeLimits +from lightcone.engine.compute.model import Offer, Request owners = local._running_owners local._running_owners = lambda: [o for o in owners() if o[1].is_relative_to(sys.argv[2])] -p = local.LocalProvider(Connection.model_validate_json(sys.argv[1])) -offer = Offer(name='small', connection='workstation', resources=Resources(cpus=1, memory_gib=0.5), - max_nodes=1, time=TimeLimits(default='1m', max='1m')) +p = local.LocalProvider(Path(sys.argv[1])) +offer = Offer.model_validate_json(sys.argv[3]) plan = p.plan(offer, Request(cpus=1, memory_bytes=512 * 1024**2)).replace(name='first') print(p.launch(plan).encode()) """ launched = subprocess.run( - [sys.executable, "-c", script, provider.connection.model_dump_json(), - str(local_allocation_scope)], + [sys.executable, "-c", script, str(provider.root), + str(local_allocation_scope), _offer(provider).model_dump_json(by_alias=True)], capture_output=True, text=True, timeout=20, check=True, ) identity = Identity.decode(launched.stdout.strip()) - other = LocalProvider(provider.connection.replace( - namespace=str(uuid4()), launch={"connection_root": str(tmp_path / "other")}, - )) + other = LocalProvider(tmp_path / "other") identities = [(provider, identity)] try: _ready(provider, identity) # The owner is found by its command, never by its Dask worker processes. assert local._running_owners() == [ - (int(identity.native_id), provider.root / identity.token), + (int(identity.native_id), provider.allocations / identity.token), ] with pytest.raises(ComputeError, match="already running") as conflict: _launch(other) @@ -201,12 +202,11 @@ def test_launch_survives_closed_standard_descriptors( import json, os, sys from pathlib import Path from lightcone.engine.compute import local -from lightcone.engine.compute.model import Connection, Offer, Request, Resources, TimeLimits +from lightcone.engine.compute.model import Offer, Request owners = local._running_owners local._running_owners = lambda: [o for o in owners() if o[1].is_relative_to(sys.argv[2])] -p = local.LocalProvider(Connection.model_validate_json(sys.argv[1])) -offer = Offer(name='small', connection='workstation', resources=Resources(cpus=1, memory_gib=0.5), - max_nodes=1, time=TimeLimits(default='1m', max='1m')) +p = local.LocalProvider(Path(sys.argv[1])) +offer = Offer.model_validate_json(sys.argv[5]) plan = p.plan(offer, Request(cpus=1, memory_bytes=512 * 1024**2)) for descriptor in json.loads(sys.argv[4]): os.close(descriptor) @@ -215,8 +215,9 @@ def test_launch_survives_closed_standard_descriptors( """ try: result = subprocess.run( - [sys.executable, "-c", script, provider.connection.model_dump_json(), - str(local_allocation_scope), str(result_path), json.dumps(closed)], + [sys.executable, "-c", script, str(provider.root), + str(local_allocation_scope), str(result_path), json.dumps(closed), + _offer(provider).model_dump_json(by_alias=True)], capture_output=True, text=True, timeout=15, ) assert result.returncode == 0, result.stderr @@ -276,22 +277,20 @@ def test_allocation_survives_launcher_and_borrowed_client_exit( provider: LocalProvider, local_allocation_scope: Path, ) -> None: script = """ -import json, sys +import sys +from pathlib import Path from lightcone.engine.compute import local from lightcone.engine.compute.local import LocalProvider -from lightcone.engine.compute.model import Connection, Offer, Request, Resources, TimeLimits +from lightcone.engine.compute.model import Offer, Request owners = local._running_owners local._running_owners = lambda: [o for o in owners() if o[1].is_relative_to(sys.argv[2])] -p = LocalProvider(Connection(**json.loads(sys.argv[1]))) -offer = Offer( - name='small', connection='workstation', resources=Resources(cpus=1, memory_gib=0.5), - max_nodes=1, time=TimeLimits(default='1m', max='1m'), -) +p = LocalProvider(Path(sys.argv[1])) +offer = Offer.model_validate_json(sys.argv[3]) print(p.launch(p.plan(offer, Request(cpus=1, memory_bytes=512 * 1024**2))).encode()) """ launched = subprocess.run( - [sys.executable, "-c", script, json.dumps(provider.connection.model_dump()), - str(local_allocation_scope)], + [sys.executable, "-c", script, str(provider.root), + str(local_allocation_scope), _offer(provider).model_dump_json(by_alias=True)], check=True, capture_output=True, text=True, @@ -320,17 +319,17 @@ def test_allocation_survives_launcher_and_borrowed_client_exit( pid not in (os.getpid(), int(identity.native_id)) for pid in worker_pids.values() ) second = """ -import json, sys +import sys +from pathlib import Path from lightcone.engine.compute.local import LocalProvider -from lightcone.engine.compute.model import Connection, Identity -p = LocalProvider(Connection(**json.loads(sys.argv[1]))) +from lightcone.engine.compute.model import Identity +p = LocalProvider(Path(sys.argv[1])) with p.connect(Identity.decode(sys.argv[2])) as client: print(client.submit(sum, [10, 11]).result(timeout=5)) """ result = subprocess.run( [ - sys.executable, "-c", second, - json.dumps(provider.connection.model_dump()), identity.encode(), + sys.executable, "-c", second, str(provider.root), identity.encode(), ], check=True, capture_output=True, @@ -378,7 +377,7 @@ def test_builtin_allocation_can_be_reopened_in_another_process_without_a_catalog ) assert result.stdout.strip() == "9" assert not (tmp_path / ".lightcone" / "compute.yaml").exists() - builtin = LocalProvider(Compute().catalog.connections["local"]) + builtin = LocalProvider(Path(Compute().catalog.connection_root)) with monkeypatch.context() as other_catalog: other_catalog.setenv("LC_COMPUTE_CONFIG", str(tmp_path / "other.yaml")) with pytest.raises(ComputeError, match="already running") as conflict: @@ -401,7 +400,7 @@ def test_cpu_allocation_without_gpu_fields_remains_discoverable_and_stoppable( identity = _launch(provider) try: _ready(provider, identity) - path = provider.root / identity.token / "identity.json" + path = provider.allocations / identity.token / "identity.json" record = read_private_json(path) del record["gpus"], record["accelerator_name"] write_private_json(path, record) @@ -447,21 +446,8 @@ def test_named_local_allocation_is_discovered_and_name_can_be_reused_after_down( catalog = tmp_path / "compute.yaml" catalog.write_text(json.dumps({ - "version": 1, - "connections": { - "workstation": { - "provider": "local", - "namespace": provider.connection.namespace, - "launch": provider.connection.launch, - }, - }, - "offers": [{ - "name": "small", - "connection": "workstation", - "resources": {"cpus": 1, "memory": 0.5}, - "max_nodes": 1, - "time": {"default": "1m", "max": "1m"}, - }], + "connection_root": str(provider.root), + "offers": [_offer(provider).model_dump(mode="json", by_alias=True)], })) monkeypatch.setenv("LC_COMPUTE_CONFIG", str(catalog)) service = Compute() @@ -473,7 +459,9 @@ def test_named_local_allocation_is_discovered_and_name_can_be_reused_after_down( assert first.name == "analysis" assert Identity.decode(first.encode()) == first # A new adapter reconstructs the name from the existing native locator. - assert [item.identity for item in LocalProvider(provider.connection).discover()] == [first] + assert [item.identity for item in LocalProvider(provider.root).discover()] == [ + first, + ] assert Compute().status("analysis", wait=True, timeout=20).identity == first with monkeypatch.context() as other_catalog: other_catalog.setenv("LC_COMPUTE_CONFIG", str(tmp_path / "other.yaml")) @@ -514,21 +502,8 @@ def test_an_unused_allocation_idles_out_despite_status_polling_and_frees_its_nam ) -> None: catalog = tmp_path / "compute.yaml" catalog.write_text(json.dumps({ - "version": 1, - "connections": { - "workstation": { - "provider": "local", - "namespace": provider.connection.namespace, - "launch": provider.connection.launch, - }, - }, - "offers": [{ - "name": "small", - "connection": "workstation", - "resources": {"cpus": 1, "memory": 0.5}, - "max_nodes": 1, - "time": {"idle": "8s"}, - }], + "connection_root": str(provider.root), + "offers": [_offer(provider, idle="8s").model_dump(mode="json", by_alias=True)], })) monkeypatch.setenv("LC_COMPUTE_CONFIG", str(catalog)) plan = Compute().plan(Request(cpus=1, memory_bytes=512 * 1024**2), name="analysis") @@ -598,7 +573,7 @@ def test_an_ended_allocation_keeps_its_record_but_not_its_secrets_or_scratch( provider: LocalProvider, ending: str, ) -> None: identity = _launch(provider, seconds=2 if ending == "walltime" else 60) - directory = provider.root / identity.token + directory = provider.allocations / identity.token scratch = Path(str(read_private_json(directory / "launch.json")["scratch"])) try: if ending == "down": @@ -711,9 +686,7 @@ def test_termination_escalates_captured_children_when_owner_exits_first( assert owner.stdout is not None child = psutil.Process(int(owner.stdout.readline())) process = psutil.Process(owner.pid) - identity = Identity( - namespace=provider.connection.namespace, native_id=str(owner.pid), token=uuid4().hex, - ) + identity = Identity(provider="local", native_id=str(owner.pid), token=uuid4().hex) monkeypatch.setattr(provider, "_record", lambda _identity: (tmp_path, {})) monkeypatch.setattr( provider, "_process", lambda *_args: process if owner.poll() is None else None, @@ -752,7 +725,7 @@ def wait(*, timeout: float) -> None: ) monkeypatch.setattr("lightcone.engine.compute.local.psutil.Process", lambda _pid: process) monkeypatch.setattr("lightcone.engine.compute.local._boot_identity", lambda: "test-boot") - identity = Identity(namespace=provider.connection.namespace, native_id="123", token=uuid4().hex) + identity = Identity(provider="local", native_id="123", token=uuid4().hex) record = {"created": 123, "boot": "test-boot"} if exited: assert provider._process(identity, provider.root, record) is None @@ -774,14 +747,14 @@ def fail(argv: list[str], **kwargs: Any) -> subprocess.Popen[bytes]: patch.setattr("lightcone.engine.compute.local.subprocess.Popen", fail) with pytest.raises(ComputeError, match="cannot execute"): _launch(provider) - assert list(provider.root.iterdir()) == [] - assert list(Path(provider.connection.launch["scratch_root"]).iterdir()) == [] + assert list(provider.allocations.iterdir()) == [] + assert list(_scratch(provider).iterdir()) == [] identity = _launch(provider) try: _ready(provider, identity) # A launcher interrupted before identity publication can also leave a # directory. This is not a published allocation or a discovery error. - interrupted = private_directory(provider.root / uuid4().hex, create=True) + interrupted = private_directory(provider.allocations / uuid4().hex, create=True) write_private_json(interrupted / "launch.json", {"identity": ""}) assert [snapshot.identity for snapshot in provider.discover()] == [identity] finally: @@ -799,8 +772,8 @@ def fail(path: Path, value: dict[str, Any]) -> None: monkeypatch.setattr(local, "write_private_json", fail) with pytest.raises(ComputeError, match="cannot start.*storage is full"): _launch(provider) - assert list(provider.root.iterdir()) == [] - assert list(Path(provider.connection.launch["scratch_root"]).iterdir()) == [] + assert list(provider.allocations.iterdir()) == [] + assert list(_scratch(provider).iterdir()) == [] def test_interrupted_launch_kills_the_unreturned_owner( @@ -845,7 +818,7 @@ def test_owner_exits_when_launcher_is_killed_before_startup_commit( import json, sys, time from pathlib import Path from lightcone.engine.compute import local -from lightcone.engine.compute.model import Connection, Offer, Request, Resources, TimeLimits +from lightcone.engine.compute.model import Offer, Request owners = local._running_owners local._running_owners = lambda: [o for o in owners() if o[1].is_relative_to(sys.argv[2])] write = local.write_private_json @@ -861,14 +834,14 @@ def pause(path, value): time.sleep(60) write(path, value) local.write_private_json = pause -p = local.LocalProvider(Connection.model_validate_json(sys.argv[1])) -offer = Offer(name='small', connection='workstation', resources=Resources(cpus=1, memory_gib=0.5), - max_nodes=1, time=TimeLimits(default='1m', max='1m')) +p = local.LocalProvider(Path(sys.argv[1])) +offer = Offer.model_validate_json(sys.argv[5]) p.launch(p.plan(offer, Request(cpus=1, memory_bytes=512 * 1024**2))) """ launcher = subprocess.Popen( - [sys.executable, "-c", script, provider.connection.model_dump_json(), - str(local_allocation_scope), str(marker), publication], + [sys.executable, "-c", script, str(provider.root), + str(local_allocation_scope), str(marker), publication, + _offer(provider).model_dump_json(by_alias=True)], stdout=subprocess.PIPE, stderr=subprocess.PIPE, ) owner: psutil.Process | None = None @@ -903,7 +876,7 @@ def test_reused_pid_and_boot_identity_are_never_signalled( provider: LocalProvider, monkeypatch: pytest.MonkeyPatch ) -> None: identity = _launch(provider) - path = provider.root / identity.token / "identity.json" + path = provider.allocations / identity.token / "identity.json" original = read_private_json(path) try: with monkeypatch.context() as patch: @@ -928,7 +901,7 @@ def test_reused_pid_and_boot_identity_are_never_signalled( psutil.Process, "cmdline", lambda _process: [ sys.executable, "-P", "-m", "lightcone.engine.compute.local_runtime", - str(provider.root / uuid4().hex), + str(provider.allocations / uuid4().hex), ], ) assert provider.inspect(identity).phase == "ended" @@ -1042,7 +1015,7 @@ def test_connection_files_are_private_and_scheduler_identity_is_authenticated( identity = _launch(provider) try: _ready(provider, identity) - directory = provider.root / identity.token + directory = provider.allocations / identity.token assert stat.S_IMODE(directory.stat().st_mode) == 0o700 for name in ("identity.json", "connection.json", "scheduler.json", "tls-key.pem"): assert stat.S_IMODE((directory / name).stat().st_mode) == 0o600 @@ -1063,7 +1036,7 @@ def test_missing_credentials_preserve_native_discovery_and_termination( identity = _launch(provider) try: _ready(provider, identity) - (provider.root / identity.token / "tls-key.pem").unlink() + (provider.allocations / identity.token / "tls-key.pem").unlink() assert provider.inspect(identity).phase == "active" assert provider.discover()[0].identity == identity with pytest.raises(ComputeError): @@ -1075,7 +1048,7 @@ def test_missing_credentials_preserve_native_discovery_and_termination( def test_a_starting_allocation_says_to_wait(provider: LocalProvider) -> None: identity = _launch(provider) - connection = provider.root / identity.token / "connection.json" + connection = provider.allocations / identity.token / "connection.json" try: _ready(provider, identity) # The owner publishes this file once its scheduler is up. @@ -1112,10 +1085,7 @@ def test_private_material_rejects_symlinks_broad_modes_and_hardlinks(tmp_path: P def test_local_plan_is_one_node_bounded_cooperative_and_does_not_allocate( provider: LocalProvider, ) -> None: - offer = Offer( - name="small", connection="workstation", resources=Resources(cpus=1, memory_gib=0.5), - max_nodes=1, time=TimeLimits(default="1m", max="1m"), - ) + offer = _offer(provider) plan = provider.plan(offer, Request(cpus=1, memory_bytes=512 * 1024**2)) assert not provider.root.exists() assert "cooperative" in plan.details["resource_enforcement"] @@ -1131,44 +1101,37 @@ def test_local_allocation_reopens_through_symlinked_configured_roots( physical = private_directory(tmp_path / "physical-home", create=True) alias = tmp_path / "home-alias" alias.symlink_to(physical, target_is_directory=True) - connection = provider.connection.replace( - launch={ - "connection_root": str(alias / ".lightcone" / "compute"), - "scratch_root": str(alias / "scratch"), - }, - ) - launcher = LocalProvider(connection) - identity = _launch(launcher) + # The catalog resolves the configured alias before any provider sees it. + root = Path(Catalog(connection_root=str(alias / ".lightcone" / "compute")).connection_root) + launcher = LocalProvider(root) + identity = _launch(launcher, scratch=alias / "scratch") try: scheduler = _ready(launcher, identity) - reopened = LocalProvider(connection) - assert reopened.root == physical / ".lightcone" / "compute" / connection.namespace + reopened = LocalProvider(root) + assert reopened.allocations == physical / ".lightcone" / "compute" / "local" assert [snapshot.identity for snapshot in reopened.discover()] == [identity] with reopened.connect(identity) as client: assert client.scheduler_info()["id"] == scheduler["id"] assert client.submit(sum, [2, 3]).result(timeout=5) == 5 - metadata = read_private_json(reopened.root / identity.token / "launch.json") + metadata = read_private_json(reopened.allocations / identity.token / "launch.json") assert metadata["scratch"] == str(physical / "scratch" / f"lc-{identity.token}") assert private_directory(Path(metadata["scratch"])).is_dir() finally: - LocalProvider(connection).terminate(identity) + LocalProvider(root).terminate(identity) _ended(launcher, identity) -def test_local_managed_namespace_symlink_is_still_refused( +def test_local_managed_provider_directory_symlink_is_still_refused( provider: LocalProvider, tmp_path: Path, ) -> None: root = private_directory(tmp_path / "configured-root", create=True) alias = tmp_path / "root-alias" alias.symlink_to(root, target_is_directory=True) - other = private_directory(tmp_path / "other-namespace", create=True) - (root / provider.connection.namespace).symlink_to(other, target_is_directory=True) - connection = provider.connection.replace( - launch={**provider.connection.launch, "connection_root": str(alias)}, - ) - configured = LocalProvider(connection) + other = private_directory(tmp_path / "other-directory", create=True) + (root / "local").symlink_to(other, target_is_directory=True) + configured = LocalProvider(Path(Catalog(connection_root=str(alias)).connection_root)) with pytest.raises(ComputeError, match="plain directory"): - _launch(configured) + _launch(configured, scratch=_scratch(provider)) with pytest.raises(ComputeError, match="plain directory"): configured.discover() assert list(other.iterdir()) == [] @@ -1182,23 +1145,18 @@ def test_default_and_configured_scratch_aliases_are_resolved_without_resolving_p alias = tmp_path / "os-temp" alias.symlink_to(physical, target_is_directory=True) monkeypatch.setattr("lightcone.engine.compute.local.tempfile.gettempdir", lambda: str(alias)) - connection = provider.connection.replace( - launch={"connection_root": str(tmp_path / "connections")}, - ) offer = Offer( - name="small", connection="workstation", resources=Resources(cpus=1, memory_gib=0.5), + name="small", provider="local", resources=Resources(cpus=1, memory_gib=0.5), max_nodes=1, time=TimeLimits(default="1m", max="1m"), ) request = Request(cpus=1, memory_bytes=512 * 1024**2) - plan = LocalProvider(connection).plan(offer, request) + plan = provider.plan(offer, request) assert plan.details["scratch_root"] == str(physical) private_directory(Path(plan.details["scratch_root"]) / "default", create=True) python = tmp_path / "python" python.symlink_to(sys.executable) - configured = connection.replace(launch={ - **connection.launch, "scratch_root": str(alias), "python": str(python), - }) - plan = LocalProvider(configured).plan(offer, request) + configured = offer.replace(config={"scratch_root": str(alias), "python": str(python)}) + plan = provider.plan(configured, request) assert plan.details["scratch_root"] == str(physical) assert plan.details["python"] == str(python) private_directory(Path(plan.details["scratch_root"]) / "configured", create=True) @@ -1222,10 +1180,7 @@ def test_local_provider_allows_interactive_nodes_but_refuses_login_nodes_before_ monkeypatch.setenv("NERSC_HOST", "perlmutter") monkeypatch.setenv("SLURM_JOB_ID", "123") monkeypatch.setattr(socket, "gethostname", lambda: "nid200021") - offer = Offer( - name="small", connection="workstation", resources=Resources(cpus=1, memory_gib=0.5), - max_nodes=1, time=TimeLimits(default="1m", max="1m"), - ) + offer = _offer(provider) plan = provider.plan(offer, Request(cpus=1, memory_bytes=512 * 1024**2)) assert plan.resources == offer.resources monkeypatch.setattr(socket, "gethostname", lambda: "login01") @@ -1243,7 +1198,7 @@ def test_local_gpu_plan_freezes_native_mask_and_publishes_the_configured_envelop order: str | None, ) -> None: monkeypatch.setattr(sys, "platform", "linux") - # Mask interpretation and agreement with the catalog belong to the operator. + # CUDA reads a device UUID and an index alike; neither is checked against hardware. mask = "GPU-opaque,3" monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) if order is None: @@ -1252,12 +1207,11 @@ def test_local_gpu_plan_freezes_native_mask_and_publishes_the_configured_envelop monkeypatch.setenv("CUDA_DEVICE_ORDER", order) probe = MagicMock(side_effect=AssertionError("local GPU planning must not probe hardware")) monkeypatch.setattr(local.subprocess, "run", probe) - offer = Offer( - name="gpu", connection="workstation", + offer = _offer(provider).replace( + name="gpu", resources=Resources.from_bytes( cpus=1, memory_bytes=512 * 1024**2, gpus=gpus, accelerator_name="A100", ), - max_nodes=1, time=TimeLimits(default="1m", max="1m"), ) plan = provider.plan(offer, Request.parse("1", "0.5", gpus=f"GPU:{gpus}" if gpus else "0")) assert plan.details["cuda_visible_devices"] == (mask if gpus else "") @@ -1288,7 +1242,7 @@ def spawned(argv: list[str], **kwargs: Any) -> SimpleNamespace: assert popen.call_args.kwargs["env"].get("CUDA_DEVICE_ORDER") == order assert os.environ["CUDA_VISIBLE_DEVICES"] == "an-ambient-mask" assert os.environ["CUDA_DEVICE_ORDER"] == "changed-after-planning" - record = read_private_json(provider.root / identity.token / "identity.json") + record = read_private_json(provider.allocations / identity.token / "identity.json") assert record["gpus"] == gpus monkeypatch.setattr(provider, "_process", lambda *_: None) snapshot = provider.inspect(identity) @@ -1297,8 +1251,8 @@ def spawned(argv: list[str], **kwargs: Any) -> SimpleNamespace: probe.assert_not_called() -@pytest.mark.parametrize("mask", [None, ""]) -def test_local_gpu_offers_require_an_explicit_nonempty_native_mask( +@pytest.mark.parametrize("mask", [None, "", "-1", "0", "0,-1,2", "0,,1"]) +def test_local_gpu_offers_need_as_many_devices_as_the_mask_exposes( provider: LocalProvider, monkeypatch: pytest.MonkeyPatch, mask: str | None, ) -> None: @@ -1308,12 +1262,13 @@ def test_local_gpu_offers_require_an_explicit_nonempty_native_mask( else: monkeypatch.setenv("CUDA_VISIBLE_DEVICES", mask) offer = Offer( - name="gpu", connection="workstation", - resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=1), + name="gpu", provider="local", + resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=2), max_nodes=1, time=TimeLimits(default="1m", max="1m"), ) - request = Request.parse("1", "0.5", gpus="GPU:1") - with pytest.raises(ComputeError, match="nonempty CUDA_VISIBLE_DEVICES"): + request = Request.parse("1", "0.5", gpus="GPU:2") + # CUDA stops at the first invalid entry, so none of these exposes two devices. + with pytest.raises(UnavailableOfferError, match="has 2 GPUs, but CUDA_VISIBLE_DEVICES"): provider.plan(offer, request) assert not provider.root.exists() @@ -1323,20 +1278,16 @@ def test_unconfigured_local_gpu_mask_does_not_hide_a_later_slurm_offer( ) -> None: monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False) local_offer = Offer( - name="local-gpu", connection="workstation", + name="local-gpu", provider="local", resources=Resources.from_bytes(cpus=1, memory_bytes=512 * 1024**2, gpus=1), max_nodes=1, time=TimeLimits(default="1m", max="1m"), ) - service = Compute.__new__(Compute) - service.catalog = Catalog( - version=1, - connections={ - "workstation": provider.connection, - "hpc": Connection(namespace=str(uuid4()), provider="slurm"), - }, - offers=[local_offer, local_offer.replace(name="batch-gpu", connection="hpc")], + catalog = Catalog( + connection_root=str(provider.root), + offers=[local_offer, local_offer.replace(name="batch-gpu", provider="slurm")], ) - plan = service.plan(Request.parse("1", "0.5", gpus="GPU:1")) + monkeypatch.setattr(Catalog, "load", lambda *args: catalog) + plan = Compute().plan(Request.parse("1", "0.5", gpus="GPU:1")) assert plan.offer.name == "batch-gpu" assert not provider.root.exists() @@ -1347,7 +1298,7 @@ def test_local_gpu_offers_are_unavailable_outside_linux( monkeypatch.setattr(sys, "platform", "darwin") monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0") offer = Offer( - name="gpu", connection="workstation", + name="gpu", provider="local", resources=Resources.from_bytes( cpus=1, memory_bytes=512 * 1024**2, gpus=1, ), diff --git a/tests/test_compute_output.py b/tests/test_compute_output.py index 407929b1..d1199406 100644 --- a/tests/test_compute_output.py +++ b/tests/test_compute_output.py @@ -26,23 +26,14 @@ def detached_cluster(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Iterator[str]: catalog = tmp_path / "compute.json" catalog.write_text(json.dumps({ - "version": 1, - "connections": { - "local": { - "provider": "local", - "namespace": str(uuid4()), - "launch": { - "connection_root": str(tmp_path / "connections"), - "scratch_root": str(tmp_path / "scratch"), - }, - }, - }, + "connection_root": str(tmp_path / "connections"), "offers": [{ "name": "small", - "connection": "local", + "provider": "local", "resources": {"cpus": 1, "memory": 1}, "max_nodes": 1, "time": {"default": "2m", "max": "2m"}, + "config": {"scratch_root": str(tmp_path / "scratch")}, }], })) monkeypatch.setenv("LC_COMPUTE_CONFIG", str(catalog)) diff --git a/tests/test_compute_slurm.py b/tests/test_compute_slurm.py index b38e99e0..b9ad10e3 100644 --- a/tests/test_compute_slurm.py +++ b/tests/test_compute_slurm.py @@ -23,7 +23,6 @@ from lightcone.engine.compute.catalog import Catalog from lightcone.engine.compute.model import ( ComputeError, - Connection, Identity, Offer, Request, @@ -31,42 +30,30 @@ TimeLimits, ) from lightcone.engine.compute.runtime import ( + configured_directory, open_client, private_directory, read_private_json, write_private_json, ) -NAMESPACE = "9d0c0fc5-9be8-407a-a3ec-f17c4110b162" TOKEN = "c82a7b8d0ccf40a4be0e57e784edb989" -IDENTITY = Identity(namespace=NAMESPACE, native_id="123", token=TOKEN) -NAME = f"lc-v1-{IDENTITY.name}" -COMMENT = f"lightcone:v1:kind=dask:token={TOKEN}" +IDENTITY = Identity(provider="slurm", native_id="123", token=TOKEN) +NAME = f"lc-{IDENTITY.name}" +COMMENT = f"lightcone:kind=dask:token={TOKEN}" @pytest.fixture def provider(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> slurm.SlurmProvider: _native(monkeypatch, {}) - return slurm.SlurmProvider( - Connection( - namespace=NAMESPACE, - provider="slurm", - context="perlmutter", - launch={ - "connection_root": str(tmp_path / "private"), - "scratch_root": str(tmp_path / "scratch"), - "cwd": str(tmp_path), - "task_slots_per_node": 126, - }, - ) - ) + return slurm.SlurmProvider(tmp_path / "private") @pytest.fixture -def offer() -> Offer: +def offer(tmp_path: Path) -> Offer: return Offer( name="batch", - connection="nersc", + provider="slurm", resources=Resources(cpus=256, memory_gib=480), max_nodes=4, time=TimeLimits(default="1h", max="4h"), @@ -75,6 +62,9 @@ def offer() -> Offer: "account": "myproject", "qos": "regular", "constraint": "cpu", + "scratch_root": str(tmp_path / "scratch"), + "cwd": str(tmp_path), + "task_slots_per_node": 126, }, ) @@ -83,8 +73,8 @@ def _live( *, job_id: str = "123", token: str = TOKEN, uid: int | None = None, state: str = "RUNNING", name: str = IDENTITY.name, comment: str | None = None, ) -> str: - comment = comment if comment is not None else f"lightcone:v1:kind=dask:token={token}" - return f"{job_id}|lc-v1-{name}|{os.getuid() if uid is None else uid}|{state}|{comment}\n" + comment = comment if comment is not None else f"lightcone:kind=dask:token={token}" + return f"{job_id}|lc-{name}|{os.getuid() if uid is None else uid}|{state}|{comment}\n" def _control( @@ -92,9 +82,9 @@ def _control( name: str = IDENTITY.name, comment: str | None = None, ) -> str: owner = os.getuid() if uid is None else uid - comment = comment if comment is not None else f"lightcone:v1:kind=dask:token={token}" + comment = comment if comment is not None else f"lightcone:kind=dask:token={token}" return ( - f"JobId=123 JobName=lc-v1-{name} UserId=alice({owner}) JobState={state}\n" + f"JobId=123 JobName=lc-{name} UserId=alice({owner}) JobState={state}\n" f" Comment={comment} \n" f" NumNodes=2 NumCPUs=512 CPUs/Task=256 MinMemoryNode=480G Restarts={restarts} " "ReqTRES=cpu=512,mem=960G,node=2 " @@ -107,9 +97,9 @@ def _history( job_id: str = "123", submitted: str = "2026-09-27T10:00:00", comment: str | None = None, uid: int | None = None, ) -> str: - comment = comment if comment is not None else f"lightcone:v1:kind=dask:token={token}" + comment = comment if comment is not None else f"lightcone:kind=dask:token={token}" owner = os.getuid() if uid is None else uid - return f"{job_id}|lc-v1-{name}|{owner}|{state}|{submitted}|{comment}\n" + return f"{job_id}|lc-{name}|{owner}|{state}|{submitted}|{comment}\n" def _native( @@ -136,12 +126,11 @@ def run(argv: list[str], **kwargs: Any) -> subprocess.CompletedProcess[str]: def _metadata(provider: slurm.SlurmProvider, *, restarts: int = 0, **changes: Any) -> Path: directory = private_directory( - slurm.attempt_directory(provider.connection, IDENTITY, restarts), create=True + slurm.attempt_directory(provider.root, TOKEN, restarts), create=True ) write_private_json( directory / "identity.json", { - "namespace": NAMESPACE, "native_id": "123", "token": TOKEN, "uid": os.getuid(), @@ -165,7 +154,6 @@ def test_plan_preserves_native_envelope_without_native_queries( assert "--cpus-per-task=256" in plan.details["native_args"] assert "--nodes=2" in plan.details["native_args"] assert "--time=01:00:00" in plan.details["native_args"] - assert "--clusters=perlmutter" in plan.details["native_args"] assert not any(arg.startswith("--partition=") for arg in plan.details["native_args"]) assert "time_policy" not in plan.details assert calls == [] @@ -229,7 +217,9 @@ def test_default_launch_assumes_a_shared_home_and_node_local_scratch( ) -> None: monkeypatch.setenv("HOME", str(tmp_path)) calls = _native(monkeypatch, {"sbatch": "123\n"}) - provider = slurm.SlurmProvider(Connection(namespace=NAMESPACE, provider="slurm")) + provider = slurm.SlurmProvider(Path(Catalog().connection_root)) + native = {"submit", "account", "qos", "constraint"} + offer = offer.replace(config={key: offer.config[key] for key in native}) plan = provider.plan(offer, Request.parse("256", "480")) root = str(tmp_path.resolve() / ".lightcone" / "compute") assert plan.details["python"] == sys.executable @@ -240,7 +230,8 @@ def test_default_launch_assumes_a_shared_home_and_node_local_scratch( payload = shlex.split(script.splitlines()[-1]) assert payload[payload.index("--connection-root") + 1] == root assert "--scratch-root" not in payload - assert slurm.attempt_directory(provider.connection, identity, 0).is_relative_to(root) + # The submission's directory exists before sbatch runs, beside no other provider's. + assert slurm.allocation_directory(provider.root, identity.token).is_dir() def test_plan_resolves_configured_roots_but_preserves_virtualenv_python( @@ -253,15 +244,11 @@ def test_plan_resolves_configured_roots_but_preserves_virtualenv_python( python = alias / "venv" / "bin" / "python" python.parent.mkdir(parents=True) python.symlink_to(sys.executable) - connection = provider.connection.replace( - launch={ - **provider.connection.launch, - "python": str(python), - "connection_root": str(alias / "private"), - "scratch_root": str(alias / "scratch"), - }, + offer = offer.replace( + config={**offer.config, "python": str(python), "scratch_root": str(alias / "scratch")}, ) - plan = slurm.SlurmProvider(connection).plan(offer, Request.parse("256", "480")) + root = Path(Catalog(connection_root=str(alias / "private")).connection_root) + plan = slurm.SlurmProvider(root).plan(offer, Request.parse("256", "480")) assert plan.details["connection_root"] == str(actual / "private") assert plan.details["scratch_root"] == str(actual / "scratch") assert plan.details["python"] == str(python) @@ -329,14 +316,12 @@ def test_plan_refuses_multiple_partitions(provider: slurm.SlurmProvider, offer: ) -def test_plan_rejects_unknown_launch_settings( +def test_plan_rejects_unknown_offer_settings( provider: slurm.SlurmProvider, offer: Offer, ) -> None: - connection = provider.connection.replace( - launch={**provider.connection.launch, "cpu_bind": "cores"} - ) - with pytest.raises(ComputeError, match="unknown Slurm launch settings: cpu_bind"): - slurm.SlurmProvider(connection).plan(offer, Request.parse("256", "480")) + offer = offer.replace(config={**offer.config, "cpu_bind": "cores"}) + with pytest.raises(ComputeError, match="unknown Slurm offer settings: cpu_bind"): + provider.plan(offer, Request.parse("256", "480")) def test_sbatch_launch_owns_payload_and_scrubs_ambient_overrides( provider: slurm.SlurmProvider, @@ -355,7 +340,7 @@ def test_sbatch_launch_owns_payload_and_scrubs_ambient_overrides( monkeypatch.setenv(name, "bad-override") monkeypatch.setenv("SLURM_CONF", "/etc/slurm/site.conf") monkeypatch.setenv("SLURM_JWT", "private-auth") - calls = _native(monkeypatch, {"sbatch": "123;perlmutter\n"}) + calls = _native(monkeypatch, {"sbatch": "123\n"}) plan = provider.plan(offer, Request.parse("256", "480", num_nodes=2)) assert provider.launch(plan) == IDENTITY argv, kwargs = next(call for call in calls if call[0][0] == "sbatch") @@ -490,9 +475,9 @@ def test_named_launch_keeps_the_full_submission_token_in_native_metadata( assert launched == IDENTITY.replace(name=name) argv = (next(argv for argv, _ in calls if argv[0] == "sbatch") if submit == "sbatch" else popen.call_args.args[0]) - assert f"--job-name=lc-v1-{name}" in argv + assert f"--job-name=lc-{name}" in argv assert f"--comment={COMMENT}" in argv - assert len(f"lc-v1-{name}") <= 69 + assert len(f"lc-{name}") <= 66 @pytest.mark.parametrize("name", ["", "Upper", "two words", "-leading", "trailing-", "a" * 64]) @@ -524,7 +509,7 @@ def test_named_submission_recovers_from_history_when_sbatch_output_is_malformed( assert provider.inspect(identity).phase == "ended" accounting = next(argv for argv, _ in calls if argv[0] == "sacct") - assert f"--name=lc-v1-{name}" in accounting + assert f"--name=lc-{name}" in accounting assert f"--uid={target_uid}" in accounting assert "--format=JobIDRaw,JobName%128,UID,State,Submit,Comment%128" in accounting assert sum(argv[0] == "sbatch" for argv, _ in calls) == 1 @@ -538,7 +523,7 @@ def test_recovery_does_not_accept_a_different_name_with_the_same_nonce( "sbatch": "garbled", "squeue": _live(name="other"), "sacct": _history(name="other"), }) plan = provider.plan(offer, Request.parse("256", "480")).replace(name="requested") - with pytest.raises(ComputeError, match=f"lc-v1-requested and comment token {TOKEN}") as raised: + with pytest.raises(ComputeError, match=f"lc-requested and comment token {TOKEN}") as raised: provider.launch(plan) assert raised.value.submission_token == TOKEN @@ -554,7 +539,7 @@ def test_named_jobs_are_discovered_and_cancelled_from_native_names( assert snapshot.identity == IDENTITY.replace(name=name) assert snapshot.identity.name == name provider.terminate(snapshot.identity) - assert f"--name=lc-v1-{name}" in calls[-1][0] + assert f"--name=lc-{name}" in calls[-1][0] assert "--format=%i|%128j|%U|%T|%128k" in next( argv for argv, _ in calls if argv[0] == "squeue" ) @@ -618,7 +603,7 @@ def test_discovery_and_cancellation_resolve_the_execution_user_once_per_provider assert identity_calls[0][0] == ["id", "-u"] assert not identity_calls[0][1].get("shell", False) - slurm.SlurmProvider(provider.connection).discover() + slurm.SlurmProvider(provider.root).discover() assert sum(argv[0] == "id" for argv, _ in calls) == 2 @@ -642,8 +627,10 @@ def test_existing_native_name_cannot_be_hidden_from_duplicate_name_checks( "scontrol": _control(name="analysis", comment=comment), }) compute = Compute.__new__(Compute) - compute.catalog = Catalog(version=1, connections={"nersc": provider.connection}, offers=[offer]) - monkeypatch.setattr(compute, "provider", lambda connection: provider) + compute.catalog = Catalog(offers=[offer]) + local = MagicMock() + local.discover.return_value = [] + monkeypatch.setattr(compute, "provider", {"slurm": provider, "local": local}.__getitem__) plan = provider.plan(offer, Request.parse("256", "480")).replace(name="analysis") expected = "already in use" if comment is None else "discovery is incomplete" @@ -732,7 +719,7 @@ def test_live_discovery_uses_native_marker_and_keeps_grant_evidence_honest( monkeypatch: pytest.MonkeyPatch, ) -> None: unrelated = f"999|another|job|{os.getuid()}|RUNNING\n" - calls = _native(monkeypatch, {"squeue": _live() + unrelated, "scontrol": _control()}) + _native(monkeypatch, {"squeue": _live() + unrelated, "scontrol": _control()}) snapshots = provider.discover() assert len(snapshots) == 1 snapshot = snapshots[0] @@ -741,7 +728,6 @@ def test_live_discovery_uses_native_marker_and_keeps_grant_evidence_honest( assert snapshot.resources == Resources.from_bytes(cpus=256, memory_bytes=480 * 1024**3) assert snapshot.evidence == "requested" assert snapshot.ready is None - assert all("--clusters=perlmutter" in argv for argv, _ in calls if argv[0] != "id") @pytest.mark.parametrize("cpus,memory", [ @@ -840,18 +826,15 @@ def test_changed_or_missing_control_comment_cannot_be_cancelled( assert all(argv[0] != "scancel" for argv, _ in calls) -@pytest.mark.parametrize("context", ["perlmutter", ""]) def test_cancel_targets_allocation_with_native_name_and_owner_filters( - provider: slurm.SlurmProvider, monkeypatch: pytest.MonkeyPatch, context: str, + provider: slurm.SlurmProvider, monkeypatch: pytest.MonkeyPatch, ) -> None: - provider = slurm.SlurmProvider(provider.connection.replace(context=context)) calls = _native(monkeypatch, {"squeue": _live(), "scontrol": _control(), "scancel": ""}) provider.terminate(IDENTITY) argv = calls[-1][0] assert argv == [ "scancel", "--ctld", - *(["--clusters=perlmutter"] if context else []), f"--user={os.getuid()}", f"--name={NAME}", "123", @@ -924,12 +907,11 @@ def test_connect_resolves_only_the_configured_root( actual.mkdir() alias = tmp_path / "home" alias.symlink_to(actual, target_is_directory=True) - connection = provider.connection.replace( - launch={**provider.connection.launch, "connection_root": str(alias / "private")}, + provider = slurm.SlurmProvider( + Path(Catalog(connection_root=str(alias / "private")).connection_root) ) - provider = slurm.SlurmProvider(connection) directory = _metadata(provider) - assert directory == actual / "private" / NAMESPACE / f"123-{TOKEN}" / "attempt-0" + assert directory == actual / "private" / "slurm" / TOKEN / "attempt-0" if managed_symlink: moved = directory.with_name("moved") directory.rename(moved) @@ -966,12 +948,26 @@ def test_connect_refuses_wrong_or_untyped_identity_metadata( def test_a_running_job_without_its_scheduler_yet_says_to_wait( provider: slurm.SlurmProvider, monkeypatch: pytest.MonkeyPatch, ) -> None: + private_directory(slurm.allocation_directory(provider.root, TOKEN), create=True) _native(monkeypatch, {"scontrol": _control()}) with pytest.raises(ComputeError, match="has not started yet") as raised: with provider.connect(IDENTITY): pytest.fail("a job without a scheduler must not connect") assert raised.value.cluster_id == IDENTITY.encode() + +def test_a_job_launched_under_another_connection_root_says_so_rather_than_to_wait( + provider: slurm.SlurmProvider, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, +) -> None: + _metadata(provider) + _native(monkeypatch, {"scontrol": _control()}) + elsewhere = slurm.SlurmProvider(tmp_path / "another-root") + with pytest.raises(ComputeError, match="launched with another connection_root") as raised: + with elsewhere.connect(IDENTITY): + pytest.fail("another root's job must not connect") + assert "has not started" not in str(raised.value) + assert raised.value.cluster_id == IDENTITY.encode() + def test_status_can_observe_a_reachable_but_degraded_scheduler( provider: slurm.SlurmProvider, monkeypatch: pytest.MonkeyPatch ) -> None: @@ -1007,7 +1003,6 @@ def test_connect_rechecks_native_attempt_after_tls( def _bootstrap_args(tmp_path: Path) -> argparse.Namespace: return argparse.Namespace( submission=TOKEN, - namespace=NAMESPACE, connection_root=str(tmp_path / "private"), scratch_root=str(tmp_path / "scratch"), num_nodes=2, @@ -1099,12 +1094,11 @@ def test_gpu_worker_advertises_native_capacity_with_the_native_mask( monkeypatch.setenv("SLURM_GPUS_ON_NODE", "2") monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "1,3") monkeypatch.setenv("CUDA_DEVICE_ORDER", "FASTEST_FIRST") - connection = Connection( - namespace=NAMESPACE, provider="slurm", launch={"connection_root": args.connection_root}, + directory = private_directory( + slurm.attempt_directory(Path(args.connection_root), TOKEN, 0), create=True, ) - directory = private_directory(slurm.attempt_directory(connection, IDENTITY, 0), create=True) write_private_json(directory / "identity.json", { - "namespace": NAMESPACE, "native_id": "123", "token": TOKEN, "uid": os.getuid(), + "native_id": "123", "token": TOKEN, "uid": os.getuid(), "restarts": 0, "num_nodes": 2, "cpus": 1, "memory_bytes": args.memory_bytes, "gpus": 2, "task_slots": 1, }) @@ -1190,10 +1184,9 @@ def test_standard_bootstrap_starts_scheduler_and_worker_on_rank_zero_and_worker_ # CPU-only submissions can omit the optional GPU count. if value is not None and key != "gpus": argv += ["--" + key.replace("_", "-"), str(value)] - connection = Connection( - namespace=NAMESPACE, provider="slurm", launch={"connection_root": args.connection_root} + directory = slurm.attempt_directory( + configured_directory(Path(args.connection_root)), TOKEN, 0, ) - directory = slurm.attempt_directory(connection, IDENTITY, 0) processes = [] client = None environment = dict(os.environ)