diff --git a/.github/workflows/check-docs.yml b/.github/workflows/check-docs.yml deleted file mode 100644 index fe9d23eb..00000000 --- a/.github/workflows/check-docs.yml +++ /dev/null @@ -1,69 +0,0 @@ -name: Documentation Health Check - -on: - pull_request: - types: [closed] - -jobs: - check-docs: - # Only run when the PR was actually merged (not just closed) - if: github.event.pull_request.merged == true - runs-on: ubuntu-latest - permissions: - contents: read - issues: write # To open an issue if docs are stale - pull-requests: write # To post a comment on the merged PR - id-token: write # Required for OIDC token (claude-code-action) - - steps: - - name: Checkout repo - uses: actions/checkout@v4 - with: - fetch-depth: 0 # Full history needed to diff against base SHA - - - name: Check documentation with Claude - uses: anthropics/claude-code-action@v1 - with: - claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} - prompt: | - A pull request has just been merged into this repository. - - PR #${{ github.event.pull_request.number }}: ${{ github.event.pull_request.title }} - Description: ${{ github.event.pull_request.body }} - - Your task: check whether the project documentation is still accurate - and complete given the code changes introduced by this PR. - - The documentation is structured, and each kind of change has a home: - - A verb's flags, output, JSON shape, or exit codes → its page in docs/cli/ - (one page per verb, plus the overview's exit-code contract). - - An engine module's responsibility, key symbols, or invariants → its page - in docs/api/ (hand-written module tours) and, for cross-cutting shifts, - docs/architecture.md. - - User-visible behavior (scaffold contents, states, refusal messages, - environment model, SLURM, publication) → docs/user/ (getting-started - quotes real console output; troubleshooting quotes real refusals) and - README.md's quick start. - - Test structure, dev workflow, or conventions → docs/contributing/. - - Steps to follow: - 1. Run: git diff --name-only ${{ github.event.pull_request.base.sha }} ${{ github.sha }} - to get the list of changed files. - 2. Read the changed source files (focus on src/**/*.py and the workflows). - 3. Read the documentation pages the map above points at for those changes. - 4. SKIP CLAUDE.md and evals/ — agent instructions and the eval harness are - maintained separately, not user-facing docs. - 5. Identify documentation that is now inaccurate, incomplete, or missing. - Two failure modes matter most here: a quoted console output or refusal - message that no longer matches what the CLI prints, and a documented - flag, verb, state, or file that no longer exists (the docs must never - describe more than the code delivers — no foreshadowing). - - Then: - - Post a comment on PR #${{ github.event.pull_request.number }} summarising - your findings (✅ if docs look fine, ⚠️ if issues found). - - If you find real documentation issues, open a GitHub Issue titled - "📚 Docs may be stale after PR #${{ github.event.pull_request.number }}" - with label "documentation", listing the specific files and what needs updating. - - Do NOT open an issue if documentation looks fine. - - Do NOT modify any files — read-only analysis only. \ No newline at end of file diff --git a/.github/workflows/docs-deploy.yml b/.github/workflows/docs-deploy.yml deleted file mode 100644 index e531ffb2..00000000 --- a/.github/workflows/docs-deploy.yml +++ /dev/null @@ -1,86 +0,0 @@ -name: Deploy Docs - -# Deploys a versioned snapshot of the docs to the gh-pages branch via -# mike; GitHub Pages serves that branch as docs.lightconeresearch.org -# (Settings → Pages must be set to "Deploy from a branch" / gh-pages). -# Each version lives under its own subdirectory (/0.4.1/, /0.5.0/, …) -# with the `latest` alias tracking the newest release, so old versions -# stay accessible after a new one ships. -# -# The site tracks the *released* CLI, not main: deploys happen on the -# same trigger as the PyPI publish, so the docs never document behavior -# that `pip install lightcone-cli` can't deliver yet. Pre-releases are -# skipped for the same reason — PyPI hands an rc only to someone who -# asks for it by name, so the site must keep serving the last full -# release. For an intermediate deploy (typo fix, clarification), -# trigger manually from the Actions tab — workflow_dispatch runs -# against the selected ref and redeploys the snapshot for the given -# version (defaults to the latest tag reachable from that ref). - -on: - release: - types: [published] - workflow_dispatch: - inputs: - version: - description: "Docs version to (re)deploy, e.g. 0.4.1 (empty: latest tag on the selected ref)" - required: false - default: "" - -# mike commits and pushes to the gh-pages branch. -permissions: - contents: write - -# Deploys append commits to gh-pages — queue them, never cancel one -# mid-push. -concurrency: - group: docs-deploy - cancel-in-progress: false - -jobs: - deploy: - # A pre-release deploys nothing. `github.event.release` is null on a - # workflow_dispatch run, so the manual path stays unconditional. - if: ${{ !github.event.release.prerelease }} - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - with: - # mike needs the gh-pages branch, and version resolution - # needs tags — fetch everything. - fetch-depth: 0 - - - name: Set up uv - uses: astral-sh/setup-uv@v6 - with: - enable-cache: true - activate-environment: true - - - name: Install docs dependencies - run: uv sync --group docs - - - name: Resolve docs version - id: version - env: - INPUT_VERSION: ${{ inputs.version }} - run: | - if [ -n "$INPUT_VERSION" ]; then - v="$INPUT_VERSION" - elif [ "${{ github.event_name }}" = "release" ]; then - v="${GITHUB_REF_NAME#v}" - else - v="$(git describe --tags --abbrev=0)" - v="${v#v}" - fi - echo "version=$v" >> "$GITHUB_OUTPUT" - echo "Deploying docs version $v" - - - name: Configure git identity - run: | - git config user.name "github-actions[bot]" - git config user.email "41898282+github-actions[bot]@users.noreply.github.com" - - - name: Deploy versioned docs to gh-pages - run: | - uv run mike deploy --push --update-aliases "${{ steps.version.outputs.version }}" latest - uv run mike set-default --push latest diff --git a/CLAUDE.md b/CLAUDE.md index f68206c6..70b35fe2 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -36,8 +36,7 @@ spec: > (rationale, substrate tradeoffs, hermeticity enforcement, the v6 > review). It stays in the sibling checkout and is **dropped when the > rebuild completes** (decision, 2026-08): the design records are not -> imported into this repo's docs — the rewritten `docs/` carries the -> current design, and this file carries the decisions. +> imported into this repo — this file carries the decisions. The pre-rebuild codebase (Snakemake shim, authored Containerfiles, `container:` in `astra.yaml`, vendored dask executor plugin, WRROC export) @@ -106,15 +105,6 @@ Each of these has been asked for in review at least once; none is optional. - **No dead code.** If nothing in the current layer calls it, it doesn't land yet. `lc --help` advertises only verbs that work. -- **`docs/` is live again** (rewritten 2026-08, PRs #185–#188; the - freeze is over). The site is two tracks — user guide + developer - corner — and a change now lands with its docs: a new or changed verb - updates its `docs/cli/` page, an engine change updates its - `docs/api/` module page, and user-visible behavior updates the user - guide. The docs' own rules match this file's: document only what - exists, quote refusals from real runs, and verify every command - block by executing it. `check-docs.yml` reviews each merged PR for - drift. - **Port with intent.** Prior implementations (this repo's git history, and the `redesign_prototype` branch of the sibling `lightcone-cli` checkout) are references, not sources of truth. Neither is the spec by @@ -206,14 +196,6 @@ evals/ # agentic eval seed: prompt.md + tasks// tests/ # pytest — mirrors src/ ``` -## Documentation versioning (mike) - -The whole docs site is versioned with [mike](https://github.com/squidfunk/mike) — specifically squidfunk's fork, which Zensical's versioning provider depends on. Each release deploys a full copy of the site to a subdirectory of the `gh-pages` branch (`/0.0.9/`, `/latest/`, etc.). Mike is enabled via `[project.extra.version] provider = "mike"` in `zensical.toml`; the version dropdown in the header is rendered natively. - -Release flow: `.github/workflows/docs-deploy.yml` runs on every published release — it runs `mike deploy --push --update-aliases X.Y.Z latest` (version taken from the tag) followed by `mike set-default --push latest`, so the bare site root always redirects to `/latest/`. A **pre-release deploys nothing**: `release: published` fires for pre-releases too (`released` is the type that skips them), and moving `latest` onto an rc would serve as default what `pip install` deliberately withholds, so the job carries `if: ${{ !github.event.release.prerelease }}`. PyPI needs no equivalent — "pre-release" there is derived from the PEP 440 version alone (an rc tag ⇒ hatch-vcs ⇒ `0.5.0rc1`), never from the GitHub checkbox, so `pypi-publish.yaml` stays unconditional. For an intermediate redeploy of an existing version, trigger the workflow manually from the Actions tab. For local/manual operations, run the same mike commands directly (`uv run mike list`, `uv run mike deploy ...`, `uv run mike delete ...` — the docs dependency group installs mike). - -Hosting: mike pushes to `gh-pages`. GitHub Pages (which serves docs.lightconeresearch.org) must be configured to "Deploy from a branch" / `gh-pages` in the repo's Pages settings, not via the Actions artifact deploy. Without this, `mike deploy` runs successfully but the site doesn't pick up versioned URLs in production. - ## Development Commands ```bash @@ -227,12 +209,9 @@ uv build # wheel + sdist (CI runs this only to publish) Test, lint and type-check are the whole loop, and they are what `.github/workflows/{tests,lint}.yml` run. There is deliberately no task runner in between — the pre-rebuild `justfile` was 90 lines of wrappers -around them. The docs build with `uv sync --group docs && uv run -zensical build`. The other workflows are `eval.yml` (the agentic eval, +around them. The other workflows are `eval.yml` (the agentic eval, on dispatch or the `run-eval` PR label; re-trigger by re-adding the -label), `check-docs.yml` (doc-drift review on merged PRs), -`pypi-publish.yaml`, and `docs-deploy.yml` (deploys on release, so the -site tracks the released CLI). +label) and `pypi-publish.yaml`. ## Key Invariants (layer 1) @@ -612,7 +591,7 @@ repository's — `uv tool install lightcone-cli` links the git-annex wheel's entry points beside `lc` (verified), so `ambient` git-annex is on `PATH` for free and the whole question disappears. `uvx` is fine for running lc and cannot support the researcher's bare `git add`; that is -what the docs and the troubleshooting entry say, and reporting it from +what the docs say, and reporting it from `lc init`/`lc status` is an open follow-up, not a promise made here. `require_git_annex` stays a `PATH` check, deliberately — it gates lc's *own* `git annex` subprocesses, which dispatch from lc's environment, @@ -2554,7 +2533,7 @@ written to" — a path the schema never defined. What changed, and why: | To... | Read | Key patterns | |---|---|---| -| Add the next layer | the spec (§11 = the layer ordering) | Land code + tests + deps together; update the layer table above and the docs pages the layer touches | +| Add the next layer | the spec (§11 = the layer ordering) | Land code + tests + deps together; update the layer table above | | Change what a scaffolded file contains | `src/lightcone/engine/templates/files/` | Edit the `.tmpl`; add new ones to `TEMPLATE_NAMES`, and a renderer only if the file needs a substituted value or a merge policy | | Add a value to the scaffold | `src/lightcone/engine/templates/__init__.py` | Derive it from the environment or our own metadata before introducing a constant | | Change what gets converged | `src/lightcone/engine/project.py` + `tests/test_project.py` | `_Converger.item` / `.file`; repairs only ever append | diff --git a/README.md b/README.md index 04bccffc..cc45a1b3 100644 --- a/README.md +++ b/README.md @@ -21,7 +21,7 @@ uv tool install lightcone-cli ``` Create an ASTRA project and launch the built-in local compute offer; no compute -configuration is needed. [Configure resource offers](docs/user/cluster.md) for +configuration is needed. [Configure resource offers](https://docs.lightconeresearch.org) for larger local allocations or Slurm: ```bash @@ -41,8 +41,6 @@ lc compute down "$CLUSTER" ASTRA specs are plain, structured YAML — they work well hand-written or drafted with any AI coding assistant. -→ [Full getting-started guide](https://docs.lightconeresearch.org/user/getting-started/) - ## Capabilities - **Multiverse analysis** — declare methodological decisions with multiple defensible options; `lc` materializes your analysis across every universe you define diff --git a/docs/api/assets.md b/docs/api/assets.md deleted file mode 100644 index 2b563085..00000000 --- a/docs/api/assets.md +++ /dev/null @@ -1,53 +0,0 @@ -# lightcone.engine.assets - -One output: its directory, its manifest, and whether it is still -current. The classification rule lives here, next to the manifest it -reads and the hashes it compares — and it is the one place in the -engine where a bug is quiet rather than loud, which is why it may not -have two implementations. - -Source: `src/lightcone/engine/assets.py`. - -## Key symbols - -| Symbol | Role | -|---|---| -| `classify(...)` | The one rule: `current` / `behind` / `stale`, with the why. Two callers — the worker and the read-only walk. | -| `Verdict.calls_for_a_remake(refresh=)` | The one place a state becomes an action: `stale` always, `behind` only when asked. | -| `data_version(path)` | Content hash of a directory or file; workers hash new outputs before they are annexed. | -| `Versions` | Per-run memo populated on the driver and serialized with tasks, so shared declared inputs are hashed once. | -| `read(sidecar)` / `write(...)` | The manifest, `..manifest.json`. Both take the sidecar's own path, so a caller holding an output path has to say `manifest_path` out loud. | -| `output_path(root, u, id, fmt)` | The output's file, guarded: any part that is not a single path component is refused, and so is a format that could not be an extension. | -| `manifest_path(output)` | The sidecar beside it, named from the id alone — so it keeps its path, and its history, across a re-declared format. | -| `ContentNotFetchedError` | An annexed file whose content is not in this clone, in either shape it takes. | - -## What must stay true - -- **One `classify`, two callers, one differing value.** The worker - hands live input digests; check mode hands `None` for anything - upstream that will run ("this is going to change"). That value is - the entire difference — never a second body of logic. History (the - foreign-write fact) enters the same way: computed by whoever has - git, handed in as a value. -- **The comparison is fourfold**: `definition_version`, the declared - input *set* (separate on purpose — a dropped dependency moves - neither hash), each recorded input digest, then `env_version`. - `stale` wins over `behind`; `behind` does not propagate and a behind - upstream still feeds its dependents. -- **A skip returns the *recorded* digest, never a recomputed one** — - on a bytes-free clone, rehashing dangling symlinks would quietly - report a different output. -- **Unfetched content refuses loudly, in both shapes.** A pointer file - hashes to a well-formed digest of the wrong thing; a dangling - symlink drops out of an `is_file()` walk without a word. Both raise - `ContentNotFetchedError` naming `git annex get`; only dangling - symlinks are added back to the directory walk. -- **`calls_for_a_remake` has three callers** (worker, check, the - cascade walk) and no inline re-spellings — the third copy is where - they start to disagree. - -## Tests - -`tests/test_assets.py` — pure; nothing on disk beyond `tmp_path`. -The pointer-file and dangling-symlink traps are pinned against real -annex shapes in `tests/test_dataset.py`. diff --git a/docs/api/compute.md b/docs/api/compute.md deleted file mode 100644 index 7ccc4749..00000000 --- a/docs/api/compute.md +++ /dev/null @@ -1,271 +0,0 @@ -# lightcone.engine.compute - -The allocation boundary shared by CLI lifecycle operations and execution. -`Compute` loads resource policy and obtains fresh native observations. -It owns no service, registry, or saved current-cluster selection. - -| Symbol | Contract | -|---|---| -| `Request.parse(...)` | Exact/minimum CPU and memory requests, exact accelerator type/count, node count, walltime, startup class. | -| `Catalog.load(path)` | Ordered fixed shapes, each naming its provider; apply local defaults or disable policy alongside configured offers. | -| `Compute.plan(request, *, name=None)` | Select an eligible offer and freeze its native launch settings and optional name without allocation. | -| `Compute.launch(plan)` | Check names across native authorities, generate one if omitted, submit once, and return a self-contained `Identity`. | -| `Compute.discover()` | Snapshots and per-provider errors, querying each provider in `Catalog.providers` once. | -| `Compute.status(cluster_id, wait=False, timeout=300)` | Resolve a name or full ID; return native allocation state plus authenticated Dask readiness. Waiting backs off from one to 30 seconds between native queries. | -| `Compute.down(cluster_id)` | Resolve a name or full ID, request native termination independent of scheduler health, and return the canonical `Identity`. | -| `connect(cluster_id, timeout=10)` | Resolve a name or full ID; borrow a standard Dask client, closing the client but never the allocation. Submits one no-op task, so a caller's preparation restarts the idle countdown. | -| `Provider` | `plan`, `launch`, `discover`, `inspect`, `connect`, `terminate`. | - -`Catalog.load()` defaults to `~/.lightcone/compute.yaml`. The built-in `local` -offer uses detected usable CPUs and RAM, one node, fast startup, and no walltime: -it ends after 30 minutes without task activity. `local.resources` overrides its -CPU/RAM budget and `local.time` its time limits; -`local.enabled: false` blocks local launch and execution while local discovery -continues for inspection and termination. Remote catalogs retain the implicit local -offer unless disabled. Explicit local offers replace it and cannot be combined -with `local.resources` or `local.time`. GPU offers require explicit configuration. -Loading creates no configuration file or allocation. -The effective local policy also disables local offers on recognized NERSC login -nodes: nonempty `NERSC_HOST` and a short hostname matching `login[0-9]+`. -Explicit enablement and Slurm job environment variables do not override this -guard; interactive compute nodes remain eligible. Local planning, launch, and -execution check the same policy, while status and termination remain available. -Missing paths selected through an argument or `LC_COMPUTE_CONFIG`, unreadable -files, and invalid catalogs remain errors. `Catalog.providers` names the native -authorities to query: those the offers use, and always `local`. Each provider -keeps its allocations in one directory under the catalog's `connection_root`, so -separate invocations discover and attach to the same allocations. - -`Compute.plan_local()` selects only local offers and defaults the name to `local`. -CLI `launch --wait` waits through `Compute.status` using the accepted immutable ID; -errors retain that ID without resubmission or termination. - -`model.py` defines the shared Pydantic models: `Offer`, `Resources`, `Accelerator`, -`TimeLimits`, `Startup`, `Request`, `Identity`, `LaunchPlan`, and `Snapshot`. -`Catalog` validates YAML directly into these objects, which providers also use. -Unknown common fields are rejected; schema errors identify paths such as -`offers.0.resources.cpus` without echoing input values. The YAML loader rejects -duplicate and non-string mapping keys before model validation. Each offer's -provider-specific `config` mapping remains the provider's responsibility. - -Units are explicit. `Resources.memory_gib` stores exact decimal GiB (the YAML key -is `memory`), and `memory_bytes` derives an exact integer. Native observations use -`Resources.from_bytes(...)`; requests store `Request.memory_bytes`. `TimeLimits` -keeps the configured `default`, `max`, and `idle` duration strings, each optional -but requiring a `default` or an `idle`, and exposes `default_seconds`, -`max_seconds`, and `idle_seconds` (`None` when unset). A `LaunchPlan` carries the -resolved hard walltime as `seconds` and derives `idle_seconds` from its offer; -either may be `None` for a local plan, while Slurm plans always have `seconds` and -never `idle_seconds`. `Startup.class_` corresponds to YAML `class`. -`Offer.provider` names a factory in `PROVIDERS`, which receives the catalog's -`connection_root`. - -Compute memory accepts bare GiB quantities and SkyPilot-style binary units: -`32`, `32GB`, and `32GiB` agree. CPU and memory requests accept a trailing `+`. -`Accelerator` accepts one `NAME[:COUNT]` or one-entry mapping, such as `A100:4` -or `{A100: 4}`, and serializes to that mapping. Counts are exact positive integers; -type matching is case-insensitive, and the generic name `GPU` accepts any model. -No accelerator registry or model alias expansion is maintained. Slurm's -`config.gpu_type` maps a named catalog accelerator to its native GRES type. - -Model constructors take keyword arguments. `replace(...)` validates updates; -`model_dump()` and `model_validate()` support internal roundtrips without changing -units. Keep the explicit `as_dict()` methods for public CLI output so internal -configuration does not leak. Value models are frozen; `Snapshot` permits validated -updates as native and scheduler observations arrive. Nested settings dictionaries -and catalog collections are not deeply immutable. - -`local.py` and `slurm.py` implement the provider protocol. Adding an adapter means -adding one provider factory and its native mapping; `run` and `materialize` only -borrow clients through the common API. Provider settings stay behind that seam. -`runtime.py` owns private files, standard TLS material, and authenticated scheduler -identity checks. `local_runtime.py` and `slurm_bootstrap.py` compose stock Dask -components; they do not define custom workers or membership protocols. - -The Slurm bootstrap runs one stock `Nanny` per rank, each with a separate worker -process; rank zero also hosts the scheduler. The Nanny restarts an exited worker -and supplies Dask's default `OMP_NUM_THREADS`, `MKL_NUM_THREADS`, and -`OPENBLAS_NUM_THREADS` values of `1`, preserving explicit launch environment -values. Sandbox policy forwards those effective values into recipe containers. -`memory_limit=0` remains deliberate: Dask's worker memory accounting excludes -the external recipe subprocesses. The payload uses -`srun --kill-on-bad-exit=0 --wait=0` to avoid terminating healthy ranks merely -because another rank exited. Site OOM policy can still terminate the step or -allocation, and there is no recovery for a dead scheduler or Nanny. Dask can -reschedule lost tasks, but surviving recipe subprocesses are not fenced from -those retries. Connection readiness still requires every expected worker. -See [Dask's Nanny](https://distributed.dask.org/en/stable/worker.html#nanny), -[Dask resilience](https://distributed.dask.org/en/stable/resilience.html), and -[Slurm's `srun` options](https://slurm.schedmd.com/srun.html). - -The configured `connection_root` and scratch roots are resolved before managed paths are -appended, so filesystem aliases such as a symlinked home directory are supported. -Managed directories and credential files retain strict symlink, ownership, and -permission checks, including modes `0700` and `0600`, respectively. - -Slurm planning includes a partition only when explicitly configured and does -not query or freeze the site's time policy. Every launch requests a finite native -`--time`; native overtime and termination grace govern actual expiry, with no -independent Lightcone deadline or guarantee of a finite overrun. - -`Snapshot` distinguishes native allocation evidence from scheduler observations. -No live allocation size is filled from today's catalog. IDs encode their provider -and native incarnation evidence, so a full ID routes without consulting the offers -and without a UUID-to-job lookup database. Exceptions retain known cluster IDs and -submission tokens for partial/ambiguous acceptance. - -`Identity.name` is a human-facing name; `Identity.encode()` is the immutable -allocation reference. Generated names use `lc-` plus 12 hexadecimal characters -from a standard-library UUID4. Launch checks every provider in `Catalog.providers` and -rejects explicit duplicates or incomplete discovery before submission. Name -resolution also requires complete discovery and exactly one current match. -Concurrent launches can still race; ambiguous names are refused. Names can be -reused after termination, while full IDs continue to identify the original -allocation without discovering other providers. Slurm carries the name -in `JobName=lc-v1-` and the submission token in -`Comment=lightcone:v1:kind=dask:token=<32hex>`; local private locators carry the -encoded identity. Slurm discovery and lifecycle checks verify both native fields -and the owner. Discovery makes one `squeue` query, then one single-job -`scontrol` lookup per managed job. A marked live job with no valid token makes -discovery incomplete. A verified live job is cancelled whatever state Slurm -reports; a job absent from live jobs must prove from accounting that it ended. -Neither is a second source of lifecycle state or a name-to-ID registry. - -The Slurm provider resolves its user ID once through `id -u` using the same -command runner as Slurm. Discovery, accounting, cancellation, and allocation -ownership checks all use that ID. The commands execute on the host running `lc`, -and filesystem ownership checks validate that process's access to connection -material. - -`slurm.py` maps native states onto the common phases: `PENDING`, `CONFIGURING`, -`SUSPENDED`, `RESIZING` and the requeue states are `pending`, `RUNNING` is -`active`, `COMPLETING` is `stopping`, and every terminal state is `ended`. Any -other state is `unknown` for `status`, while `terminate` still cancels such a -job once its owner, name and token verify. `connect` pins the job's current -restart count and reads only that attempt's connection material, then checks the -count again after the TLS handshake, so a requeued job cannot hand over an -earlier attempt's scheduler. A job that is running but has not yet published its -scheduler, like a local allocation still starting, is refused with -`runtime.NOT_STARTED`. - -Historical Slurm identity requires the accounting `Comment` field. Slurm stores -it when `AccountingStoreFlags` includes `job_comment`; without a matching retained -token, a missing live job remains unknown and cannot authorize cancellation. -See [Slurm's accounting field documentation](https://slurm.schedmd.com/sacct.html). - -Execution submits ordinary tasks through the borrowed client's `submit` method. -Dask chooses the workers and handles dependencies; invocation-specific keys prevent -unintended reuse across commands. There is no worker-selection layer, per-worker -preflight orchestration, or source fingerprinting. The local login-node guard -does not restrict remote Slurm execution from a login node. Driver-side -preparation and the existing task runtime/sandbox checks remain in their owners. - -Workers advertise standard Dask `CPU`, `MEMORY`, and `GPU` resources; memory is measured -in bytes. `engine.execution_resources.TaskResources` validates ASTRA's -`recipe.resources` into whole CPUs, bytes, and a whole GPU count at -execution admission. `plan.Task` preserves the ASTRA mapping so read-only -classification does not impose executor restrictions. `worker_capacities(workers)` -normalizes advertised budgets once; `requirements(capacities)` checks that one -worker can satisfy a task and returns its `Client.submit` resource dictionary. -Omitted memory adds no `MEMORY` reservation. `whole_worker=True` reserves CPU, -memory, and GPUs for a probe. Recipe GPU counts default to zero; a GPU recipe -reserves the full GPU budget of a fitting worker -and inherits its whole allocation mask. The requested count is a minimum, not a -visibility limit. This serializes GPU recipes per worker without device assignment. -Unsupported disk/type requests and fractional CPU/GPU counts fail before execution. - -The driver reuses the read-only classification walk before admission. Known -current or unrefreshed behind outputs become `TaskResult` values, without Dask -submission or resource reservations. Tasks that may execute, including dependents -of potentially rebuilt outputs, have their resource requests validated before -preparation. Workers recheck actual upstream digests and may still skip a reserved -task if its inputs prove unchanged. Allocation and task requests share byte -conversion utilities; their models remain distinct because allocation selection supports minimum quantities -and node counts. Standard Dask scheduling accounts for -concurrent CPU, memory, and GPU reservations; Dask execution-thread counts remain a -separate concurrency cap. Reservations do not impose hard limits on recipe -subprocesses. Recipe `time_limit` is unsupported and explicitly refused before -preparation or execution; allocation walltime remains supported. - -Recipe memory remains ASTRA-style: `8Gi` is binary, `8GB` is decimal, and units -are required. Allocation memory follows the compute convention above; keep the -two parsers' contracts explicit even though they share exact byte arithmetic in -`units.py`. Allocation duration parsing stays in `compute.model.duration`, raising -`ValueError` for Pydantic; `Request.parse` converts it to `ComputeError`. - -## GPU allocation and visibility - -Local GPU offers require Linux and an explicit nonempty `CUDA_VISIBLE_DEVICES`. -Planning freezes that mask and optional `CUDA_DEVICE_ORDER`; launch passes them -to the worker unchanged. Count and model are catalog declarations, not hardware -observations. The built-in local offer remains CPU-only. - -Slurm requests native GPU GRES and validates `SLURM_GPUS_ON_NODE` before -advertising the worker's GPU budget. Bootstrap preserves Slurm's CUDA mask and -sets `CUDA_DEVICE_ORDER=PCI_BUS_ID`. There is no CUDA probe, device inventory, or -custom Dask worker. - -The sandbox's `use_gpus` policy option inherits the worker's mask for GPU commands -and supplies an empty mask for CPU commands, without modifying the reusable -worker's environment. Direct GPU policies grant native NVIDIA character devices. -Container GPU execution uses podman-hpc's `--gpu`. Explicit GPU recipes on ordinary -Docker or Podman are refused before image preparation; probes use a CPU policy and -report that GPU access is unavailable while retaining their whole-worker reservation. -Native permissions and cgroups remain authoritative. NVIDIA devices, including UVM, -must already exist; policy construction does not load drivers or create devices. -See [GPU deployment requirements](../user/cluster.md#gpu-allocations). - -## Execution output and teardown - -The driver still saves each output in its own Git/annex commit while Dask runs -the submitted graph. This serial storage work can dominate many short recipes; -adding workers does not accelerate it. - -`output.py` transports byte chunks through standard Dask events so detached -workers' output reaches the invoking CLI. It uses the borrowed client's event -topic, which the schedulers lc launches drop as soon as the client disconnects -(`runtime.SCHEDULER_CONFIG`), rather than retaining a separate topic for every -command. A driver that exits before every task reports says so with -`UNSTOPPED`: closing a client cannot prove that a remote subprocess stopped. Probes preserve both streams; -materialization sends recipe output to stderr to leave stdout for its report. - -Local allocations are limited to one per user on each machine, independent of -connection roots. Before spawning, the launcher scans the process -table for a live owner of the same user: a session leader running `-P -m -lightcone.engine.compute.local_runtime `, which excludes workers forked -from it. The process table spans every catalog and connection root and needs no -file lock, which some shared home filesystems, NERSC's included, do not support. -The owner's directory argument locates its identity record, so a refused launch -names the running cluster, its connection root, and the catalog it was launched -with. A record not yet written means the owner is still starting; one that is -missing or unreadable after that never recovers, so the refusal names the -owner's PID instead. Two limits are accepted rather than closed with a lock: -launches that overlap can both pass the scan, and the scan covers one PID -namespace, so a container sharing the home directory does not see the host's -owner. - -A startup pipe lets the owner proceed only after the launcher publishes its -identity and launch records. If the launcher dies before completing publication, -the pipe closes and the owner exits. Failures before identity -publication remove the launcher's private files; a published identity remains -inspectable after a startup failure. - -Local teardown drains the allocation's validated process group rather than -assuming the owner's exit proves every child stopped. Boot UUID, UID, process -session and the exact command containing a random allocation token establish -identity without depending on hostname or wall-clock creation time. Discovery -skips other boot sessions; explicit operations refuse them because this process -cannot establish their state on another host. Failed unpublished launches -are cleaned up, and incomplete locator directories do not hide healthy allocations. -An allocation verified as ended, by `down` or by discovery, is retired: its TLS -material, scheduler files and scratch are removed, and a marker lets discovery -skip it unread. Its identity record stays, so a full ID still reports `ended`. -Cancellation and concurrent project writers are not made safe by allocation -management; callers must respect the documented execution limits. Containers -managed outside that process group can survive local teardown. - -Tests cover deterministic selection, malformed identities and catalogs, partial -native failures, acceptance ambiguity, PID reuse, detached local walltime and idle -expiry, standard -Dask bootstrap, and explicit execution through borrowed clients. Slurm command -contracts are simulated; a real NERSC submission remains a deployment check. diff --git a/docs/api/container.md b/docs/api/container.md deleted file mode 100644 index 0bb0f553..00000000 --- a/docs/api/container.md +++ /dev/null @@ -1,76 +0,0 @@ -# lightcone.engine.image & container - -The container hatch, split down the pure/impure line. `image.py` is -what a containerized project *declares* and how that becomes an -identity — pure, no subprocess anywhere. `container.py` is building, -storing and entering images — impure, every command through -`project._run`. The exec side (the mount table) lives with the other -backends in `sandbox/oci.py`. - -Sources: `src/lightcone/engine/image.py`, -`src/lightcone/engine/container.py`, `src/lightcone/engine/sandbox/oci.py`. - -## Key symbols - -| Symbol | Role | -|---|---| -| `image.declaration(root)` | The `[tool.lightcone.image]` table, validated — a closed key set (`base`, `apt-install`, `run-commands`, `env`), because every key is hashed. | -| `image.tag(root)` | `lc-env-<16 hex>` over the rendered Containerfile *and* the identity document. | -| `image.archive_path(root, tag)` | `.datalad/environments//image` — the `datalad containers-add` layout. | -| `container.build(root)` | Build + save + commit, idempotent; returns `(Runtime, "built" \| "present")`. | -| `container.runtime_for_run(root, *, build, use_gpus=False)` | Resolve the runtime, refusing unsupported explicit GPU requests before preparing the image. Materialize may build and commit; probes and reruns only find, fetch, and load. | -| `container.backend(...)` | The single construction point for the exec backend — the only mode branch. | -| `container.sync(...)` | The in-container environment converge: network on, project `:rw`, host uv cache mounted, into `.lightcone/venv`. | -| `Runtime` | Resolved execution facts; `supports_gpus` is true for direct mode and podman-hpc. | - -## What must stay true - -- **The user never sees a Containerfile.** The render exists only in a - transient build context; the image's `LABEL` carries the identity - document so the archive stays self-describing. There is deliberately - no `pip-install` key — the Python environment is the lock's - business, never the image's. -- **The engine never enters the image.** The container is the - *recipe's* world: driver, git, annex, and classification stay the - host's `lc`; exactly two things run in-image — the sync and each - exec. Network is uncontrolled on every mechanism, symmetrically, and - the attestation says so — no consumer may read a promise into - "containerized". -- **No project file enters the build context** — that is what makes - "code edits never rebuild" structural rather than incidental. -- **The dataset is the store; runtime stores are caches.** Execution - pins the archive's config-blob **id** (readable with no runtime), - never a tag; a dropped archive never substitutes — a rebuild is a - new archive under a new id. -- **Builds and archive commits happen only on a clean tree, and only - after the graph resolves** — a refusal over a typo must not cost a - minutes-long build, and `dataset.save` commits the whole index. -- **The mount table is the mechanism** (`sandbox/oci.py`): project - `:ro`, the write scope `:rw` — the directory a recipe's output lands - in, or `results/` for a probe — declared inputs `:ro`, private HOME, - `--tmpfs /tmp`, over a `--read-only` rootfs — without that flag a - stray write *succeeds* into the ephemeral layer and vanishes while - the attestation claims `fs: declared`. Mounts are resolved source, - **declared** destination — the one policy shape that keeps its paths - unresolved, because they are addresses the recipe uses. -- **Runtime differences are spellings, never shapes.** One - `OCIBackend`, data-parameterized; the podman family is stated once - (`_PODMAN_FAMILY`) and asked positively, so a new runtime falls - outside it by default. podman-hpc adds exactly one step (`migrate`, - outside the load branch). Execution verifies the prepared image on each worker. - Detection order podman-hpc → podman → docker; docker's daemon is - probed at detection. -- **The architecture gate refuses before the load** — a wrong-arch - `load` succeeds and then dies as `exec format error` deep inside a - recipe. Ignorance passes; a recorded mismatch refuses, naming the - fix. - -## Tests - -`tests/test_image.py` (pure: structure and ordering, tag sensitivity -both ways, the `env_version` frame), `tests/test_container.py` -(lifecycle against the stubbed `_run` — every refusal on recorded -argv), `tests/test_sandbox_oci.py` (the mount table, pure), and -`tests/test_container_smoke.py` — the runtime's answer, gated by -`LC_CONTAINER_TESTS_REQUIRED=1` in CI, building a real image and -proving the record on a bytes-free clone with a real `datalad rerun`. diff --git a/docs/api/crate.md b/docs/api/crate.md deleted file mode 100644 index c403edb5..00000000 --- a/docs/api/crate.md +++ /dev/null @@ -1,68 +0,0 @@ -# lightcone.engine.crate - -The publication view: the repository described as a Workflow Run -RO-Crate. The project *is* the crate — `ro-crate-metadata.json` sits at -the root, describes what the repository already holds, and a deposit is -`git archive`, not an export step. lc's manifests stay the canonical -record; the crate is the same facts in schema.org vocabulary for -archives and viewers that will never run `lc`. - -Source: `src/lightcone/engine/crate.py` (converged by -`materialize._converge_crate`). - -## Key symbols - -| Symbol | Role | -|---|---| -| `render(root, graph, *, license, dsid, writer)` | The document, as bytes. A pure function of repository state — git comes in as the `writer` callable, the dataset id as a value. | -| `license_of(root)` | `[project].license` from `pyproject.toml`; empty means no crate is maintained. Presence is publication intent. | -| `CRATE_FILENAME` | `ro-crate-metadata.json`. | - -## What must stay true - -- **The clock never enters the render.** `datePublished` is the newest - manifest `finished_at` (the spec file's last-commit date for a - never-materialized project) and must override rocrate's - construction-time default. Entities build in sorted order, - serialization is `sort_keys` — render-twice-identical is the one - byte-level claim, and it is what makes convergence sound. The - serialization also compacts every one-element array to its value, - as RO-Crate 1.1 recommends: which properties hold one value depends - on the project, so the rule lives in one place, not in each builder. -- **Maintenance is derived, never configured.** RO-Crate requires a - license; materialize must not refuse to run science over a missing - key, and inventing one asserts terms over someone's data. Absent ⇒ - one report line; removed later ⇒ the file is left, and the line says - it is no longer maintained. -- **Run identity comes free from `git_sha`** — the driver reads HEAD - once per run, so grouping manifests by it *is* grouping by run: one - `OrganizeAction` per materialize, a `ControlAction` per execution, a - `HowToStep` per output id (deduped across universes — a step is spec - structure, an action is one execution). -- **The `Person` is the author of the output's *saving* commit** (via - `writer`), never the manifest's `git_sha` — that is the commit the - run *started* at and can be someone else's. -- **An output is a `File`, not a `Dataset` of parts.** It is one file, - so there is one annex key to look up and one `sha256` to publish — - the same number its manifest records as `data_version`, and the one - `sha256sum` prints. -- **The manifest is not transliterated.** `env_version`, - `definition_version` and `hermeticity` get no invented schema.org - spelling — the manifest itself is in the crate as a `File`, - `subjectOf` its output. Real vocabulary comes from the workflow-run - `@context`, without which `containerImage` and `sha256` are - undefined terms JSON-LD silently drops — the pre-rebuild exporter's - failure mode. -- **The rerun entry point does not regenerate the crate** — it is one - task's executor, so the crate lags until the next materialize. - Recorded residue, not a bug. - -## Tests - -`tests/test_crate.py` — pure: fixture manifests, a hand-built graph, a -stub writer, no git anywhere; structure and ordering assertions plus -the single render-twice byte check. `tests/test_crate_smoke.py` — the -official `rocrate-validator` against Provenance Run Crate 0.5: -REQUIRED clean, RECOMMENDED pinned to the recorded `_FLOOR` set (a new -failure is a regression, a disappearing one is the floor to shrink), -required in CI via `LC_CRATE_TESTS_REQUIRED=1`. diff --git a/docs/api/dataset.md b/docs/api/dataset.md deleted file mode 100644 index 28da872a..00000000 --- a/docs/api/dataset.md +++ /dev/null @@ -1,69 +0,0 @@ -# lightcone.engine.dataset - -The git + git-annex seam: how a project stores what it produced. -Storage follows the DataLad model — git carries the pointers and the -history, git-annex carries the bytes — reached through ordinary `git` -commands. Every command goes through `project._run`, so there is one -monkeypatch point and every invocation is inspectable. - -Source: `src/lightcone/engine/dataset.py` (+ -`templates/files/gitattributes.tmpl` for the routing policy). - -## Key symbols - -| Symbol | Role | -|---|---| -| `save(root, paths, message)` | Stage scoped, commit — with `-c annex.thin=true` and `-c annex.dotfiles=true`, per-add and never written to config. | -| `restore(root, paths)` | `git clean` always; `git checkout HEAD --` only when HEAD has the path. Never `-- .`. | -| `status(root)` | The dirty question, scoped to the project (`-- .`, prefix-stripped) so a project inside a larger repository works. | -| `head(root)` | The commit a run started at — read once per run, by the driver. | -| `last_writer(root, *paths)` | Who last touched an output or its manifest — the foreign-write question. Answers "cannot say" as empty, never an error. | -| `require_committer(root)` | Refuses a repository with no git identity, before any recipe spends time. Asked as `git var`, the question a commit itself asks. | -| `dataset_id(root)` | The DataLad dataset UUID, read via `git config -f`. | -| `set_annex_filter_required(root)` | Set `filter.annex.required=true`, so a `git add` that cannot reach git-annex fails loudly instead of staging raw bytes. | -| `annex_filter_required(root)` | Whether that flag is already set — `lc init --check`'s question. | - -## What must stay true - -- **Nobody is ever asked to run a git-annex command.** `filter=annex` - plus the `.gitattributes` policy make an ordinary `git add` do the - right thing; `annex.largefiles=nothing` comes first and outputs and - data opt out — last match wins, and - `test_analysis_code_stays_in_git_and_stays_writable` pins it against - a real annex. -- **Manifests stay in git**, exempted back out of the annex, so a - bytes-free clone can classify a whole project. -- **An unfetched file exists, in two shapes** — an unlocked pointer - file (readable, hashes to the wrong thing) and a locked dangling - symlink (drops out of naive walks silently). `assets.data_version` - refuses both with `ContentNotFetchedError`; detection handles both - regardless of which shape lc writes, because `annex.thin` and - `git annex lock` are the researcher's to set. -- **Thin is per-add and only where lc writes.** Thin's hazard is an - in-place write rewriting the annex object under its own key; lc - always resets output directories rather than writing in place, but - `data/` is the researcher's, and their tools (`h5py`, astropy - `mode='update'`) do open files for update — so the flag never - reaches repository config. -- **`restore` is asymmetric on purpose:** a first materialization has - no HEAD version to go back to, and a failed task must not discard - edits made elsewhere while the graph ran. -- **Committing an archive or dot-named file needs `annex.dotfiles`** — - git-annex routes dotfiles to git whatever `largefiles` says, and - without the flag an image archive lands as a git blob, silently. -- **`filter.annex.required=true` is the storage policy's safety net.** - Without it, a `git add` whose shell cannot resolve git-annex prints - an error, exits 0, and stages the raw bytes into git history — - measured, and pinned by - `test_stock_plumbing_without_required_stages_raw_bytes_silently`. - It is the *only* thing convergence adds to what `git annex init` - wrote: no filter driver is rewritten, and no hook is touched, so how - git dispatches git-annex stays git-annex's own business and stays - resolved from `PATH`. - -## Tests - -`tests/test_dataset.py`, deliberately against **real tools** -(`real_tools` fixture): whether bytes land in the annex or as a blob -in git is not a question a stub can answer, and every bug this seam -has had was invisible to one. diff --git a/docs/api/identity.md b/docs/api/identity.md deleted file mode 100644 index e7f865ca..00000000 --- a/docs/api/identity.md +++ /dev/null @@ -1,53 +0,0 @@ -# lightcone.engine.identity - -What a materialized output is identified by: two hashes that answer -different questions, and the lock scan that decides whether an -environment can be audited at all. - -Source: `src/lightcone/engine/identity.py`. - -## Key symbols - -| Symbol | Role | -|---|---| -| `definition_version(recipe, decisions)` | What the spec says an output *is* — the rebuild trigger. | -| `env_version(root)` | What it ran under: lock bytes ‖ interpreter pin ‖ install settings ‖ image document. The `behind` trigger. | -| `scan_lock(root)` | Refusals, reports and advisories about what the lock pins. | - -## What must stay true - -- **`env_version` is not part of `definition_version`.** That is the - whole shape of the invalidation model: an environment edit stales - nothing, it makes outputs *behind*. (The original design nested - them; staling every output in every project on an engine upgrade was - the bug, not the cost.) -- **Both hashes are length-framed** — label, length, bytes per field — - so a boundary shift between adjacent fields cannot yield the same - digest from different inputs. Mutation-checked in the suite. -- **The lock is hashed as raw bytes, never parsed.** A comment reflow - moves `env_version`, deliberately: over-invalidation costs a report - line, while a parse of our own can silently disagree with uv. -- **The install-settings list is closed** (`_INSTALL_SETTINGS`), every - key hashed whether or not the project sets it — a setting outside - the list must not move the hash, one merely *matching* today's - default must. Settings are read where uv reads them (`uv.toml` - **replaces** `[tool.uv]`, measured); only values are hashed, never - which file supplied them. User-level uv config is deliberately out - of reach — machine state, not project state — and the residue is - tracked as issue #176. -- **The git commit is recorded, never hashed, and never a signal** — - one sha covers the whole tree, so hashing it stales everything on a - README edit. The honest consequence: editing `src/fit.py` remakes - nothing unless the file is declared as an ASTRA input. Do not add a - heuristic that scans recipes for repo paths. -- **The lock scan refuses only what cannot be audited** — path, - directory, and editable dependencies (two syncs of one lock can - install different code). A registry package with no wheel is a - report; a non-default group is advisory; the project's own package - is exempt. Names compare in PEP 503 form, or a project named - `my_project` fails to recognise itself. - -## Tests - -`tests/test_identity.py` — pure, and written as sensitivity tests in -both directions: what must move each hash, and what must not. diff --git a/docs/api/index.md b/docs/api/index.md deleted file mode 100644 index d79ee66b..00000000 --- a/docs/api/index.md +++ /dev/null @@ -1,39 +0,0 @@ -# Engine Internals - -The `lightcone.engine.*` modules, one page each: what the module owns, -its key symbols, and the invariants a change must keep. These are -hand-written tours, not generated API dumps — the engine is not a -public API (projects don't depend on lightcone-cli), so what matters -is responsibility and contract, not every signature. - -## The map - -| Module | Owns | Character | -|---|---|---| -| [`project`](project.md) | What a project is: convergence, discovery, mode, the `_run` seam | impure | -| [`dataset`](dataset.md) | How a project stores: git + git-annex, run records, restore | impure | -| [`identity`](identity.md) | `env_version`, `definition_version`, the lock scan | pure | -| [`plan`](plan.md) | The spec, read as a graph of tasks (through ASTRA) | pure | -| [`assets`](assets.md) | One output: its directory, manifest, and state | pure | -| [`worker`](worker.md) | Making one output; the rerun entry point | impure | -| [`materialize`](materialize.md) | The driver: gates, scheduling, the save/restore loop, status | impure | -| [`compute`](compute.md) | Resource requests, native allocation lifecycle, borrowed Dask clients | impure | -| [`execution_resources`](compute.md) | Task resource admission | pure | -| [`sandbox`](sandbox.md) | The exec boundary: policy, backends, attestation, denials | mixed | -| [`image` & `container`](container.md) | The container hatch: declaration → image → archive → runtime | pure / impure | -| [`crate`](crate.md) | The publication view: the repo as an RO-Crate | pure | - -"Pure" here is a testing fact: pure modules are tested with nothing on -disk beyond `tmp_path` and nothing spawned; impure ones go through the -one subprocess seam (`project._run`) that the suite stubs — see -[Testing](../contributing/testing.md). - -Two files sit outside the engine on purpose: - -- **`lightcone/_sandbox_exec.py`** — the Landlock shim. Stdlib-only, - zero lightcone imports; it runs on every sandboxed exec, and an - engine import there would put click and the astra stack on that - path. Pinned by tests. -- **`lightcone/cli/commands.py`** — the CLI: flags, rendering, exit - codes. Imports the engine inside callbacks so `lc --help` stays - cheap; never contains logic worth testing beyond rendering. diff --git a/docs/api/materialize.md b/docs/api/materialize.md deleted file mode 100644 index d767404b..00000000 --- a/docs/api/materialize.md +++ /dev/null @@ -1,83 +0,0 @@ -# lightcone.engine.materialize - -Making a whole analysis: what runs, in what order, and what gets -committed. The driver refuses dirt, hands the graph to Dask, and owns -git alone — plus the read-only halves (`check`, `status`) that share -its classification walk. - -Source: `src/lightcone/engine/materialize.py`. - -Recipes are ordinary Dask tasks. Their stdout/stderr is forwarded as bytes to the -driver's stderr, independently of success or failure, leaving stdout for the report. - -## Key symbols - -| Symbol | Role | -|---|---| -| `materialize(root, targets, *, cluster_id, refresh)` | Project checks → graph → cluster connection → fetch/converge → schedule → save/restore → crate converge. | -| `check(root, targets, *, refresh)` | The same classification without executing, committing, or fetching. Exempt from the dirty refusal. | -| `status(root)` | The report: every output's state and provenance commit, plus the mode/image/sandbox header facts. | -| `MaterializeReport` / `StatusReport` | The JSON surfaces; `ok` and `up_to_date` first. | -| `cluster_for_run(cluster_id)` | Borrow the cluster; expose resource validation, submission, and completion. | -| `run_record(...)` / `datalad_run_subject(...)` | The commit message `datalad rerun` replays, and the one spelling of its subject line — shared with the foreign-write comparator, because two strings here would drift. | -| `_engine_requirement()` | How a record pins its engine: by version for a release, by source commit (hatch-vcs) for a dev build. | - -## The run's order, and why - -1. **Read-only project checks before connecting** — tool, committer, dirty-tree, - spec and lock errors do not require a reachable cluster to report. The shared - classification walk identifies outputs already current or left behind. -2. **Explicit cluster before preparing the environment** — validate native - allocation identity, connect, and validate CPU/memory/GPU requests for tasks - that may execute. Known skips become values without resource reservations; - dependents of potentially rebuilt outputs still need admission. Explicit GPU - recipes must also have a supported runtime before any image build. - The dirty refusal has already run: in - containerized mode the converge can commit an image archive, and - `dataset.save` commits the whole index; on a dirty tree the user's - staged edits would be swept in. -3. **Converge before the graph runs** — `uv run --locked --no-sync` - in workers would otherwise execute recipes against a drifted - `.venv` while manifests record the new lock (measured; the state - is made impossible rather than detected). -4. **Graph (validation, lock scan) before the image** — a refusal - over a typo must not cost a minutes-long build. -5. **HEAD, runtime, input hashes and foreign-write facts read once, handed down** - — the driver commits as results arrive, so any per-task read could - answer differently mid-run. Nondeterminism in a provenance field is - worse than either answer. The populated input-hash memo travels with each - task; independent worker processes do not rehash shared inputs. Unreadable - inputs still fail only the tasks that need them. -6. **Save on `ok`, restore reported failures** — unreported outputs are retained - after interruption because their tasks may still be writing. Allocation - management does not provide concurrent-writer or cancellation guarantees. - -## What must stay true - -- **The driver owns git, alone** — one thread, as results arrive. - A dependent may start while its upstream is being annexed; that is - measured-safe (the clean filter renames over the path, which never - stops existing) and must not be "fixed" by moving the save into the - task. -- **`up_to_date` is `ok and not made and not planned`** — a run where - every recipe failed must not report "nothing to do", and `behind` - never counts against it. -- **A read-only verb never tracebacks.** Anything `check`/`status` - cannot read classifies as "will be remade" and the real error - belongs to the recipe that follows. -- **The run record is genuinely re-runnable**: engine pinned by - requirement, project environment rebuilt by the worker from the - rerun commit's own lock, format tested *through datalad's parser* - and a real `datalad rerun` — a golden test over our own JSON stays - green through a silent break. -- **The crate converge is contained**: it runs after the loop, on the - full graph, and a failure there is a warning — the outputs are - already committed, and the crate is the publication view, not the - run. - -## Tests - -`tests/test_materialize.py` — real repositories, real recipes, a real -`LocalCluster` through the seam exactly once, real `datalad rerun` for -the record's whole claim. `cluster_for_run` is the one monkeypatch -point for allocation-free tests. diff --git a/docs/api/plan.md b/docs/api/plan.md deleted file mode 100644 index 49f06a78..00000000 --- a/docs/api/plan.md +++ /dev/null @@ -1,65 +0,0 @@ -# lightcone.engine.plan - -The spec, read as a graph of tasks. `astra.yaml` × `universes/*.yaml` -gives one task per `(universe, output)` pair that has a recipe; a task -carries everything executing it needs — the rendered command, where its -bytes go, what it reads, its decisions, its `definition_version`, and its -resource requirements — and nothing about *how* it will be executed. - -Source: `src/lightcone/engine/plan.py`. - -## Key symbols - -| Symbol | Role | -|---|---| -| `build(root)` | Validate the spec with ASTRA's own validators, resolve every universe, return the `Graph`. | -| `Graph` | Tasks keyed on `(universe_id, output_id)`; `order()` for the read-only topological walk, `resolve(targets)` for what a user typed, `closure(keys)` to narrow a run. | -| `Task` | One output in one universe, frozen, retaining ASTRA's resource declaration in `resources`. | -| `declared_path(root, path)` | The one rule that names a path: project-relative inside the tree, absolute outside, never resolved. | - -## What must stay true - -- **What the spec *means* is ASTRA's to say.** `astra.resolve` settles - decisions, resolves inputs, drops `when:`-excluded outputs, and - renders the placeholder grammar. This module holds only what - *execution* adds. A prior in-house interpretation diverged three - ways (couldn't build ASTRA's own nested example, ignored `when:`, - invented an input spelling `astra validate` rejects) — that history - is why re-derivation is banned. Missing semantics → PR to - astra-tools. -- **A spec ASTRA rejects never reaches a recipe.** `build` runs the - schema, file, and universe validators before resolving anything — - resolution answers what a *valid* spec means and does not re-check - that it is one. -- **Resource declarations survive resolution.** `build` reads - `recipe.resources` from ASTRA's resolved output definition and preserves the - mapping. A valid declaration remains readable by `status` and - `materialize --check` even when this executor cannot honor it. Execution - validates supported requirements through `TaskResources.parse` and checks - cluster capacity for tasks that may execute before preparing the project. - Already-current outputs need no resource admission. No worker placement or executor-specific resource validation belongs - in this module. -- **The layout is flat and path-addressed.** - `results//.`, and the path in a - rendered recipe *is* the path on disk — no staging, no relocation. -- **`declared_path` is lexical, never `resolve()`d.** A declared input - under `data/` is an annex symlink; resolving it writes - `.git/annex/objects/…` into the run record — the storage instead of - the input. This shipped once. -- **Two universes cannot share an id** (the id names a directory; - `build` refuses, naming both files), and an out-of-tree absolute - input is **reported, not refused** — its bytes still hash and - cascade, but the repository cannot bring it back, and saying so is - the whole obligation. -- **A target that matches nothing is an error** listing what exists — - quietly making nothing is the least useful thing a build tool can - do. - -## Tests - -`tests/test_plan.py` — pure; tests what lc *adds* (directories, edges, -versions, resource preservation, the validation gate), never what a spec means — -that coverage lives in astra-tools' own suite, and re-asserting it here -would recreate the second implementation this module deleted. Every -fixture must be a spec `astra validate` accepts; the gate enforces it -for free. diff --git a/docs/api/project.md b/docs/api/project.md deleted file mode 100644 index 9b33e8be..00000000 --- a/docs/api/project.md +++ /dev/null @@ -1,60 +0,0 @@ -# lightcone.engine.project - -What a project is: the convergence engine behind `lc init`, project -discovery, mode detection, and the one subprocess seam the whole -engine shares. - -Source: `src/lightcone/engine/project.py` (+ -`engine/templates/` for the scaffold's file content). - -## Key symbols - -| Symbol | Role | -|---|---| -| `converge(dir, *, write)` | The whole scaffold operation. `write=False` is check mode — the *same* decision path with side effects off. | -| `ConvergenceReport` | `created` / `repaired` / `unchanged` / `blocked` / `warnings`, plus `.converged` and `.as_dict()`. | -| `current_project()` | The cwd as a project: requires `pyproject.toml`, `uv.lock`, `.venv`. | -| `declared_project()` | The weaker question — what the repository carries, without `.venv`. One caller: the worker entry point, which builds the venv a moment later. | -| `mode(root)` | `"direct"` or `"containerized"` — presence of `[tool.lightcone.image]`, nothing else. | -| `uv_prefix(root)` | The one spelling of the project uv hop, `--no-sync` because the driver converges the environment before any probe or recipe runs. | -| `project_name(dir)` | PEP 503-ish name from the directory name. | -| `_run` / `_check_call` | Every external tool invocation, and the suite's one monkeypatch point. | -| `ProjectError` | The engine's one exception; the CLI translates it once. | - -## What must stay true - -- **Everything routes through the converger.** Every scaffold item - goes through `_Converger.item` / `.file` / `.blocked`; nothing - writes or records outside that mechanism. `.file` takes a *thunk*, - so check mode renders no template at all. -- **Derived artifacts converge by correctness, not existence.** - `uv.lock` and `.venv` are probed with uv's own no-write checks - (`uv lock --check`, `uv sync --locked --exact --check`); drift - reports as `repaired`. Check mode may probe but never mutates — - pinned by `test_check_mode_only_probes`. -- **A warning is advisory; a blocked item counts.** Convergence never - claims a project is converged while something it owns is absent or - unfixable — and repairs only ever append (`.gitignore` / - `.gitattributes` are converged entry-wise, order judged against the - template). -- **Only what git can carry is converged.** No `src/`, no empty - directories — a clone must need nothing but `.venv`, `git annex - init` and the annex filter (all three local state git does not - clone), and `test_a_clone_of_a_converged_project_is_converged` pins - it. -- **The annex filter is one config key.** `filter.annex.required=true`, - always, so a `git add` that cannot reach git-annex refuses instead of - silently staging raw bytes. Nothing else about how git dispatches - git-annex is lc's to write: the filter drivers and hooks stay exactly - as `git annex init` left them, resolved from `PATH`. -- **There is no discovery.** The invoked directory is the project or - it is a clean error; every uv call carries an explicit `--project`. -- **Templates are files** (`templates/files/*.tmpl`, `string.Template` - with strict substitution), and a template gets a function only when - there is a value to decide or a merge policy to hold. - -## Tests - -`tests/test_project.py` (semantics, against the stubbed `_run`), -`tests/test_templates.py` (content, substitution, repair logic), -`tests/test_cli.py` (the `lc init` surface). diff --git a/docs/api/sandbox.md b/docs/api/sandbox.md deleted file mode 100644 index 8709d6b7..00000000 --- a/docs/api/sandbox.md +++ /dev/null @@ -1,91 +0,0 @@ -# lightcone.engine.sandbox - -The exec boundary: what a command may touch, and how that is enforced. -A `Policy` says *what* in mechanism-free path sets; a `Backend` turns -it into **a different argv that sandboxes itself**; `boundary` picks -one, runs it, and reports what was actually enforced. `run.py` (the -`lc run` engine) and the worker are the two consumers. - -Source: `src/lightcone/engine/sandbox/` — `model.py`, `policy.py`, -`boundary.py`, `landlock.py`, `seatbelt.py`, `oci.py`, `denial.py` — -plus `lightcone/_sandbox_exec.py`, the Landlock shim. - -## Key symbols - -| Symbol | Role | -|---|---| -| `Policy` | What we will enforce: path sets, env overlay, exec allowlist. No mechanism ever appears in it. | -| `Capability` | What this host can do — `detect()`'s answer, the only `sys.platform` branch. | -| `Attestation` | What was actually enforced, derived from the flags applied — never from what the matrix says should have happened. | -| `Backend.wrap(policy, argv)` | The pure rewrite. `contains_prefix` declares whether the uv hop rides inside (a container is a world; a host mechanism trusts host plumbing). | -| `exec_policy(...)` | Shared policy builder with caller-supplied write scope and `use_gpus` setting. Building it creates a private `$HOME`; `scope()` owns its cleanup. | -| `Unavailable` | A real backend that wraps to the same argv and attests `fs: open`. Saying so is the caller's job; pretending is nobody's. | -| `denial.explain()` / `denial.trailer()` | Best-guess remedies (allowed to return nothing) and the unconditional trailer on every nonzero sandboxed exit. | - -An optional output receiver gets stdout/stderr byte chunks. Capturing output never -decodes or normalizes stdout; only the retained stderr tail is decoded for denial -classification. Without a receiver, stdout remains inherited. - -`exec_policy(..., use_gpus=True)` passes the allocation's `CUDA_VISIBLE_DEVICES` -and optional `CUDA_DEVICE_ORDER` through unchanged. CPU commands receive an empty -CUDA mask. Direct GPU policies grant existing NVIDIA character device nodes. -Native permissions still apply; visibility is cooperative, and the reusable -worker's environment is never modified. - -The pure OCI rewrite adds podman-hpc's `--gpu` for GPU commands. Runtime selection -refuses explicit GPU recipes with ordinary Docker or Podman; probes on those -runtimes receive a CPU policy and an explanatory note. CPU containers remain -supported on all runtimes and set `NVIDIA_VISIBLE_DEVICES=void` to override image defaults. -See [GPU allocations](../user/cluster.md#gpu-allocations). - -Container recipes and probes explicitly receive the worker's effective -`OMP_NUM_THREADS`, `MKL_NUM_THREADS`, and `OPENBLAS_NUM_THREADS` values when set. -This preserves Dask Nanny's numerical-library thread settings across the OCI -boundary. Container environment variables remain an explicit allowlist. - -## What must stay true - -- **`wrap` stays pure** — no temp files, no FDs, no global state - (pinned by `test_wrap_is_pure`). That is what makes every backend - testable on a host that cannot run it, and it is why the Landlock - policy travels as JSON on argv rather than an inherited ruleset FD. -- **The shim stays alone**: stdlib only, zero lightcone imports, setup - failures exit the reserved 97, and it never falls through to running - the command unsandboxed. -- **Never grant EXECUTE on a directory that could be a system - prefix.** Landlock unions rights over ancestors, so one EXECUTE on - `/usr` outranks the whole per-file allowlist — with every test still - green, because the allowlisted binaries are exactly the ones that - were going to work. This shipped once (a venv on a system python); - the rule and its test are the fix. -- **SBPL is last-match-wins; Landlock unions.** The asymmetry decides - where a rule can live: the macOS guard takes back writes the - vendored defaults hand out, and the write tier is restated *after* - the guard — get the order wrong and layer 4 materializes on Linux - and refuses on macOS with the golden test still green. -- **Anything every backend must do belongs to the seam** — the env - overlay is composed in `boundary.env_argv()` once, for every - mechanism, so a mechanism added later cannot forget what it never - had to remember. (While each backend applied its own, `Unavailable` - applied none.) -- **The macOS profiles are vendored, not authored** (codex-derived, - provenance header, single delta) — the read baseline is a list of - things that break, found one production failure at a time. Put our - rules in the generator, keep `diff` against upstream as the re-sync - tool. -- **A denial is never invisible**: `explain()` may find nothing, so - the trailer fires on every nonzero exit, unconditionally. Remedies - name only what exists today. - -## Tests - -The suite splits along the seam: -`test_sandbox_policy/wrap/denial.py` (pure, every OS), -`test_sandbox_shim.py` (the shim as a real subprocess), -`test_sandbox_oci.py` (the mount table, pure), and -`test_sandbox_enforcement.py` — **the kernel's answer**, one suite for -both mechanisms, run against the *real* `exec_policy`, with -`LC_SANDBOX_TESTS_REQUIRED=1` turning "no mechanism, skip" into a hard -failure in CI. Every denial test is mutation-checked through -`Unavailable()` — a denial test that would pass unsandboxed is testing -nothing, silently. diff --git a/docs/api/worker.md b/docs/api/worker.md deleted file mode 100644 index 63ecaba8..00000000 --- a/docs/api/worker.md +++ /dev/null @@ -1,83 +0,0 @@ -# lightcone.engine.worker - -Making one output — the unit of work, and the only thing that runs a -recipe. Also an entry point: - -```text -python -m lightcone.engine.worker / -``` - -which is what the `[DATALAD RUNCMD]` record in every materialization -commit names, behind an engine-pinning `uv run --no-project --with …`. -It is a module rather than an `lc` verb on purpose: it makes the -output unconditionally, commits nothing, and leaves the tree dirty by -design — precisely the state `lc materialize` refuses to start from — -so advertising it would hand people a footgun. - -Source: `src/lightcone/engine/worker.py`. - -Cluster execution supplies an output receiver to `materialize`/`execute`, which -passes byte chunks from the sandbox back to the invocation. Standalone reruns -retain direct terminal output. The driver submits each cluster task with its -CPU, memory, and GPU reservations. Before resetting outputs, `execute` validates -resource syntax and builds the command policy with GPU access enabled only when -the recipe requests it. Standalone reruns apply the same checks but do not perform -Dask resource admission. GPU reruns require an explicit `CUDA_VISIBLE_DEVICES` in -the rerun environment, for example `CUDA_VISIBLE_DEVICES=0 datalad rerun`. Recipe -`time_limit` is explicitly refused. - -## Key symbols - -| Symbol | Role | -|---|---| -| `materialize(root, task, context, ...)` | The unit: classify → reset → sandbox → recipe → check the payload → hash → manifest. Returns a `TaskResult`, always. | -| `execute(root, task, input_versions, context)` | Validate resource syntax and GPU policy, run a recipe unconditionally, then record its payload and manifest. | -| `TaskResult` | `ok` / `current` / `behind` / `failed` / `blocked`, the output's `data_version`, reason, and diagnostic notes. `.usable` is what dependents check. | -| `main(argv)` | The rerun entry point: guards, converges the project environment from the commit's own lock, resolves its own HEAD and runtime, executes. | -| `lc_version()` | The engine version every manifest records. | - -## What must stay true - -- **The worker never raises** — enforced at the unit boundary, so the - contract holds for failure modes nobody enumerated. Raising would - make Dask abort every task in flight; reporting all independent - failures in one run is most of what owning the loop buys. -- **Device visibility belongs to each command.** CPU recipes receive an empty - `CUDA_VISIBLE_DEVICES`; GPU recipes inherit the allocation's whole mask through - the sandbox policy. Never modify the reusable worker's shared environment. - Admission reserves the worker's full GPU budget for one GPU recipe at a time, - even when its minimum requested count is smaller. -- **`data_version` is computed here, before anything is staged** — the - dependent's argument *is* this return value, so the digest must - exist while the files are still unannexed. Deriving it from - `git annex find` records `sha256([])` for everything, silently, with - green tests — and couples the digest to the annex backend, which is - deliberately not pinned. -- **The reset takes what the output's id names, never the directory** — - outputs share a directory and Dask writes them concurrently, so a - whole-directory delete would take a neighbour's bytes with it. The - glob is `.*` plus the sidecar: an id cannot contain a dot, - so it cannot reach a sibling, and it *does* reach a payload left by a - run that declared another `format`. -- **A payload that is not a regular file fails the task.** `data_version` - hashes a directory perfectly happily, so `mkdir {output}` would - otherwise commit a well-formed digest of something that is not the - output — and exit 0 is not evidence that anything was written. -- **No git in here.** The driver commits; a worker that asked git - would race the index lock and could read a HEAD this same run moved. -- **`main`'s "no output ``" message covers the task lookup only.** - It once wrapped the whole body, and a `KeyError` from anywhere - inside astra surfaced as "bad target" — a rerun misdiagnosing itself - at the one place nobody is watching. -- **Keep it cheap to import — no click, no rich.** It is on the path - of every task and every rerun; two tests pin the imports and the - absence from `--help`. (Nothing pins the absence of a - `[project.scripts]` entry — treat that as a review item.) - -## Tests - -`tests/test_worker.py` — real recipes through the real boundary -against a real repository (the `analysis` fixture): whether gates -hold and bytes land are not questions a stub can answer. -`tests/test_gpu_execution.py` checks real subprocess masks and unsupported runtime -refusal before output deletion; it requires no physical GPU. diff --git a/docs/architecture.md b/docs/architecture.md deleted file mode 100644 index f5fdd193..00000000 --- a/docs/architecture.md +++ /dev/null @@ -1,232 +0,0 @@ -# Architecture - -How lightcone-cli is put together, for someone about to change it. The -[user-guide concepts page](user/concepts.md) covers what the tool -promises; this page covers how the promises are kept. - -## The split that everything else follows - -```text -lc (CLI) engine ASTRA -───────────── ───────────────────── ───────────── -flags, rendering, ──► what a project is, ──► what a spec -exit codes how outputs are made *means* -``` - -- **`cli/commands.py`** owns flags, console rendering, and exit codes — - nothing else. It imports the engine *inside* command callbacks, so - `lc --help` stays cheap. The engine never imports click and never - prints. -- **The engine** owns everything about what a project is and how - outputs get made. It raises `ProjectError`; the CLI's group class - translates that into a clean error message, once, for every verb. -- **ASTRA** owns what a spec means. Scoping, `from:` references, - conditional outputs, universe resolution, and the recipe placeholder - grammar are all answered by `astra.resolve` and checked by - `astra.validation` — never re-implemented here. When the spec's - *meaning* looks wrong, the fix is a PR to astra-tools. - -The engine ships as the `lightcone.*` PEP 420 namespace — -`src/lightcone/` has **no `__init__.py`**, so sibling distributions can -share the namespace. The engine is the host's `uv tool`, never a -project dependency: a project's lock carries only what the analysis -imports. - -## One run, end to end - -```text -lc materialize "$CLUSTER" - │ guard: tools? git identity? - │ refuse: dirty tree - │ plan: astra validate + resolve → Graph of Tasks - │ (no tasks → converge the crate and stop; nothing connects) - │ classify: current/behind outputs become values; other outputs may run - │ connect: native identity + Dask readiness - │ admit: resource requests for outputs that may run fit a worker - │ fetch: git annex get (declared inputs not in this clone) - │ converge: uv.lock ⇄ .venv (and the image, containerized) - ├─► workers: reset output file → sandbox → recipe → hash → manifest - │ (never raise; return ok/current/behind/failed/blocked) - └─ driver: consume results in one thread - ok → dataset.save (commit + run record) - failed → dataset.restore (tree as clean as it started) - then → converge ro-crate-metadata.json (if licensed) -``` - -The division of labor is strict and load-bearing: - -- **The driver owns git, alone.** Workers execute and return a - `TaskResult`; the driver commits as results arrive, in one thread. - Concurrent git operations race on the index lock — this split is not - a preference. -- **Dask owns the ordering.** Tasks that may execute are submitted with their - upstream futures or already-current values as arguments; there is no ready-set loop or - hand-rolled topological sort on the execution path. -- **The worker never raises.** A recipe failure, a gate failure, an - unreadable manifest — all come back as a state, so one failure - doesn't abort every task in flight, and a run reports *all* its - independent failures. -- **Dask accounts for task resources.** Workers advertise CPU, memory, and GPU - budgets; submissions reserve the recipe's requirements. Omitted memory adds no - RAM reservation. These coordinate scheduling rather than imposing per-recipe - OS limits or numerical-library thread counts. - Recipe `time_limit` is explicitly refused; allocation walltime remains supported. - A GPU recipe reserves the worker's full GPU budget and inherits its allocation - mask, even when it requests fewer GPUs. CPU recipes expose none; probes reserve - the whole worker budget. Direct and podman-hpc probes use the allocation mask, - while ordinary Docker and Podman probes use no GPUs. -- **Values are resolved once and handed down.** HEAD, the container - runtime, and the foreign-write facts are read by the driver and - passed to workers as values — a worker that asked git itself could - get a different answer mid-run, and workers have no git anyway. - -## Identity: two hashes, three states - -`identity.py` computes two digests that deliberately answer different -questions: - -- **`definition_version`** = hash(rendered recipe ‖ decisions) — what - the spec says the output *is*. When it moves, the artifact - contradicts the spec: **stale**, remade. -- **`env_version`** = hash(lock bytes ‖ interpreter pin ‖ install - settings ‖ image document) — what the output *ran under*. When it - moves, the artifact is merely from another time: **behind**, - reported, left alone. - -`assets.classify` is the one implementation of the rule, with two -callers: the worker (live input digests) and the read-only walk -(`None` for anything upstream that will run — "this is going to -change"). That single value is the entire difference between run and -check, which is what keeps `--check` honest. `behind` does not -propagate; `stale` wins when both apply; and a foreign write (an -output whose file or manifest was last touched by a commit that is not its own run -record) classifies stale through the same rule, as one more input -value. - -Both hashes are length-framed (label, length, bytes per field), so a -boundary shift between concatenated fields cannot produce a collision. -The lock is hashed as raw bytes, never parsed — over-invalidation -costs a report line; a parse that disagrees with uv costs correctness. - -## Storage: the repository is the record - -`dataset.py` is the whole git + git-annex seam. The model is DataLad's: -git carries pointers and history, the annex carries bytes, and -`.gitattributes` routes content (`annex.largefiles=nothing` by -default; `data/` and `results/` opt out). A researcher only ever types -ordinary `git add` / `git commit`. - -That ordinary `git add` dispatches git-annex from the *researcher's* -`PATH`, and a shell that cannot resolve it stages the raw bytes into -git history while exiting 0 — so `lc init` sets -`filter.annex.required=true`, which makes git refuse loudly instead -(every filtered command, not only `git add`). -Getting git-annex onto that `PATH` is the install's job, not the -repository's: `uv tool install lightcone-cli` puts it there alongside -`lc`. - -Each output is committed with a **run record** — a `[DATALAD RUNCMD]` -commit message whose `cmd` reconstructs the engine -(`uv run --no-project --with lightcone-cli==`) and re-executes the -worker entry point, so `datalad rerun` replays the making of an output -with the gates, the sandbox, and the manifest intact. Results are -committed *thin* (hard-linked to their annex object), which is safe -precisely because lc never writes an output in place — the worker -resets the directory first. - -## The exec boundary - -Every recipe and every `lc run` command goes through -`engine/sandbox/`: a `Policy` (mechanism-free path sets) is turned -into *a different argv that sandboxes itself* by a `Backend` — -Landlock via the stdlib-only shim `lightcone/_sandbox_exec.py`, -Seatbelt via `sandbox-exec`, the OCI mount table in containerized -mode, and `Unavailable` (wrap = identity) where no mechanism exists. -Because every backend is a pure argv rewrite, all of them are testable -on a host that can't run them, and the manifest's `hermeticity` field -records what was *actually* enforced — never what should have been. - -There is one policy builder, `exec_policy`: probes and recipes share environment -and filesystem rules, with write scope and GPU visibility supplied by the caller. -GPU recipes and supported probes inherit the worker's allocation mask; CPU -commands get an empty mask. - -## The container hatch - -Containerized mode changes the recipe's world and nothing else. -`image.py` (pure) turns the `[tool.lightcone.image]` declaration into -a rendered Containerfile, an identity document, and a content tag; -`container.py` (impure) builds it, saves it as a `docker-archive` -inside the repository (`.datalad/environments//image`, annexed), -and enters it. The engine never enters the image — driver, git, and -classification stay on the host; exactly two things run in-image: the -environment sync and each recipe exec, over a read-only rootfs with -the mount table as the whole policy. Execution pins the archive's -config-blob id, never a tag. - -## Compute allocations - -`engine.compute` owns allocation lifecycle through a small provider protocol. -A YAML catalog supplies ordered resource offers and stable native service -namespaces. When the implicit default file is absent, a built-in local catalog -provides all detected usable CPUs and RAM without setup. The `local` policy can -override that budget or disable local compute. Remote catalogs retain the default -local offer; explicit local connections supply their own offers. A runtime guard -disables local compute on recognized NERSC login nodes while permitting interactive -compute nodes. GPU offers need an explicit catalog. Missing explicit paths and invalid files -remain errors. No catalog is written and -no allocation starts until `compute launch` resolves resources and submits once. Slurm queries -and validated local OS identities are authoritative for allocations; Dask is the -authority for connected workers. Private scheduler/TLS files are connection -material, not a registry. - -Allocation requests use SkyPilot-style CPU/memory exact or minimum quantities -and one accelerator type/count. Providers translate those requests into native -allocations; Lightcone does not depend on SkyPilot or carry its GPU alias registry. -Stock Dask workers inherit the allocation's native CUDA mask; Lightcone does not -probe GPU hardware. Container GPU access uses podman-hpc's native `--gpu` option. -See [GPU setup](user/cluster.md#gpu-allocations). - -`compute.connect(CLUSTER_ID)` borrows a standard Dask client and closes only that -client on exit. Both execution commands require a cluster ID. The materialization -scheduler validates resource requests, then keeps its `submit`/`completed` seam. -Driver preparation and existing task runtime/sandbox checks remain unchanged. -Tasks use ordinary Dask scheduling; -there is no separate worker-selection or preflight layer, or site-marker guard. No execution command implicitly allocates compute. -See [compute internals](api/compute.md) and [deployment limits](user/cluster.md). - -## The publication view - -`crate.py` renders the repository as a Provenance Run Crate — a pure -function of repository state (sorted iteration, no clock, git injected -as a callable), which is what lets `materialize` converge -`ro-crate-metadata.json` byte-for-byte and commit only differences. -Run identity comes free from the manifests' `git_sha` (the driver -reads HEAD once per run), so one materialize maps onto one -`OrganizeAction` with no new manifest field. - -## Repository at a glance - -```text -src/lightcone/ # namespace — NO __init__.py -├── _sandbox_exec.py # the Landlock shim — stdlib only, zero lightcone imports -├── cli/commands.py # flags, rendering, exit codes — nothing else -└── engine/ - ├── project.py # what a project is: convergence, discovery, mode - ├── dataset.py # the git + git-annex seam - ├── identity.py # env_version, definition_version, the lock scan - ├── image.py # the system layer, declared → rendered — pure - ├── container.py # runtimes, the build, the archived image — impure - ├── crate.py # the publication view — pure - ├── assets.py # an output: directory, manifest, state - ├── plan.py # the spec, read as a graph of tasks - ├── worker.py # making one output; the rerun entry point - ├── materialize.py # the driver: gates, Dask, the save/restore loop - ├── run.py # what `lc run` is - ├── compute/ # common allocation API, local and Slurm providers - ├── sandbox/ # the exec boundary - └── templates/ # the scaffold's file content, as real files -``` - -Each module's page in [Engine Internals](api/index.md) carries its -public surface and the invariants that bind it. diff --git a/docs/assets/favicon.svg b/docs/assets/favicon.svg deleted file mode 100644 index 011c383a..00000000 --- a/docs/assets/favicon.svg +++ /dev/null @@ -1 +0,0 @@ - \ No newline at end of file diff --git a/docs/assets/logo.svg b/docs/assets/logo.svg deleted file mode 100644 index 011c383a..00000000 --- a/docs/assets/logo.svg +++ /dev/null @@ -1 +0,0 @@ - \ No newline at end of file diff --git a/docs/cli/build.md b/docs/cli/build.md deleted file mode 100644 index f4603a20..00000000 --- a/docs/cli/build.md +++ /dev/null @@ -1,90 +0,0 @@ -# lc build - -Build the project's system-layer image, and commit it. Containerized -projects only — a project containerizes by declaring a -`[tool.lightcone.image]` table in `pyproject.toml`, and on a direct -project this verb just says so and exits. - -## Synopsis - -```text -lc build [OPTIONS] -``` - -Idempotent: an image that is already built and committed is left -alone. - -## What the image is - -The image is the *system layer* only: the declared base (digest-pinned, -or the default), the declared apt packages, and the pinned Python -interpreter. Your analysis environment is not in it — recipes' Python -packages come from the project's lock, synced into the container at run -time — and neither is `lc` itself. That is what makes "editing code -never rebuilds the image" structural: no project file enters the build -context at all. - -The declaration is a closed set of keys, hashed into the image's -identity: - -```toml -[tool.lightcone.image] -base = "docker.io/library/debian@sha256:..." # optional; default pinned by lc -apt-install = ["libfftw3-dev"] # optional -run-commands = ["curl -L ... | tar xz"] # optional, the bounded escape -env = { OMP_NUM_THREADS = "1" } # optional -``` - -## The archive is the store - -`lc build` saves the built image into the repository — -`.datalad/environments//image`, a `docker-archive` committed -through git-annex — so the exact bytes travel with the project: a -clone obtains them with a fetch, no registry and no credentials -involved. Execution always pins the image's content *id*, never a tag, -so nothing can substitute a different image under the same name. - -The archive records the architecture it was built for, and a host that -can't execute that architecture is refused up front — build where the -architecture matches the machines that will run recipes (on NERSC, a -login node). - -## Requirements - -- A clean tree — the image commit must not sweep your staged edits in, - and the tag derives from the committed declaration. -- A build-capable runtime: `podman-hpc`, `podman`, or `docker` - (detected in that order; nothing to configure). - -`lc materialize` also builds as a preflight when the committed -declaration has no image yet, announcing it first — `lc build` exists -so you can pay the minutes when *you* choose to. - -## Options - -| Option | Default | Effect | -|--------|---------|--------| -| `--json` | off | Emit the result as JSON on stdout. | - -## The JSON result - -```json -{ - "mode": "containerized", - "tag": "lc-env-1a2b3c4d5e6f7a8b", - "id": "sha256:...", - "archive": ".datalad/environments/lc-env-1a2b3c4d5e6f7a8b/image", - "action": "built" -} -``` - -`action` is `"built"` when this invocation built and committed the -image, `"present"` when it was already there. On a direct project the -result is just `{"mode": "direct"}`. - -## Examples - -```bash -lc build # build + commit, or confirm it's already there -lc build --json # the machine-readable form -``` diff --git a/docs/cli/compute.md b/docs/cli/compute.md deleted file mode 100644 index e30cc802..00000000 --- a/docs/cli/compute.md +++ /dev/null @@ -1,149 +0,0 @@ -# lc compute - -Manage explicitly allocated Dask clusters using resource offers. -No project is required for these commands. - -```text -lc compute resources [--json] -lc compute launch [--cpus VALUE --memory VALUE] [--gpus NAME[:COUNT]|0] - [--name NAME] [--num-nodes N] [--time DURATION] [--startup fast] [--dry-run] [--wait] [--timeout SECONDS] [--json] -lc compute status [CLUSTER] [--wait] [--timeout SECONDS] [--json] -lc compute down CLUSTER [--json] -``` - -With no resource flags, `lc compute launch` starts the default local CPU offer, -names the cluster `local`, and uses all detected usable logical CPUs and RAM. -The built-in offer has one node, fast startup, and no fixed lifetime: it ends -after 30 minutes without task activity. `--name` overrides the name, and `--time` -adds a hard lifetime that ends the cluster even while work is running. -CPU and RAM are cooperative scheduling budgets, not exclusive reservations. -GPUs still require explicit offers and, locally, a `CUDA_VISIBLE_DEVICES` mask. - -`~/.lightcone/compute.yaml` configures resource offers; `LC_COMPUTE_CONFIG` selects -another file for all compute and execution commands. The top-level `local` block -can override the built-in CPU/RAM budget or time limits, or disable local compute. -Each offer names its `provider` (`local` or `slurm`). The built-in local offer is -appended after configured offers, unless the catalog lists its own local offers; -the shortcut chooses the first eligible local offer. See -[local configuration](../user/cluster.md#customize-resource-offers). -Missing explicit paths and invalid catalogs are errors. Loading a catalog or -planning with `--dry-run` creates no files or compute. - -Local launch and execution are automatically disabled on recognized NERSC login -nodes, including the first run without a catalog. Interactive compute nodes remain -eligible. This runtime guard creates no configuration file and cannot be overridden -by `local.enabled: true`; Slurm, status, and termination remain available. -See [local allocations](../user/cluster.md#local-allocations) for detection details. - -| Command | Behavior | -|---|---| -| `resources` | Ordered available offers, per-node shape, node limit, default/maximum walltime, idle timeout, and startup class. Free capacity remains unknown. | -| `launch` | Resolve one resource request and submit exactly once; print only the cluster name to stdout on acceptance. | -| `launch --wait` | Submit once, then wait for all expected workers. `--timeout` sets the readiness deadline (default 300 seconds). | -| `launch --dry-run` | Show the resolved shape and native launch parameters without allocation. | -| `status` | List one `name: status` line per current allocation across every provider the catalog's offers use, and always local; report the providers that could not be queried. | -| `status CLUSTER` | Resolve a name or full ID, inspect native state, and probe Dask readiness separately. | -| `status CLUSTER --wait` | Wait for readiness: the allocation is active and every node's worker is connected. The default deadline is 300 seconds, and queries grow less frequent as the wait goes on (up to every 30 seconds). Exits 1 on timeout, or at once if the allocation is ending or has ended; the allocation is left unchanged. | -| `down CLUSTER` | Request native termination even if the scheduler is unavailable. An allocation that has already ended is a successful no-op when addressed by full ID; its name no longer resolves. A Slurm job that has left the queue is refused unless accounting confirms it ended. | - -The local shortcut defaults to `local`. For explicit CPU/memory requests, omitting -`--name` generates `lc-` followed by 12 random hexadecimal characters. -Use `--name analysis` to choose either name yourself. Names contain 1–63 lowercase ASCII letters, -digits, or hyphens; they start with a letter and end with a letter or digit. -Launch guidance goes to stderr, so the default output can be captured directly: - -```bash -CLUSTER=$(lc compute launch --wait) -lc compute down "$CLUSTER" -``` - -Launch checks current allocations across all of the catalog's providers. An explicit -name already in use is rejected; generated collisions are retried before the -single submission. This check is not an atomic reservation: concurrent launches -can race. Name lookup refuses ambiguous or incomplete discovery instead of -choosing a cluster. A name can be reused after its allocation ends; it is not a -durable reference to that allocation. Use the full immutable `id` from launch or -status JSON to address one allocation directly, including when other providers -are unavailable or no offer uses its provider any more. No name registry is -maintained. - -Supply both `--cpus` and `--memory`, or omit both for the local shortcut. -The shortcut never selects a remote offer, including when local compute is disabled. - -Resource quantities are **per node**, and `--num-nodes` defaults to one. CPU and -memory requests follow SkyPilot's exact/minimum convention: `4` is exact and `4+` -means at least four. Compute memory uses binary units: `16`, `16GB`, and `16GiB` -all mean 16 GiB; `16GB+` permits a larger offer. - -`--gpus A100:4` requests exactly four GPUs from an offer whose accelerator type is `A100`; -`--gpus A100` means one, `--gpus GPU:4` accepts any GPU model, and the default -`--gpus 0` selects CPU-only offers. Names match case-insensitively. GPU counts -are positive whole numbers, with no `+` or fractional form. Lightcone does not -maintain SkyPilot's accelerator alias registry: use the labels configured in -`resources` or use `GPU:N`. Catalog shapes use `accelerators: A100:4` or -`accelerators: {A100: 4}`. See [GPU allocations](../user/cluster.md#gpu-allocations). - -Time accepts positive durations with day/hour/minute/second units, such as `30m`, -`1h30m`, or `45s`. Without -`--time`, the chosen offer's default walltime applies, if it has one. `fast` is a -service class, not a queue-time promise. Limits apply to each allocation; aggregate -quotas remain with the native backend. - -A local allocation ends at its walltime, after its idle timeout, or at whichever -comes first when it has both. The idle timeout is Dask's scheduler -`idle-timeout`: running or queued tasks keep the allocation alive, and new work -restarts the countdown; connected clients and `status` queries do not. `lc run` -and `lc materialize` restart it when they connect, so their preparation (fetching -inputs, building the image, syncing the environment) starts with the full timeout. -When it expires, the allocation ends, `status` gives that as the reason, and both -its name and this machine's one local allocation are free again. A walltime ends -the allocation even during active work. `down` still ends it at once. - -For Slurm, time is a finite native `--time` request, so a Slurm offer needs a -`time.default` and cannot declare `time.idle`: - -```text -Error: Slurm allocations end at their native walltime: set the offer's time.default and remove time.idle -``` - -Slurm's overtime and -termination-grace policy determines actual expiry and can allow unlimited -overrun; Lightcone supplies no independent Slurm runtime deadline. A partition -is passed only when explicitly set in the offer's configuration. - -The first eligible offer wins; an offer this host cannot provide is skipped, and -the error lists why when nothing matches. An invalid configuration or failed -submission is an error, with no automatic resubmission elsewhere. An uncertain submission error -includes its token and any known cluster ID. Inspect existing allocations before -retrying it. - -On `status`, `--wait` requires a CLUSTER. On both commands, `--timeout` requires -`--wait`. Launch rejects `--wait --dry-run`. A waiting launch prints its cluster -name only once ready; JSON adds `ready: true`. A timeout or startup failure exits -1 and includes the accepted immutable ID in the error. Waiting never resubmits -or terminates the accepted allocation. - -Only one local cluster may run per user on a machine, across names and -configured roots. A launch that finds one of your local clusters running in -the process table fails until that cluster ends; launches that overlap can both -succeed. - -`--json` emits versioned (`schema_version: 1`), allowlisted data without -scheduler credentials: - -| Command | Keys | -|---|---| -| `launch` | `plan`, `id`, `name`, `accepted` (`ready: true` after `--wait`; only `plan` with `--dry-run`) | -| `status CLUSTER` | `id`, `name`, `phase`, `allocation`, `dask`, `reason`, `native_state` | -| `status` | `clusters` (a list of the above) and `errors` (by provider name) | -| `down` | `id`, `name`, `termination_requested` | -| any failure | `error`, `id`, `submission_token`, on stdout, with exit 1 | - -`phase` is `pending`, `active`, `stopping`, `ended`, or `unknown`. `allocation` -holds `num_nodes`, per-node `resources`, and their `evidence` (`configured`, -`requested`, or `unknown`). Resource objects contain `cpus`, `memory` in GiB, -and `accelerators` as a one-entry type/count mapping, or `null` for CPU-only shapes. -`dask` is observed separately: `observation` -(`unverified`, `reachable`, or `unreachable`), `ready`, and `workers`. A -discovery that partially succeeds still exits 1. Native errors, invalid -requests, and readiness timeouts also exit 1. diff --git a/docs/cli/index.md b/docs/cli/index.md deleted file mode 100644 index 0ab2b46d..00000000 --- a/docs/cli/index.md +++ /dev/null @@ -1,54 +0,0 @@ -# CLI Reference - -The `lc` CLI is a thin wrapper around the engine. The user-facing -surface is small on purpose — `astra.yaml` carries the analysis -description, and the CLI is the durable, scriptable way to execute -and audit it. - -## Global behavior - -- **The current directory is the project.** Every command except - `init` assumes it is invoked from the project root; there is no - walk-up and no global configuration. Outside a project, a command - errors cleanly. -- **Nothing waits on a human.** No command prompts or opens an - interactive shell — every verb runs to completion on its arguments - alone, which is what makes the CLI safe to drive from scripts and - agents. -- **Refusals carry their remedy.** When a command refuses (a dirty - tree, an unavailable cluster, a missing image), the message names the exact - command that fixes it. - -## Commands - -| Command | Purpose | -|---------|---------| -| [`lc init`](init.md) | Converge a directory into a Lightcone project (idempotent). | -| [`lc materialize`](materialize.md) | Make the analysis's outputs; commit each one as it lands. | -| [`lc status`](status.md) | Report the state of every output. Reads only; always exits 0. | -| [`lc compute`](compute.md) | Allocate resources, inspect clusters, and end allocations. | -| [`lc run`](run.md) | Run an ad-hoc command in the project environment, under isolation. | -| [`lc build`](build.md) | Containerized projects: build the image and commit it. | - -## Global options - -```text -lc [OPTIONS] COMMAND [ARGS]... - -Options: - --version Show the version and exit. - --help Show this message and exit. -``` - -## Exit codes - -- `0` — the command did what it says. -- `1` — a refusal or a failure, with the reason on stderr. For - `lc materialize --check` and `lc init --check`, exit 1 means "work - would be done" — the gate form scripts branch on. -- `lc run` is a proxy: it exits with the command's own code - (`128 + N` for a signal), so pipelines read it exactly as they would - the bare command. - -Every verb with a report takes `--json` for the machine-readable form; -each verb's page shows its shape. diff --git a/docs/cli/init.md b/docs/cli/init.md deleted file mode 100644 index 84990e06..00000000 --- a/docs/cli/init.md +++ /dev/null @@ -1,135 +0,0 @@ -# lc init - -Converge a directory into a Lightcone project. Idempotent — safe to run -at any time, on an empty directory, a half-scaffolded one, an existing -project, or a fresh clone. - -## Synopsis - -```text -lc init [OPTIONS] [DIRECTORY] -``` - -`DIRECTORY` defaults to `.` (the current directory). - -## Convergence semantics - -Each run creates whatever is missing, repairs the pieces lightcone -manages, and never overwrites files you own: - -- **Created if missing** — every item in the tree below. A directory - that already holds an `astra.yaml` is *adopted*: the spec is left - untouched and only the missing lightcone pieces are added. A - directory inside an existing git repository adopts that repository - rather than nesting a new one. -- **Repaired** — derived artifacts that have drifted: a `uv.lock` that - no longer matches `pyproject.toml`, a `.venv` that no longer matches - the lock, a managed `.gitignore` or `.gitattributes` entry that a - newer `lc` added, an annexed repository still missing the - `filter.annex.required` flag. Repairs only ever append or rebuild - derived state; - hand-written lines are never reordered or removed. -- **Blocked** — something convergence can see but must not fix by - appending: a `.gitignore` rule that would silently swallow - `results/`, a `.gitattributes` whose ordering would misroute storage. - A blocked item names the file and line at fault, counts against - convergence, and is yours to resolve. -- **Warned about** — advisory facts (e.g. uv falling back to file - copies across filesystems). Warnings never affect the exit code. - -`--check` computes the same report without writing anything and exits -`1` when the project is not converged. `--json` prints it -machine-readable: - -```json -{ - "converged": true, - "created": [], - "repaired": [], - "unchanged": ["astra.yaml", "pyproject.toml", "..."], - "blocked": [], - "warnings": [] -} -``` - -Agents driving a project should run `lc init --check --json` at the -start of a session to make sure the directory is workable. - -## What it creates - -Inside `DIRECTORY` (creating it if needed): - -```text -astra.yaml # an empty analysis spec, ready to fill in -universes/ - baseline.yaml # the default universe (selects nothing yet) -pyproject.toml # the uv project — the environment's source of truth -.python-version # the exact interpreter, pinned -uv.lock # derived: converged by correctness, not existence -.venv/ # derived: built from the lock (local, never committed) -.gitignore # managed entries, converged line-wise -.git/ # a git repository, with git-annex initialized -.gitattributes # the storage policy: what the annex carries -.datalad/config # dataset identity (a DataLad dataset from birth) -data/ + README.md # declared input data lives here -results/ + README.md # outputs land here — lc's to write -myst.yml # MyST report configuration -index.md # template report, to reference astra.yaml from -``` - -Two things it deliberately does *not* create: a `src/` directory -(where analysis code lives is your layout, and git doesn't track empty -directories), and any dependency in `pyproject.toml` — the lock -carries only what *your* analysis imports, added with `uv add`. - -Inside `.git`, convergence sets one configuration key — reported as the -`annex-filter` item: - -- `filter.annex.required=true`, always. Without it, a `git add` whose - shell cannot find git-annex prints an error, **exits 0, and stages - the raw bytes into git history** — a 2 GB dataset in git proper, on - every clone, forever. With it, the same situation is a hard, loud - failure and nothing is staged. Once the project holds committed - annexed content that refusal covers every command that must run the - filter, `git status` and `git diff` included — a project you cannot - use until git-annex is back, rather than one that silently absorbed - your data. - -That is the only thing `lc init` adds to what `git annex init` wrote. -How git finds git-annex is still ordinary `PATH` resolution, which is -why `lc` should be installed with `uv tool install lightcone-cli` — it -puts `git-annex` on your `PATH` alongside `lc`. If your `git add` ever -refuses, see -[`fatal: … clean filter 'annex' failed`](../user/troubleshooting.md#fatal-clean-filter-annex-failed) -in the troubleshooting guide. - -## Options - -| Option | Default | Effect | -|--------|---------|--------| -| `--check` | off | Report drift without writing; exit 1 if not converged. | -| `--json` | off | Emit the convergence report as JSON on stdout. | - -There is deliberately nothing else — no `--no-git`, no template -selection. The project layout is the contract the other verbs rely on. - -## Examples - -```bash -lc init # converge cwd -lc init my-analysis # scaffold/converge ./my-analysis -lc init --check --json # is this directory workable? (for scripts/agents) -lc init # in a fresh clone: rebuild .venv + the annex -``` - -## Next steps - -```bash -cd my-analysis -# Describe your analysis in astra.yaml — inputs, outputs, recipes, -# decisions — and write the scripts the recipes name. -uv add numpy # declare what the scripts import -git add -A && git commit -m "First analysis" -lc materialize "$CLUSTER" # make the outputs -lc status # see where everything stands -``` diff --git a/docs/cli/materialize.md b/docs/cli/materialize.md deleted file mode 100644 index a323913c..00000000 --- a/docs/cli/materialize.md +++ /dev/null @@ -1,141 +0,0 @@ -# lc materialize - -Make the analysis's outputs, and commit each one as it lands. This is -the build verb: it validates the spec, converges the environment, runs -every recipe that needs running — in dependency order, in parallel -where the graph allows — and commits each result together with its -manifest, in a commit whose message is a replayable run record. - -## Synopsis - -```text -lc materialize [OPTIONS] CLUSTER [TARGETS]... -lc materialize --check [OPTIONS] [TARGETS]... -``` - -Execution requires a cluster name or full immutable ID from `lc compute launch`. -No cluster is chosen or started implicitly, and the run never waits for one: a -cluster that is not active with every expected worker connected is refused -(`lc compute status CLUSTER --wait` waits for readiness). `--check` needs no -cluster. Project validation and the dirty-tree check run before connecting to -compute. A run whose spec selects no outputs still takes the CLUSTER argument -but never connects to it: it only updates the publication view. A run whose -outputs are all current still validates the cluster connection, but submits no -Dask tasks and reserves no recipe resources. - -With no targets, everything the spec declares, across every universe. -A target narrows the run to an output and whatever it depends on: - -- `fit` — the output `fit` in every universe that has it. -- `robust/fit` — exactly one universe's output. - -A target that matches nothing is an error listing what exists — -quietly making nothing is the least useful thing a build tool can do. - -## What gets remade - -An output is remade when it is `stale` — the analysis defines it -differently than it was made (a changed recipe or decision), one of -its declared inputs changed content, or it was edited by hand since. -Inputs are compared by content, so a rebuild that comes out -byte-identical stops the cascade there. - -An output that is `behind` — still exactly what the spec asks for, -but made under an earlier environment — is reported and left alone; -`--refresh` widens the run to remake those too. A `current` output is -never touched, under any flag. - -## The run's contract - -- **Starts clean.** A dirty tree is a refusal. A recipe that returns a - failure has its partial work restored. After a cluster interruption, - unreported outputs are retained because tasks may still be running. The - same holds when a commit fails while other recipes are still running: the - error says so. - Stop the allocation with `lc compute down` and its full ID (a name can - already belong to a newer allocation), and confirm its recipes have - stopped before cleaning results. Local containers may need separate - termination through their runtime; see [execution limits](../user/cluster.md#execution-requirements-and-limits). -- **Honors recipe resources.** CPU, memory, and GPU requests must fit one worker - and are reserved through standard Dask scheduling. GPU recipes run one at a - time per worker and inherit the whole allocation's CUDA mask, which may expose - more GPUs than requested. Recipes without `gpus` see none. Resource checks apply - to outputs that may rebuild; already-current outputs need no reservation. - Omitted memory reserves no RAM, so CPU requests and task slots control concurrency. - Recipe `time_limit` is unsupported and refused; allocation walltime remains supported. - See [recipe resource requirements](../user/cluster.md#recipe-resource-requirements). -- **Fetches what it needs.** Declared inputs whose annexed content is - not in this clone are fetched before workers hash or execute. -- **Commits as it goes.** Each output lands in its own commit, written - by the driver in one thread while other recipes keep running. -- **Forwards recipe diagnostics.** Recipe stdout and stderr reach the invoking - terminal on stderr, including failed recipes. stdout remains available for the - report, so `--json` stays machine-readable. -- **Reports every independent failure.** One failing recipe doesn't - abort the rest; its dependents report `blocked` and the run exits 1 - with all of it listed. -- **Maintains the publication view.** With a `[project].license` - declared, the run converges `ro-crate-metadata.json` in a trailing - commit. - -On a containerized project, the run resolves the committed image first -(building it as a preflight if the declaration is committed but the -image never built). Tasks use the explicitly selected cluster and the client -detaches on completion, leaving that cluster available — see [Running on a Cluster](../user/cluster.md). - -## Check mode - -`--check` classifies every output without executing, committing, or -fetching anything, and exits `1` if a run would do work — the gate a -script or CI job branches on. It is exempt from the dirty-tree -refusal: reading the state of a project before deciding what to commit -is what it is for. - -## Options - -| Option | Default | Effect | -|--------|---------|--------| -| `--check` | off | Report what would run and why; exit 1 if anything is out of date. | -| `--refresh` | off | Also remake `behind` outputs. Never touches `current` ones. | -| `--json` | off | Emit the report as JSON on stdout. | - -There is deliberately no `--jobs` (task concurrency belongs to the configured -cluster), no `--force`, and no flag to -*skip* a stale output — deleting its file is your own file -operation, and stronger consent than a flag. - -## The JSON report - -```json -{ - "ok": true, - "up_to_date": true, - "made": [], - "current": ["baseline/fit", "robust/fit", "baseline/fit_plot", "robust/fit_plot"], - "behind": {}, - "failed": [], - "blocked": [], - "planned": {}, - "warnings": [], - "notes": [] -} -``` - -The first two keys are the ones to branch on: `ok` — everything -attempted finished; `up_to_date` — nothing needed doing (a failed run -is never up to date, and `behind` outputs don't count against it). -`planned` is check mode's answer, mapping each would-run output to why; -`behind` maps each left-alone output to the commit that can rebuild its -environment. `notes` carries sandbox messages verbatim — denial -remedies are built to be pasted. - -## Examples - -```bash -lc materialize "$CLUSTER" # everything, all universes -lc materialize "$CLUSTER" fit # one output (and upstreams), every universe -lc materialize "$CLUSTER" robust/fit # one universe's output -lc materialize --check # would anything run? (exit 1 = yes) -lc materialize "$CLUSTER" --refresh # also remake behind outputs -lc materialize --check --json # the machine-readable gate -``` diff --git a/docs/cli/run.md b/docs/cli/run.md deleted file mode 100644 index 312fa587..00000000 --- a/docs/cli/run.md +++ /dev/null @@ -1,91 +0,0 @@ -# lc run - -Run an ad-hoc command in the project environment, under isolation. -This is the probe verb: it executes exactly one command the way a -recipe would be executed — same environment, same sandbox. Recipes must also -declare the resources they need, including their GPU count. - -## Synopsis - -```text -lc run CLUSTER -- COMMAND... -``` - -The first argument is a cluster name or full immutable ID from `lc compute launch`. -Everything after `--` is the command, verbatim — flags included. -Argv, the `docker run` / `uv run` convention: a single quoted string -would be exec'd as one filename, so probe shell syntax through -`bash -c` instead. Set `CLUSTER` to your allocated cluster's name or full ID: - -```bash -lc run "$CLUSTER" -- python -c "import numpy; print(numpy.__version__)" -lc run "$CLUSTER" -- python src/fit.py --points data/points.csv --outliers keep --output /tmp/probe -``` - -The command is submitted as an ordinary task to the cluster's Dask scheduler, -which chooses a worker. The command uses the prepared project environment and -the same sandbox as a recipe. stdout/stderr are forwarded as bytes, preserving binary output and -line endings when redirected. The client detaches on completion; the allocation -stays available until `lc compute down` or its time limit. A missing cluster, -or one that is not yet active with every expected worker connected, is an -error: nothing waits and nothing runs locally instead. Use -`lc compute status CLUSTER --wait` first. See [compute](compute.md). - -The command receives EOF on stdin; terminal input and pipes into `lc run` are not -forwarded. Pass input files through the project's declared inputs instead. -For direct execution, ambient environment variables come from the worker's -allocation environment. Prefixing the CLI with `NAME=value` does not forward -that variable to an existing cluster. Set command-specific values inside the -command, for example `lc run "$CLUSTER" -- env NAME=value python script.py`. -Containerized commands use the image's environment and the sandbox overlays. - -The command reserves one worker's full CPU, memory, and GPU budgets for its duration. -Direct and podman-hpc probes inherit the allocation's whole CUDA mask. Ordinary -Docker and Podman probes run without GPUs and report that limitation, even on a -GPU allocation. CPU-only allocations expose no GPUs. A recipe declares its -minimum GPU count explicitly; unsupported GPU recipes fail before image preparation. -See [GPU allocations](../user/cluster.md#gpu-allocations) for container prerequisites. - -Interrupting the CLI detaches its client; the remote command may still be running. -Stop the allocation with `lc compute down` and its full ID (a name can already -belong to a newer allocation) before working with files the interrupted command -could still be writing. Confirm that the command has -stopped; local containers may require separate termination through their runtime -(see [execution limits](../user/cluster.md#execution-requirements-and-limits)). - -## What it does - -- **Converges the environment first.** The probe syncs `.venv` to the - lock before executing, so what you probe is what a recipe gets. -- **Applies the recipe policy.** The project tree is read-only apart - from `results/`, declared inputs are readable, undeclared tools - don't execute. On a containerized project, the command runs inside - the committed image (which must already be built — the probe never - builds). -- **Proxies the exit code.** `lc run` exits with the command's own - code — `128 + N` when a signal killed it — so scripts and pipelines - read it exactly as they would the bare command. -- **Explains denials.** On a nonzero exit, a note on stderr says the - command ran sandboxed; when the failure looks like a denial, the - note names the path and the remedy (`uv add` for a missing package, - an ASTRA input declaration for data, `results/` or - `tempfile.mkdtemp()` for writes). - -A probe has no output and writes no manifest: nothing it does is -recorded anywhere. Any uv project works — `lc run` doesn't require an -`astra.yaml`, only `pyproject.toml`, `uv.lock` and `.venv` in the -current directory. - -## What it is not - -There is no sandbox opt-out and no flag surface — a command that needs -more than the policy grants is a command that would fail as a recipe, -and the fix (declare the dependency) is the same in both places. - -## Examples - -```bash -lc run "$CLUSTER" -- python -c "import scipy" # is the package in the lock? -lc run "$CLUSTER" -- bash -c 'echo $HOME' # see the private HOME a recipe gets -lc run "$CLUSTER" -- python src/fit.py --help # exercise a script exactly as a recipe would -``` diff --git a/docs/cli/status.md b/docs/cli/status.md deleted file mode 100644 index fae5fdb9..00000000 --- a/docs/cli/status.md +++ /dev/null @@ -1,90 +0,0 @@ -# lc status - -Report what state each of the analysis's outputs is in. Reads only: it -runs nothing, commits nothing, transfers no data, does not mind an -unclean tree, and always exits `0` — a state is not a failure. The -moment you most need to know where a project stands is when it isn't -clean, so this verb works there. - -## Synopsis - -```text -lc status [OPTIONS] -``` - -## Output - -```text - mode: direct - sandbox: landlock (fs: declared, network: allowed) - - · current baseline/fit a3f1f11 - · current baseline/fit_plot a3f1f11 - · behind robust/fit 00cc14e made under an earlier environment - ! stale robust/fit_plot — no manifest — it has never been materialized - -2 current · 1 behind · 1 stale -``` - -The header is repository facts: which mode the project executes in -(and, for a containerized project, the image's tag and state), and -what enforcement a run on this host would get. No runtime and no -network is needed to answer either. - -Then one line per output the spec declares, in dependency order: its -state, **the commit it was made at**, and — for anything not current — -why. The commit column is the verb's reason to exist: "which code made -this?" has an answer for a current output too, and for a `behind` -output that commit is where the environment that produced it can be -read back. - -## States - -- `current` — exactly what the spec asks for. Nothing to do. -- `behind` — still what the spec asks for; the environment moved since. - Left alone by runs; `--refresh` remakes. -- `stale` — contradicts the project: definition changed, an input's - content changed, or the output was edited by hand since it was made - (a *foreign write* — the offending commit is named). - -## Report vs gate - -`lc status` reports; **`lc materialize --check` gates.** Two verbs -answering the same question with different exit codes is how a script -comes to depend on the wrong one, so the split is sharp: use status for -eyes, check for exit codes. - -## Options - -| Option | Default | Effect | -|--------|---------|--------| -| `--json` | off | Emit the report as JSON on stdout. | - -## The JSON report - -```json -{ - "mode": "direct", - "image": null, - "sandbox": "landlock (fs: declared, network: allowed)", - "counts": {"current": 4, "behind": 0, "stale": 0}, - "outputs": [ - { - "output": "baseline/fit", - "status": "current", - "why": "", - "git_sha": "a3f1f11791430d1becbe5548477b5910ab59a94a", - "data_version": "sha256:939e9a55...", - "foreign_write": "" - } - ], - "warnings": [] -} -``` - -Per output: the state, the reason (empty for `current`), the commit it -was materialized at and its content identity (both empty if it never -was), and `foreign_write` — the sha of a hand-edit's commit when one -was detected, which the prose `why` cannot carry for a machine -consumer. For a containerized project, `image` is -`{"tag": ..., "state": "present" | "absent" | "unfetched"}`. diff --git a/docs/contributing/extending.md b/docs/contributing/extending.md deleted file mode 100644 index d3f650dd..00000000 --- a/docs/contributing/extending.md +++ /dev/null @@ -1,55 +0,0 @@ -# Extending the Codebase - -Where each kind of change belongs, what to read first, and the -invariant it must keep. The engine has one implementation per rule — -most review feedback is some form of "that spelling already exists; -use it". - -## The map - -| To change… | Edit | Keep true | -|---|---|---| -| What a scaffolded file contains | `engine/templates/files/*.tmpl` (+ `test_templates.py`) | A template gets a function only when a value must be decided or a merge policy held. | -| What gets converged | `engine/project.py` (+ `test_project.py`) | Everything through `_Converger.item`/`.file`/`.blocked`; repairs only append; only what git can carry. | -| How a project stores bytes | `engine/dataset.py` + `gitattributes.tmpl` (+ `test_dataset.py`, real annex) | Every command through `project._run`; nobody is asked to run git-annex. | -| How an output is identified | `engine/identity.py` (+ `test_identity.py`) | Sensitivity both ways: what must move the hash, what must not. Length-framing stays. | -| When an output is remade | `engine/assets.py` (+ `test_assets.py`) | One `classify`; callers differ by one input value, never by logic. Ask first: does the change *contradict* the project (stale) or is it *circumstance* (behind)? | -| How the spec becomes a graph | `engine/plan.py` (+ `test_plan.py`) | Ask `astra.resolve`; a missing answer is a PR to astra-tools; ambiguity is a `ProjectError`, never a guess. | -| How a recipe runs | `engine/worker.py` (+ `test_worker.py`) | Never raises; no git; mutation-check every denial test. | -| What a run commits | `engine/materialize.py` (+ `test_materialize.py`) | The driver owns git alone; the tree ends as clean as it started. | -| Where a run executes | `engine/compute/` + `cluster_for_run` (+ `test_compute*.py`) | Explicit allocation IDs; implement the provider protocol and register one factory. Execution borrows standard clients. | -| Supporting a new HPC center | Compute catalog | Expose resource offers with the site's native Slurm settings; there is no hostname-based placement guard. | -| What a sandboxed command may touch | `sandbox/policy.py` (+ `test_sandbox_policy.py`) | Path sets only — no mechanism leaks in. | -| Adding a sandbox mechanism | one module in `sandbox/` + one line in `detect()` | `wrap` pure, `attest` honest, `contains_prefix` answered. Nothing above the seam changes. | -| A denial message | `sandbox/denial.py` (+ `test_sandbox_denial.py`) | Remedies copy-pasteable and real *today*; the trailer stays unconditional. | -| What the image is made of | `engine/image.py` (+ `test_image.py`) | Pure; every declaration key hashed; structure tests, never byte goldens. | -| How images are built/stored/entered | `engine/container.py` + `sandbox/oci.py` (+ `test_container.py`) | `runtime_for_run`'s two strictnesses; runtime differences are spellings inside `OCIBackend`, never new shapes. | -| What the crate says | `engine/crate.py` (+ `test_crate.py`) | Pure builder: sorted, no clock, git injected; render-twice-identical. The validator floor lives in `test_crate_smoke._FLOOR`. | -| How a foreign write is detected | `dataset.last_writer` + `materialize._foreign_write` | History, never hashing; `datalad_run_subject` is the one spelling of the record's subject. | -| A CLI verb | `cli/commands.py` (+ `test_cli.py`) | Logic in the engine; raise `ProjectError`; render here; engine imports stay inside callbacks. | - -## Rules that apply everywhere - -- **Land code, tests, and dependencies together.** A dependency enters - `pyproject.toml` with the change that needs it, never speculatively. -- **No dead code, no foreshadowing.** Nothing references a verb, flag, - or feature that doesn't exist yet; `lc --help` advertises only what - works. -- **No escape hatches.** Enforcement ships without a flag to turn it - off; there is deliberately no `--no-sandbox`, no `--force`, no - rebuild-the-world flag. -- **Nothing waits on a human.** No prompt, no interactive shell — - either is a hang for the agents that run these verbs most. -- **Refusals carry remedies, and remedies are verified.** A message - that tells someone to run a command has been run; a center's - spellings come from its documentation. -- **Docstrings are Google-style, comments carry *why*.** A design - decision gets a sentence; its history belongs in the design record, - not the code. - -## Conventions - -Ruff (E, F, I, N, W, UP; line length 100), mypy strict with -`namespace_packages = true`. `src/lightcone/` must never gain an -`__init__.py` — the namespace is shared with future sibling -distributions, and a real package there breaks the contract. diff --git a/docs/contributing/setup.md b/docs/contributing/setup.md deleted file mode 100644 index 188f1cc9..00000000 --- a/docs/contributing/setup.md +++ /dev/null @@ -1,78 +0,0 @@ -# Development Setup - -Everything runs through [uv](https://docs.astral.sh/uv/); there is no -task runner and no other build tooling. - -## Clone & install - -```bash -git clone https://github.com/LightconeResearch/lightcone-cli.git -cd lightcone-cli -uv sync --group dev -``` - -That resolves the engine and the dev tools (pytest, ruff, mypy, -datalad, the rocrate validator) into `.venv`. `uv run lc --version` -runs the checkout's `lc`. - -You also need `git` on `PATH` (the one tool uv cannot install); -git-annex arrives as a wheel with the sync. - -## The loop - -```bash -uv run pytest # the suite -uv run ruff check src/ tests/ # lint (--fix to apply) -uv run mypy src/ # strict mode -``` - -These three are exactly what CI runs (`tests.yml`, `lint.yml`) — green -locally means green there, modulo the gated suites below. - -Most of the suite is hermetic: an autouse fixture stubs the engine's -one subprocess seam, so tests spawn nothing and touch no network. The -exceptions opt in explicitly — see [Testing](testing.md). - -### The gated suites - -Three suites answer questions only a real mechanism can, and each -skips where its mechanism is missing — with an environment variable CI -sets to turn the skip into a hard failure: - -| Variable | Suite | Needs | -|---|---|---| -| `LC_SANDBOX_TESTS_REQUIRED=1` | `test_sandbox_enforcement.py` | Landlock (Linux) or Seatbelt (macOS) | -| `LC_CONTAINER_TESTS_REQUIRED=1` | `test_container_smoke.py` | podman or docker | -| `LC_CRATE_TESTS_REQUIRED=1` | `test_crate_smoke.py` | nothing beyond dev deps | - -## Building the docs - -```bash -uv sync --group docs -uv run zensical build # renders into site/ -uv run zensical serve # live preview -``` - -The site deploys on release (`docs-deploy.yml`), so docs track the -released CLI, not `main`. A pre-release deploys nothing — the site keeps -serving the last full release. - -## Building the wheel - -```bash -uv build -``` - -CI runs this only to publish. The version comes from hatch-vcs — the -git tag for a release, tag-plus-commit for a dev build — which is also -what lets a run record pin a dev engine by its source commit. - -## Pre-PR checklist - -1. `uv run pytest` — including, if your change touches the sandbox, - containers, or the crate, the relevant gated suite on a host that - can run it. -2. `uv run ruff check src/ tests/` and `uv run mypy src/`. -3. New behavior lands with its tests, in the same PR. -4. Read [Extending](extending.md) — it says where each kind of change - belongs, and the invariants it must keep. diff --git a/docs/contributing/testing.md b/docs/contributing/testing.md deleted file mode 100644 index 1e180c0d..00000000 --- a/docs/contributing/testing.md +++ /dev/null @@ -1,77 +0,0 @@ -# Testing - -The suite's shape follows the engine's: pure modules get pure tests, -the subprocess seam gets a stub, and the questions only a kernel, a -runtime, or a validator can answer get real ones — gated so they can't -pass by not running. - -## The one seam - -`tests/conftest.py`'s autouse `tools` fixture stubs -`engine.project._run` — the single choke point every external command -goes through — emulating each tool's observable effect (`uv lock` -writes `uv.lock`, `git init` makes `.git`, …) and recording every -argv. Under the stub the suite is hermetic: no network, no resolution, -no subprocesses. - -The `real_tools` fixture opts back out, putting the real `_run` back. -Everything built on it (the `analysis` fixture, the rerun tests) does -spawn and may touch the network — that is the deliberate price of -testing execution. - -## Where a question belongs - -| Question | File | Character | -|---|---|---| -| Convergence semantics | `test_project.py` | stubbed | -| Template content & repair | `test_templates.py` | pure | -| Do bytes land in the annex? | `test_dataset.py` | **real tools** — every bug this seam had was invisible to a stub | -| Identity sensitivity | `test_identity.py` | pure, both directions | -| The graph, the gate | `test_plan.py` | pure — tests what lc *adds*, never what a spec means (that's astra-tools' suite) | -| Classification | `test_assets.py` | pure | -| One output, real recipe | `test_worker.py` | real boundary, real repo | -| The run, the record | `test_materialize.py` | real repos; one real `LocalCluster`; real `datalad rerun` | -| Compute lifecycle | `test_compute*.py` | real detached local clusters; fake Slurm commands; real stock Dask bootstrap | -| Policy / wrap / denial | `test_sandbox_*.py` | pure, run on every OS | -| The kernel's answer | `test_sandbox_enforcement.py` | gated | -| Image identity | `test_image.py` | pure — structure and ordering, never byte goldens | -| Runtime lifecycle | `test_container.py` | stubbed; refusals asserted on recorded argv | -| The runtime's answer | `test_container_smoke.py` | gated | -| The crate | `test_crate.py` | pure; the one byte claim is render-twice-identical | -| The validator's answer | `test_crate_smoke.py` | gated | -| CLI surface | `test_cli.py` | `CliRunner`; assert short unwrappable fragments | - -## The enforcement suite - -`test_sandbox_enforcement.py` is the only file that can tell you the -sandbox works, and four properties keep it honest: - -1. **One suite, both mechanisms** — parameterized by `detect()` alone; - a leak only Linux catches is a leak, and macOS CI is the sole place - the generated SBPL ever executes. -2. **The real policy** — always `exec_policy`, never one hand-built to - make the point. (`/usr` once sat in the exec set through a fully - green suite built the other way.) -3. **Real leaks, tried literally** — undeclared tools executed, - undeclared libraries `dlopen`ed, undeclared data read. -4. **It cannot pass by not running** — `LC_SANDBOX_TESTS_REQUIRED=1` - in CI turns the skip into a failure, and two tests cover the guard - itself. - -**Mutation-check every denial test**: run the same command through -`Unavailable()` and confirm it *succeeds*. A denial test that would -pass unsandboxed is testing nothing, and the failure mode is silent. -Two related traps: a write-denial must target a path the OS would let -you write (a `/etc` write pins nothing), and enforcement fixtures must -not live under `/tmp`, which is inside the write baseline — the -`outside` fixture roots at `$HOME` for exactly this reason. - -## Conventions - -- Don't add a flag whose only user is a test — stub `project._run` - instead. -- A forged-output test must break the annex hard link before writing - (`test_materialize._forge` shows how) — results are committed thin, - so an in-place write dirties every byte-identical sibling. -- Record formats are tested through their consumer (datalad's parser, - the rocrate validator), never as golden files of our own JSON. diff --git a/docs/index.md b/docs/index.md deleted file mode 100644 index 31464072..00000000 --- a/docs/index.md +++ /dev/null @@ -1,61 +0,0 @@ -# lightcone-cli - -**lightcone-cli** is [Lightcone Research][lr]'s execution layer for -[**ASTRA**][astra] (Agentic Schema for Transparent Research Analysis). -It serves as the machinery that ties an analysis `astra.yaml` specification to a tree -of materialized outputs. - -!!! warning "Alpha development" - lightcone-cli is in **early alpha**. The CLI and the execution layer are - still moving — expect breaking changes between minor versions. Bug reports, design - challenges, and use cases the tooling doesn't yet cover are exactly what we want to - hear at this stage; please open an issue on the - [GitHub repo](https://github.com/LightconeResearch/lightcone-cli/issues). - -## Choose your path to the documentation - -
- -- __I want to try it out__ – :lucide-rocket: - - --- - - Installation instructions, step-by-step tutorial, and fast tour of the lightcone framework and its workflow capabilities. - - [User Guide](user/index.md){ .md-button .md-button--primary } - -- __I want to contribute__ – :lucide-cog: - - --- - - In depth tour of the software architecture and API docs, as well as contribution instructions, aimed for - contributors and maintainers. - - [Developer corner](maintainer.md){ .md-button .md-button--primary } - -
- ---- - -## Two libraries, one toolchain - -
- -- __lightcone-cli__ - - The library that ships the `lc` CLI: project scaffolding, locked environments, sandboxed execution, and the provenance layer. Depends on [**astra-tools**][astra-tools], the SDK for working with ASTRA analysis specifications. - - [:fontawesome-brands-github: Repository][cli]{ .md-button } - -- __astra-tools__ - - The SDK for working with [**ASTRA**][astra] analysis specifications. This library provides the `astra` CLI which handles the [**ASTRA**][astra] lifecycle and validation process (schema, prior insights & findings, evidence verification helpers). - - [:fontawesome-brands-github: Repository][astra-tools]{ .md-button } - -
- -[lr]: https://lightconeresearch.org/ -[astra]: https://astra-spec.org/latest/ -[astra-tools]: https://github.com/LightconeResearch/astra-tools -[cli]: https://github.com/LightconeResearch/lightcone-cli diff --git a/docs/maintainer.md b/docs/maintainer.md deleted file mode 100644 index fdfac4a5..00000000 --- a/docs/maintainer.md +++ /dev/null @@ -1,58 +0,0 @@ -# Developer corner - -`lightcone-cli` is a small engine with strong opinions: one way to -identify an output, one way to store it, one boundary to execute it -behind. This guide covers everything below the user surface — how the -engine is put together, what each module owns, and how to get a -working dev loop. - -If you're looking for the user-facing docs, the -[user guide](user/index.md) is the other half of this site. - -## What this covers - -- [Architecture](architecture.md) — the CLI/engine/ASTRA split, the - run pipeline, identity, storage, the exec boundary, and the - invariants that hold them together. -- [CLI Reference](cli/index.md) — every `lc` command: flags, JSON - report shapes, exit codes. -- [Engine Internals](api/index.md) — the `lightcone.engine.*` - modules: what each owns, its key symbols, and what must stay true - of it. -- [Contributing](contributing/setup.md) — clone, install, run the - test suite; [how the suite is shaped](contributing/testing.md); and - [where a change belongs](contributing/extending.md). - -## Get started in three commands - -!!! tip "Dev loop" - - ```bash - git clone https://github.com/LightconeResearch/lightcone-cli.git - cd lightcone-cli - uv sync --group dev && uv run pytest - ``` - -Test, lint (`uv run ruff check src/ tests/`) and type-check -(`uv run mypy src/`) are the whole loop — there is deliberately no -task runner in between. - -## The house rules - -A few conventions run through every module; changes are reviewed -against them: - -- **No dead code, no foreshadowing.** Nothing lands before the layer - that calls it, and no message names a verb or flag that doesn't - exist yet. `lc --help` advertises only what works. -- **No escape hatches around guarantees.** A feature that enforces - something ships without a flag to turn the enforcement off. -- **Literal behavior over invented convenience.** The current - directory is the project; erroring beats walking up or guessing. - Nothing prompts — a verb is run by an agent more often than a - person, and a prompt is a hang. -- **One implementation per rule.** Classification, path naming, the - run-record subject, tool resolution — each has exactly one spelling, - and a second copy is where the two start to disagree. -- **Honest reporting.** What was enforced, what was skipped, and what - a clone can't see are all recorded or said — never assumed. diff --git a/docs/stylesheets/extra.css b/docs/stylesheets/extra.css deleted file mode 100644 index e454c122..00000000 --- a/docs/stylesheets/extra.css +++ /dev/null @@ -1,57 +0,0 @@ -@import url('https://fonts.googleapis.com/css2?family=EB+Garamond:ital,wght@0,400..800;1,400..800&family=Libre+Baskerville:ital,wght@0,400;0,700;1,400&family=Inter:wght@300;400;500;600&family=JetBrains+Mono:wght@400;500&display=swap'); - -/* ── Light mode ─────────────────────────────────────────────────────────── */ - -:root > * { - --md-primary-fg-color: #4e5a70; - --md-primary-fg-color--light: #6b7a8d; - --md-primary-fg-color--dark: #3a4456; - --md-accent-fg-color: #426b78; - --md-default-bg-color: #f8f7f3; - --md-default-bg-color--light: #ffffff; - --md-default-bg-color--dark: #f1efe9; - --md-text-font: "Libre Baskerville", Georgia, serif; - --md-code-font: "JetBrains Mono", "Fira Code", monospace; -} - -body, -.md-header, -.md-main, -.md-main__inner, -.md-content, -.md-tabs, -.md-sidebar { - background-color: #f8f7f3; -} - -/* ── Dark mode ──────────────────────────────────────────────────────────── */ - -[data-md-color-scheme="slate"] { - --md-default-bg-color: #221f20; - --md-hue: 219; -} - -[data-md-color-scheme="slate"] body, -[data-md-color-scheme="slate"] .md-header, -[data-md-color-scheme="slate"] .md-main, -[data-md-color-scheme="slate"] .md-main__inner, -[data-md-color-scheme="slate"] .md-content, -[data-md-color-scheme="slate"] .md-tabs, -[data-md-color-scheme="slate"] .md-sidebar { - background-color: #221f20; -} - -[data-md-color-scheme="slate"] .md-nav__link--active, -[data-md-color-scheme="slate"] .md-nav__item--active > .md-nav__link { - background-color: rgba(106, 147, 160, 0.20); - color: #f8f7f3; -} - -[data-md-color-scheme="slate"] .md-content a { - color: #85c0d0; -} - -[data-md-color-scheme="slate"] .md-content :not(pre) > code { - background-color: rgba(76, 63, 70, 0.15); - color: #c8dde5; -} diff --git a/docs/user/cluster.md b/docs/user/cluster.md deleted file mode 100644 index 1856b3a1..00000000 --- a/docs/user/cluster.md +++ /dev/null @@ -1,584 +0,0 @@ -# Running on a Cluster - -Allocate compute explicitly, then pass the returned cluster name to either execution -command. The same commands work for a local workstation and Slurm. No cluster is -started by `lc run` or `lc materialize`, even when Slurm environment variables are -present. `lc materialize --check` and `lc status` remain local project inspection. - -## Start locally - -No configuration is needed on a fresh installation. `lc compute launch` uses all -detected usable logical CPUs and RAM on this machine and names the cluster `local`. -A local cluster has no fixed lifetime: it ends after 30 minutes without task -activity, so a two-hour recipe finishes normally and the cluster stops 30 minutes -later if no further work arrives. Running or queued tasks keep it alive; connecting -a client or checking its status does not. Use `--time` to add a hard lifetime, which -ends the cluster even while work is running, or `lc compute down` to stop it now. -GPUs require explicit offers; see [GPU allocations](#gpu-allocations). - -```bash -lc compute resources -lc compute launch --dry-run -CLUSTER=$(lc compute launch --wait) -lc run "$CLUSTER" -- python -c 'print("hello from the cluster")' -lc materialize "$CLUSTER" -lc compute down "$CLUSTER" -``` - -Run the execution commands from your project root. A launch returns when native -allocation is accepted; `launch --wait` or `status --wait` waits for Dask readiness. -Both accept `--timeout SECONDS` (default 300). Waiting failures retain the accepted -cluster ID and leave the allocation unchanged. Execution never -waits: `lc run` and `lc materialize` refuse a cluster that is not active with -every expected worker connected, for example: - -```text -Error: this allocation's Dask scheduler has not started yet; wait for readiness with `lc compute status CLUSTER --wait` -``` - -Finishing a run detaches its client and leaves the cluster available for another -command. The allocation ends at its time limit or when you call `down`. - -`lc compute status` lists allocations as `name: status`, one per line. -Use `lc compute status NAME` for resource details and Dask readiness. - -## Local allocations - -Local resources are cooperative limits, not an exclusive CPU/RAM reservation. -Only one local cluster can run per user on each machine. A launch checks the -process table for a running local cluster of yours and refuses if it finds one, -including one launched through a different name, catalog, or connection root. -End the existing cluster before launching another; once its -owner process exits, including on failure, idle expiry, or walltime expiry, a -new launch proceeds. A refusal identifies the running cluster and its original catalog and -connection root. Use that catalog to inspect or stop the cluster if the current -catalog sets a different `connection_root`. If the cluster's record is missing or -damaged, the refusal names its process ID instead, to stop with `kill`. -Launches that overlap can both succeed, and the check sees only the processes -visible where `lc` runs: a launch inside a container does not see a cluster -started outside it. -An allocation owns a detached process session and standard `LocalCluster`: one -worker process with `task_slots_per_node` threads, and a scheduler that listens -on `127.0.0.1` over TLS. Its own logs are discarded; a startup failure, or the -scheduler closing after its idle timeout, is kept and shown as the reason by -`lc compute status`. At its walltime the whole process session is killed with -SIGKILL, so a recipe still running stops mid-write. The idle timeout ends the -session the same way. Dask tracks tasks, not processes: after an interrupted -`lc run` or `lc materialize`, a recipe can keep running once its task is gone, -and the idle timeout stops it too. -`down` sends SIGTERM, waits three seconds, then sends SIGKILL. -Private process locators are checked against the native boot UUID, UID, process -session, and exact command containing the allocation's random token before -attachment or termination. Hostname changes and clock adjustments do not change -that identity. Manage a local allocation from the host and boot session that -launched it. Other boot sessions are excluded from discovery, and an explicit -ID from one is refused rather than reported as stopped. Once an allocation has -ended, its credentials and scratch directory are removed; its full ID still -reports `ended`. -Local compute requires an enabled local policy and a valid local offer. -On NERSC login nodes it is disabled automatically, even with no catalog or with -`local.enabled: true`. The guard recognizes a nonempty `NERSC_HOST` and a short -hostname matching `login[0-9]+`; it does not perform DNS or scheduler queries. -The guard permits compute nodes such as `nid200021`, including interactive -sessions. A `SLURM_JOB_ID` variable does not exempt a login node. -See NERSC's [environment conventions](https://docs.nersc.gov/environment/) and -[interactive sessions](https://docs.nersc.gov/connect/vscode/). - -A local offer's optional `config` settings are `scratch_root` (default: the -temporary directory), `python` (default: the interpreter running `lc`), and -`task_slots_per_node` (default: all of the offer's CPUs). - -## Cluster names - -The local shortcut defaults to `local`. Explicit CPU/memory requests generate a -short name such as `lc-a1b2c3d4e5f6`. Override either with `--name`: - -```bash -lc compute launch --name analysis --wait -lc compute down analysis -``` - -Names contain 1–63 lowercase ASCII letters, digits, or hyphens, starting with a -letter and ending with a letter or digit. Launch writes only the name to stdout; -readiness guidance goes to stderr. All execution and lifecycle commands accept -either that name or the full immutable ID available in launch and status JSON. - -Names are checked against current allocations across every provider the catalog's -offers use, and always local, before launch. An explicit duplicate is refused, and an autogenerated collision -is regenerated before submission. Discovery failures prevent this check from -succeeding. Concurrent launches can still choose the same name, so lookup also -refuses ambiguous names or incomplete discovery. Use a full ID to select a known -allocation directly when another provider cannot be queried. - -This includes local allocations when `local.enabled: false`: they remain visible -and can still have conflicting names. Repair a provider's discovery error before -launching another cluster or resolving names. - -A name may be reused once its allocation has ended. Keep the full ID when you -need a durable reference to one allocation; a later cluster with the same name -has a different ID. Names are discovered from allocation metadata, without a -separate name registry. Native state still decides whether an allocation exists. - -## Customize resource offers - -Create `~/.lightcone/compute.yaml`, or select a file with `LC_COMPUTE_CONFIG`. -For a smaller default local budget: - -```yaml -version: 1 -local: - resources: {cpus: 4, memory: 8GiB} -``` - -Both CPU and memory are required in `local.resources`; omit that block to use -detected capacity. The same capacity validation applies to configured budgets. -This controls the default offer, not hard OS resource limits. - -`local.time` replaces the default offer's time limits, for example a longer idle -timeout with a ceiling on `--time`: - -```yaml -version: 1 -local: - time: {idle: 1h, max: 8h} -``` - -NERSC login nodes are guarded without setup. For other sites, or to disable local -compute on every node using the catalog, set: - -```yaml -version: 1 -local: - enabled: false -# Add Slurm offers as shown below. -``` - -This blocks local launches, including explicit local offers, and new execution -commands on local clusters. Existing local allocations remain inspectable and -stoppable; disabling does not kill them. Bare launch reports that local compute is -disabled; supply CPU/memory requirements to select a Slurm allocation. Select this -catalog on login nodes through `LC_COMPUTE_CONFIG`. This is Lightcone configuration -policy; native site permissions enforce machine-wide restrictions. -Leave `local.enabled` at its default to use local compute inside a NERSC -interactive compute-node session. The login-node guard still applies, and it -creates no configuration file. - -By default, a built-in `local` offer follows configured offers, which take -selection priority. If the catalog lists its own local offers, those replace the -built-in offer; omit `local.resources` and `local.time`, and size and time those -offers directly. The no-resource shortcut selects the first eligible local offer -and defaults its cluster name to `local`. - -Resource requests can select the built-in local offer when no earlier remote -offer is eligible. Set `local.enabled: false` for catalogs that must use only remote -compute. While the built-in offer is enabled, the offer name `local` is reserved. - -For example, this catalog defines its own local offer: - -```yaml -version: 1 -offers: - - name: workstation - provider: local - resources: {cpus: 4, memory: 8} - max_nodes: 1 - time: {idle: 30m} - startup: fast -``` - -Allocations launched from the built-in offer stay visible and can still be -stopped after this file exists: both are the same local provider. - -Set `LC_COMPUTE_CONFIG` to choose another file for all commands, including -`lc run` and `lc materialize`, which find clusters through the same catalog. A -missing explicit path or an invalid catalog is an error. An absent implicit default -file uses the default local policy. - -A catalog has `version: 1`, an optional `connection_root`, an optional `local` -policy, and an ordered `offers` list, empty by default: - -- `connection_root` (default `~/.lightcone/compute`) holds every allocation's - private scheduler files and credentials, one directory per provider. Stop - running allocations before changing it: allocations under the old root are no - longer found. -- An offer has a unique `name`, its `provider` (`local` or `slurm`), per-node `resources` - (`cpus`, `memory`, and optional `accelerators`), `max_nodes`, and `time`. `time` - holds an optional walltime `default`, no longer than an optional `max`, and an - optional `idle` timeout; it needs a `default` or an `idle`, so every allocation - can end. Local offers end at whichever limit comes first. Slurm offers need a - `default` and refuse `idle`. `startup` is optional (`fast`, `batch`, or the default - `unknown`), written either as a bare class or as `{class: …, source: …}`. - `config` holds provider-specific settings. - -Each provider reaches one native authority: `local` is this machine, and `slurm` -is the cluster the Slurm client on this machine reaches by default. - -Catalog errors identify the invalid field, for example `offers.0.resources.cpus`. -Unknown common fields and duplicate YAML keys are rejected. CPU and node counts -must be positive integers. Compute memory follows SkyPilot's binary-unit -convention: `32`, `32GB`, and `32GiB` mean 32 GiB. Fractional quantities must -represent an exact number of bytes. An accelerator declaration names one type -and a positive whole count: `accelerators: A100:4` or `accelerators: {A100: 4}`; -`accelerators: A100` means one. Omit it for CPU-only offers. Durations use ordered -day/hour/minute/second units, such as -`30m`, `1h30m`, or `45s`. - -Selection takes the first offer in catalog order that matches the request. An -offer this host cannot provide is skipped: a local offer with more nodes, CPUs or -memory than the host has. When nothing matches, the error lists why each skipped -offer was unavailable: - -```text -Error: no configured offer matches this resource request; see lc compute resources; huge: the local offer exceeds this host's CPU or RAM capacity -``` - -## Configure Slurm - -The CLI runs the native `sbatch`, `salloc`, `squeue`, `sacct`, `scontrol`, and -`scancel` commands as the current user. It needs a compatible Slurm client -installation, and it uses the cluster that installation reaches by default. - -Those commands run without inherited request settings: every `SBATCH_*`, -`SALLOC_*`, `SRUN_*`, `SQUEUE_*`, `SACCT_*`, `SCANCEL_*`, and `SLURM_*` variable -is removed, except `SLURM_CONF`, `SLURM_CONF_SERVER`, and `SLURM_JWT`. An -`SBATCH_ACCOUNT` in your shell profile therefore has no effect; put the account -in the offer. - -This illustrative NERSC configuration requires a deployment-specific account and -resource sizing. It has not been validated by submitting a job at NERSC: - -```yaml -version: 1 -local: - enabled: false -offers: - - name: quick - provider: slurm - resources: {cpus: 256, memory: 480} - max_nodes: 2 - time: {default: 1h, max: 4h} - startup: fast - config: - submit: salloc - account: myproject - constraint: cpu - qos: interactive - - name: batch - provider: slurm - resources: {cpus: 256, memory: 480} - max_nodes: 16 - time: {default: 1h, max: 12h} - startup: batch - config: - submit: sbatch - account: myproject - constraint: cpu - qos: regular -``` - -A Slurm offer's `config` accepts `submit` (`sbatch`, the default, or `salloc`), -`account`, `partition`, `qos`, `constraint`, `reservation`, `gpu_type`, and the -launch settings below. -For a named accelerator offer, `gpu_type` maps the public type to the site's -native Slurm GRES name; it is required even when the spellings happen to match. -A generic `GPU` offer may omit it. Slurm offers must state memory as a whole -number of MiB. - -Every launch setting is optional. The defaults assume a home directory that the -login and compute nodes share: - -- `python`: the interpreter that ran `lc compute launch`, so workers use the - driver's own Lightcone installation. `uv tool install lightcone-cli` places it - under `$HOME`. Avoid launching through `uvx`, whose environment lives in uv's - cache and can be pruned while the allocation runs. -- `scratch_root`: each node's own temporary directory (`$TMPDIR`, usually - `/tmp`), which holds the Dask workers' files. -- `cwd`: your home directory, as the job's working directory. -- `task_slots_per_node`: one fewer than the offer's CPUs, leaving room for the - scheduler. Lower it when recipes are multithreaded or memory-heavy. -- `interface`: unset, so Dask listens on the node's hostname. Name a network - interface instead if nodes cannot reach each other by hostname; Perlmutter's - high-speed network is `hsn0`. - -The catalog's top-level `connection_root` (default `~/.lightcone/compute`) must -be reachable from the driver and every node too: the scheduler's connection -files, TLS credentials, and batch logs live in private directories there. - -The offered CPU and memory shape is per node. Bare resource quantities request an -exact match; a trailing `+` permits a larger offered shape. Selection takes the -first eligible offer in catalog order. `--startup fast` filters to that service -class; it does not guarantee a queue wait. Inspect the resolved plan before launch: - -```bash -lc compute launch --cpus 32+ --memory 128+ --num-nodes 2 --time 1h --dry-run -``` - -One allocation contains one `srun` step with one rank per node, bound with -`--cpu-bind=threads` to exactly the hardware threads Slurm allocated. Each rank -starts a standard Dask `Nanny`, which supervises a separate `Worker` process. -Rank zero also runs the scheduler, so a worker process exiting does not take the -scheduler with it. A one-node allocation has both scheduler and worker. -Neither serves a dashboard or any other HTTP route. -The scheduler consumes part of the offered resources; `task_slots_per_node` -controls Dask task concurrency independently of the allocation's logical CPUs. - -Dask's Nanny defaults `OMP_NUM_THREADS`, `MKL_NUM_THREADS`, and -`OPENBLAS_NUM_THREADS` to `1`, preventing each concurrent recipe from requesting -the whole node's numerical-library threads. Values explicitly set in the launch -environment take precedence and also reach containerized recipes. For recipes -that need more threads, set those values before launching and declare enough -recipe CPUs for each task; `task_slots_per_node` can further cap concurrency. See -[Dask's Nanny environment settings](https://distributed.dask.org/en/stable/worker.html#nanny). - -The Nanny restarts an exited worker, and Dask can reschedule its tasks. The step -uses `--kill-on-bad-exit=0 --wait=0` so an exited rank does not itself trigger -termination of the remaining ranks. This does not provide complete failure -isolation: site OOM policy can still kill the step or allocation, and the -scheduler and Nanny processes are not restarted if they die. Dask memory -management remains disabled because it does not account for the recipe -subprocesses' memory. Reduce task concurrency for memory-heavy recipes; there is -no per-recipe memory limit. A lost worker can also leave a recipe subprocess -running while Dask reschedules its task, so retries do not guarantee exclusive -access to output files. New commands still require every expected worker to be -connected. See [Dask's failure behavior](https://distributed.dask.org/en/stable/resilience.html) -and [Slurm's step termination settings](https://slurm.schedmd.com/srun.html). - -Lightcone always requests a finite native `--time`. Actual termination follows Slurm's -`OverTimeLimit` and `KillWait` policy, which can permit an unlimited overrun. -Lightcone does not impose an independent Slurm runtime deadline or require a -preflight time-policy query. - -`config.partition` is optional. Lightcone passes `--partition` only when it is -explicitly configured; otherwise the site selects the partition. Omit it at -NERSC so site routing can select from the QOS and constraint. During deployment -testing, inspect the submitted job's actual `Partition` with `scontrol show job`. -[NERSC's workflow guidance](https://docs.nersc.gov/jobs/workflow/maestro/) -describes its QOS-driven partition selection. - -At NERSC, move uv's cache off `$HOME` before launching. Every recipe and probe -runs through `uv run`, which locks uv's cache, and Perlmutter's compute nodes -cannot lock files in `$HOME`, where the cache lives by default. Workers inherit -the environment `lc compute launch` runs in, so set the variable there, for -example in your shell profile: - -```bash -export UV_CACHE_DIR=$PSCRATCH/uv-cache -``` - -An allocation launched without it has to be relaunched. `$PSCRATCH` is purged -when idle; a purged cache is only downloaded again. - -Slurm displays `lc-v1-` as the job name, for example `lc-v1-analysis`. -Its native comment carries the random submission token as -`lightcone:v1:kind=dask:token=<32hex>`. Lightcone verifies the name, token, and owner -before attaching or cancelling; a job name alone does not establish identity. -The opaque cluster ID encodes the provider, native job ID, token, and name. -There is no job registry to reconcile. Removing an offer prevents new launches; -`status` and `down` with a full cluster ID still reach the job, while name lookup -and the `status` listing query only the providers the catalog's offers use. -Native job state and live Dask readiness are separate observations. A worker loss -can leave a job active but not ready. Unknown native state is reported as unknown. - -Live discovery requires the native comment. Historical inspection also needs -Slurm accounting to retain it through `AccountingStoreFlags=job_comment`. -If accounting has no matching token, Lightcone reports unknown rather than -assuming the allocation ended or cancelling a job with a reused ID. -[Slurm documents this comment-retention setting](https://slurm.schedmd.com/sacct.html). - -An `salloc` launch retains native `salloc`/`srun` processes on the submit host. -Its survival across logout, Jupyter shutdown, and site session cleanup must be -checked on the deployment. Batch jobs are independent of the submitting CLI. -An ambiguous submission reports its token; inspect native state before retrying, -since the original allocation may have been accepted. - -A batch launch submits with `sbatch --parsable --no-requeue`. An `salloc` launch -starts `salloc --kill-command=TERM` detached and returns once Slurm lists the job, -waiting at most ten seconds. Everything lives under the connection root: - -- Submission logs: `submissions//.out` for `sbatch`, or - `submissions//salloc.log`. -- Scheduler connection files and TLS credentials: - `slurm/-/attempt-/`. - -Each worker's files go under `//attempt-/`. - -Before starting Dask, every rank checks that Slurm gave it what the plan -requested: the node count, CPUs per task and its actual CPU affinity, and memory -per node. GPU allocations also validate Slurm's native GPU count. -Ranks other than zero wait up to 120 seconds for the scheduler, which -has as long to start. A failed check or timeout logs -`Slurm Dask startup failed: …` to the submission log and exits nonzero. Look -there when a job is active but never becomes ready. - -## GPU allocations - -GPU support uses NVIDIA CUDA devices on Linux. `lc compute resources` shows -configured accelerator types and counts; Lightcone does not probe CUDA or discover -local hardware. There is no fractional GPU or MIG management. - -Requests use SkyPilot-style `NAME[:COUNT]`: `--gpus A100` means one A100, -`--gpus A100:4` means exactly four, and `--gpus GPU:4` accepts any configured model -with exactly four. Names are case-insensitive catalog labels; GPU counts do not -accept `+`. Omitting `--gpus`, or passing `0`, selects CPU-only offers. - -For local GPUs, add an offer to the [workstation catalog above](#customize-resource-offers): - -```yaml - - name: workstation-gpu - provider: local - resources: {cpus: 4, memory: 8GB, accelerators: 'GPU:1'} - max_nodes: 1 - time: {idle: 30m} - startup: fast -``` - -Set the devices available to that allocation when launching it: - -```bash -CUDA_VISIBLE_DEVICES=0 lc compute launch --cpus 4 --memory 8GB --gpus GPU:1 -``` - -Lightcone freezes the nonempty mask and `CUDA_DEVICE_ORDER`, if set, at launch. -You are responsible for matching the catalog's count and model to those devices; -Lightcone does not verify them. Local allocations do not reserve GPUs exclusively -against other allocations or programs on the host. Before launch, the host must -have loaded the NVIDIA driver and created its character devices, including UVM. -Lightcone grants existing device nodes and does not initialize them; see -[NVIDIA's device setup utility](https://github.com/NVIDIA/nvidia-modprobe/blob/main/nvidia-modprobe.1.m4). - -On Slurm, the offer's `config.gpu_type` maps its catalog label to a native GRES -type. For example, add this offer with settings adjusted to your site: - -```yaml -- name: gpu-batch - provider: slurm - resources: {cpus: 32, memory: 128GB, accelerators: 'A100:4'} - max_nodes: 2 - time: {default: 1h, max: 4h} - startup: batch - config: - account: myproject - constraint: gpu - gpu_type: a100 -``` - -```bash -lc compute launch --cpus 32 --memory 128GB --gpus A100:4 --dry-run -``` - -This is an illustrative shape, not a tested site configuration. Slurm allocates -GPUs through GRES; Lightcone checks the native count and passes Slurm's CUDA mask -through unchanged, using `CUDA_DEVICE_ORDER=PCI_BUS_ID`. - -GPU commands inherit the allocation's whole CUDA mask. CPU commands receive an -empty mask. These are cooperative visibility settings; native OS and cgroup -permissions remain authoritative. - -Containerized GPU execution currently supports **podman-hpc** through its native -`--gpu` option. See [NERSC's GPU container guidance](https://docs.nersc.gov/development/containers/podman-hpc/overview/#using-nvidia-gpus-in-podman-hpc). -Recipes explicitly requesting GPUs with ordinary Docker or Podman are refused -before image preparation. `lc run` probes on those runtimes remain usable on a -GPU cluster: they run without GPUs and report that limitation. CPU execution -supports all three runtimes and sets `NVIDIA_VISIBLE_DEVICES=void` to override -GPU-enabled image defaults. Physical GPU execution remains a deployment -validation step. - -A standalone GPU rerun needs a device mask in its own environment, for example -`CUDA_VISIBLE_DEVICES=0 datalad rerun`. It does not inherit an old allocation's -mask or reserve devices through Dask. - -## Recipe resource requirements - -Declare each recipe's needs in `astra.yaml`: - -```yaml -recipe: - command: python src/fit.py {output} - resources: - cpus: 4 - memory: 8Gi - gpus: 1 -``` - -Each recipe runs on one worker. Its CPU, memory, and GPU request must fit that -worker, even when the cluster has several nodes. Dask reserves CPU and memory -while the task runs, so recipes can run together only when their combined -requests fit. `task_slots_per_node` also caps concurrent tasks; it does not -limit how many CPUs a single recipe may request. - -CPUs must be positive whole numbers and default to one. Memory needs units: -`512Mi` and `8Gi` are binary sizes; `8GB` is decimal, unlike compute memory. -Bare quantities are not accepted. Without a memory declaration, no RAM is -reserved: CPU requests and `task_slots_per_node` control concurrency. - -Recipe `gpus` is a nonnegative whole count, defaulting to zero; accelerator type -selection belongs to cluster allocation. A GPU recipe reserves the worker's -entire GPU budget, so only one GPU recipe runs on that worker at a time. The -requested count is a minimum capacity requirement: the command inherits the -worker's whole allocated CUDA mask and may see more GPUs than requested. CPU -recipes may still run alongside it when CPU, memory, and task slots permit; their -CUDA mask is empty. `lc run` reserves the worker's entire CPU, memory, and GPU -budgets; direct and podman-hpc probes inherit that allocation mask. - -Recipe `time_limit` is not supported and is refused before preparation or -execution. Bound the allocation instead, with `lc compute launch --time`. -Fractional CPU/GPU counts, GPU model requests inside a recipe, and disk requests -are also rejected rather than ignored. - -`lc materialize` first classifies the selected graph, then checks resources for -outputs that may rebuild before preparation or submission. Already-current or -unrefreshed behind outputs reserve nothing. Dependents of an output that may -change still need resources; the worker may later skip them if the actual -upstream digest is unchanged. Use `lc materialize --check` to inspect currency -without allocation. Read-only `status` and `--check` accept valid ASTRA resource -declarations even when this executor cannot satisfy them. - -These are scheduling reservations, not per-recipe CPU or RAM enforcement. -Recipes must respect their declarations; a subprocess can otherwise exceed -its request. Slurm enforces the overall allocation, while local execution -uses cooperative budgets. Leave capacity for the scheduler, workers, and other -overhead when declaring recipe requirements. - -A CPU reservation does not set numerical-library thread counts. Local and Slurm -clusters use Dask's Nanny defaults of `1` for `OMP_NUM_THREADS`, `MKL_NUM_THREADS`, -and `OPENBLAS_NUM_THREADS` when those variables are unset. Container recipes -receive the worker's effective values too. Set the variables before -`lc compute launch`, or in the recipe command, to choose another value, and -declare enough recipe CPUs for those threads. See -[Dask's defaults](https://docs.dask.org/en/stable/configuration.html#distributed.nanny.pre-spawn-environ.OMP_NUM_THREADS) -and [environment precedence](https://distributed.dask.org/en/stable/_modules/distributed/nanny.html). - -## Execution requirements and limits - -Driver and workers must see the same project, prepared environment, and inputs -at the same absolute paths. They need matching Lightcone code, Python major/minor, -and Dask versions. By default, Slurm workers run the driver's own installation -(see the `python` launch setting above). Relaunch allocations after upgrading -the worker installation. -Commands and recipes are ordinary tasks submitted to the Dask -scheduler, which chooses their workers; task runtime and sandbox checks still -apply. Recipe output is forwarded to the invoking terminal on stderr; `run` -preserves the command's stdout and stderr bytes separately. -Containerized projects also require the prepared image and runtime on each -worker; `podman-hpc` can expose its migrated image across NERSC nodes. -Direct recipes inherit the allocation workers' environment, not variables added -to the invoking CLI after launch. Remote execution does not forward stdin. - -The catalog contains policy, not credentials or live state. Scheduler connection -material is private and uses standard Dask TLS and scheduler files. The configured -`connection_root` and scratch roots can contain symlinks, including a symlinked home -directory: Lightcone resolves the root before appending managed paths. Allocation -directories and credential files still reject symlinks, retain ownership and -ancestor-permission checks, and require modes `0700` and `0600`, respectively. -The catalog's location is independent of the private connection files. - -Use one execution invocation per project at a time. Concurrent writers, -comprehensive cancellation, task fencing, and recovery after client/worker loss -are not guaranteed. A lost client does not prove its subprocesses stopped. -Unreported partial outputs are retained after interruption rather than restored -while a task may still write them. End the allocation and establish that work has -stopped before inspecting or repairing that project's outputs. -For local containerized execution, `down` and walltime expiry stop the managed -process group but do not guarantee termination of containers managed by an -external runtime. A Podman container that ignores SIGTERM can survive. Inspect -and stop such containers through the container runtime before cleaning results. diff --git a/docs/user/concepts.md b/docs/user/concepts.md deleted file mode 100644 index 8dda4ae1..00000000 --- a/docs/user/concepts.md +++ /dev/null @@ -1,151 +0,0 @@ -# Core Concepts - -The mental model behind `lc`, in one page. Nothing here is required to -follow [Getting Started](getting-started.md) — come back when you want -to know *why* the tool behaves the way it does. - -## A project is three files - -A lightcone project is a directory holding an ASTRA spec and a uv -project: - -- **`astra.yaml`** describes the analysis — inputs, outputs, recipes, - methodological decisions. It is the single source of truth: everything - `lc` does is downstream of it. -- **`pyproject.toml` + `uv.lock`** describe the environment — every - package a recipe may import, resolved to exact versions. The `.venv` - is built *from* the lock and is disposable; the lock is what's real, - and it travels in git. - -There is no global configuration, no registry, no state outside the -project. Clone the repository and you have everything except two pieces -of local machinery (`.venv` and the git-annex initialization), which -`lc init` rebuilds. - -Adding a dependency is a uv operation, not an `lc` one: - -```bash -uv add numpy -``` - -That updates `pyproject.toml`, re-locks, and syncs `.venv` in one step. -Recipes import from the locked environment and nothing else — a stray -`pip install` on your machine changes nothing they can see. - -## An output has an identity, and three facts about it - -Every materialized output records, in its -`..manifest.json`: - -1. **What it is** — a hash of its recipe and the decision values that - shaped it (its *definition*). -2. **What it was made from** — a content hash of each declared input. -3. **What it ran under** — a hash of the environment (the lock, the - interpreter, the image declaration if any), plus the git commit the - run started at. - -Those three facts are deliberately not one fact, because they age -differently — and that is what the three states mean: - -| state | means | what `lc materialize` does | -|---|---|---| -| `current` | the output is exactly what the spec asks for, made from these inputs, under this environment | nothing | -| `stale` | the output **contradicts** the project: the spec now defines it differently, or an input's content changed | remakes it | -| `behind` | the output is still exactly what the spec asks for — only the **environment** moved since it was made | reports it, leaves it alone | - -The line between `stale` and `behind` is contradiction versus -circumstance. A stale output is mislabelled — keeping it would be a -lie, so it is remade. A behind output is not wrong in any way: one -`uv add` for a plotting script rewrites the lock for the whole project, -and remaking a week of computation over that buys nothing. Its manifest -records exactly which environment and commit produced it, and that -commit's own `uv.lock` reconstructs the environment if you ever need -it. - -When you *do* want behind outputs remade — before a release, say — -that is one flag: - -```bash -lc materialize "$CLUSTER" --refresh -``` - -`--refresh` only ever widens a run: a `current` output stays current -under it, and there is deliberately no flag in the other direction — -nothing suppresses the rebuild of a stale output. - -One more way an output can be stale: a hand edit. Every output is -committed by the run that made it, so a file changed by hand and -committed shows up in history under a commit that is not a run record — -and the output classifies `stale` everywhere, with `lc status` naming -the foreign commit. - -## Everything is committed, and the tree stays clean - -`lc` versions results in the project's own git repository: git carries -the history and the small files, git-annex carries the data bytes — -transparently, behind the ordinary `git add` / `git commit` you already -type. - -That model has two consequences you'll feel: - -- **A run starts from a clean tree.** Every output is committed - together with the code that produced it; a run that started from - uncommitted edits could not say what that code was. So: commit, then - materialize. -- **A run ends with a clean tree.** Each output is committed as it - lands — with its manifest, in a commit whose message is a *run - record* that `datalad rerun` can replay. A failed recipe's partial - work is rolled back. Your `git log` is the build log. - -`results/` is `lc`'s to write. Don't put files there by hand — a -hand-placed file has no manifest and no run record, and the foreign -write check above exists precisely to catch it. - -## Two modes, derived from the project - -How recipes execute is never configured — it is read off the project: - -- **Direct mode** (the default): recipes run on your machine, in the - project's `.venv`, under an OS sandbox — Landlock on Linux, Seatbelt - on macOS. The project tree is read-only except each recipe's own - directory its output lands in; undeclared tools don't execute. -- **Containerized mode**: declaring a `[tool.lightcone.image]` table in - `pyproject.toml` *is* the switch. Recipes then run inside a - content-addressed image built from that declaration — and the image - itself is saved into the repository as versioned content, so a clone - obtains the exact bytes with no registry and no credentials. - `lc status` shows the mode and the image's state. - -Either way, every manifest records what enforcement actually ran -(`hermeticity`) — a host with no sandbox mechanism runs the recipe and -says so, rather than pretending. - -## Reading and gating are different verbs - -- **`lc status`** reports. It always exits 0 — a state is not a - failure — runs nothing, and doesn't mind a dirty tree, because the - moment you most need it is when things aren't clean. It's also the - verb that shows the commit each output was made at. -- **`lc materialize --check`** gates. It classifies everything without - running anything and exits 1 if a run would do work — the thing a - script or CI job branches on. - -Both have `--json`; the first two keys of the check report, `ok` and -`up_to_date`, are the ones to branch on. - -## Publication is a license away - -Declaring a `license` under `[project]` in `pyproject.toml` is -declaring the intent to publish. From then on, every `lc materialize` -maintains `ro-crate-metadata.json` at the project root — an -[RO-Crate](https://www.researchobject.org/ro-crate/) describing the -project, its outputs, and the runs that produced them. The repository -*is* the crate; depositing it is `git archive` on something you already -have. - -## Where to next - -- [Running on a Cluster](cluster.md) — the same model on SLURM. -- [Troubleshooting](troubleshooting.md) — the refusals quoted, with - their remedies. -- [Glossary](glossary.md) — the terms, one at a time. diff --git a/docs/user/getting-started.md b/docs/user/getting-started.md deleted file mode 100644 index a0e6e6f6..00000000 --- a/docs/user/getting-started.md +++ /dev/null @@ -1,411 +0,0 @@ -# Getting Started - -Let's go from nothing on your disk to a working, reproducible analysis. -You can read this top to bottom without running anything, or follow along — -every command is copy-paste ready. - -**What you'll build:** a small two-output analysis that fits a line to a -noisy dataset and sweeps one methodological decision — whether points far -from an initial fit are kept or clipped. The result is two universes, -`baseline` and `robust`, each with its own fitted slope and figure, and a -project that ends published as an [RO-Crate](https://www.researchobject.org/ro-crate/). - -Make sure you've finished the [install](install.md) first. - -## 1. Create a project - -```bash -lc init line-fit-demo -cd line-fit-demo -``` - -`lc init` converges the directory to a small, opinionated layout and -stops; it doesn't ask any questions, and it's idempotent — re-running -it later only fills in whatever is missing. - -``` -line-fit-demo/ -├── astra.yaml # the spec, empty for now — this is where everything lives -├── pyproject.toml # the project's environment: its dependencies… -├── .python-version # …and the exact interpreter, locked by uv -├── uv.lock -├── .venv/ # built from the lock (local, never committed) -├── .git/ # a git repository, with git-annex initialized -├── .gitattributes # the storage policy: what the annex carries -├── .gitignore -├── .datalad/ # dataset identity — the project is a DataLad dataset -├── data/ # declared input data lives here -├── results/ # outputs materialize here — lc's to write, not yours -├── universes/ -│ └── baseline.yaml # one universe, selecting nothing yet -├── myst.yml # MyST report configuration -└── index.md # template report, to reference the spec from -``` - -Two things are worth registering now: - -- **The project is a git repository.** Every - output `lc` makes is committed together with the code that produced - it; large files ride in git-annex behind the scenes, but you only - ever type ordinary `git add` and `git commit`. -- **The environment is the lock.** `pyproject.toml` + `uv.lock` define - exactly what your recipes can import, and `.venv` is built from them. - You'll add packages with `uv add` in a moment — never `pip install`. - -The file you'll actually work in is **`astra.yaml`** — the single source -of truth for your analysis. Inputs, outputs, methodological decisions, -recipes: everything else lightcone-cli does is downstream of this file. - -## 2. Add the data - -A real project starts from a dataset; ours will generate a small one — -200 points on a line, with a few outliers thrown far off it: - -```bash -python3 - <<'EOF' -import random -random.seed(0) -rows = ["x,y"] -for _ in range(200): - x = random.uniform(0, 10) - y = 2.5 * x + 1.0 + random.gauss(0, 1.5) - if random.random() < 0.04: - y += random.gauss(0, 15) - rows.append(f"{x:.6f},{y:.6f}") -open("data/points.csv", "w").write("\n".join(rows) + "\n") -EOF -``` - -`data/` is where declared inputs live. When you commit, the -`.gitattributes` policy routes the file's bytes into git-annex -automatically — the file stays an ordinary readable, writable file in -your tree, and the repository stays light. - -## 3. Write the spec - -`astra.yaml` was scaffolded as an empty analysis. Fill it in with ours: - -```yaml -version: "0.0.13" # ASTRA schema version — keep what the scaffold wrote -name: "line_fit" -description: | - Fit a straight line to a small synthetic dataset and sweep one - methodological decision: whether points far from an initial fit are - kept or clipped before the final fit. - -inputs: - - id: points - type: data - source: data/points.csv - description: "200 synthetic (x, y) points, a few of them far off the line" - -outputs: - - id: fit - type: metric - format: json - description: "Slope and intercept of the least-squares line" - inputs: [points] - decisions: [outliers] - recipe: - command: python src/fit.py --points {inputs.points} --outliers {decisions.outliers} --output {output} - - - id: fit_plot - type: figure - format: png - description: "The points and the fitted line" - inputs: [points, fit] - recipe: - command: python src/plot.py --points {inputs.points} --fit {inputs.fit} --output {output} - -decisions: - outliers: - label: "Outlier handling" - rationale: "A few points sit far off the line; keeping or clipping them shifts the slope." - default: keep - options: - keep: - label: "Keep every point" - clip: - label: "Drop points beyond 3 sigma of an initial fit" -``` - -A few things to notice: - -- Each output declares its full dependency contract: `fit` depends on - the `points` input and the `outliers` decision; `fit_plot` depends on - `points` and on the sibling output `fit`. That contract is how `lc` - orders the build — and how it knows what to rebuild when something - changes. -- Recipes reference those dependencies through placeholders — - `{inputs.points}`, `{decisions.outliers}`, `{output}` — which are - expanded at execution time. `{output}` is the output's own file, - `results//.`; the engine creates the - directory before the recipe runs, and the recipe writes that one path. -- Each output declares a `format` — the extension its artifact is - written with. It is what names the file, so a consumer knows what an - output *is* from the spec alone, and one output is always one file. -- The decision's options aren't hardcoded anywhere in code; the scripts - will take them as command-line arguments. - -`universes/baseline.yaml` was scaffolded empty, so give it a value for -our decision: - -```yaml -id: baseline -description: "Every point kept — the decision defaults." -decisions: - outliers: keep -``` - -Each universe is one complete selection of decision values; its results -materialize to `results//.`. - -Check the spec is well-formed: - -```bash -astra validate astra.yaml -``` - -(`astra` is the spec-side CLI; it ships with `astra-tools`, a dependency -of lightcone-cli.) - -## 4. Write the scripts - -Two short scripts, in a `src/` directory (`mkdir src` — the scaffold -doesn't create it; where code lives is your choice, the recipes above -just happen to point there). First `src/fit.py`: - -```python -import argparse -import json -from pathlib import Path - -import numpy as np - -parser = argparse.ArgumentParser() -parser.add_argument("--points", required=True) -parser.add_argument("--outliers", choices=["keep", "clip"], required=True) -parser.add_argument("--output", required=True) -args = parser.parse_args() - -x, y = np.loadtxt(args.points, delimiter=",", skiprows=1, unpack=True) -if args.outliers == "clip": - slope, intercept = np.polyfit(x, y, 1) - residuals = y - (slope * x + intercept) - mask = np.abs(residuals) < 3 * residuals.std() - x, y = x[mask], y[mask] -slope, intercept = np.polyfit(x, y, 1) - -Path(args.output).write_text( - json.dumps({"slope": slope, "intercept": intercept, "n_used": len(x)}, indent=2) -) -``` - -Then `src/plot.py` — reads the upstream output's file, makes the -figure: - -```python -import argparse -import json -from pathlib import Path - -import matplotlib - -matplotlib.use("Agg") -import matplotlib.pyplot as plt -import numpy as np - -parser = argparse.ArgumentParser() -parser.add_argument("--points", required=True) -parser.add_argument("--fit", required=True) -parser.add_argument("--output", required=True) -args = parser.parse_args() - -x, y = np.loadtxt(args.points, delimiter=",", skiprows=1, unpack=True) -fit = json.loads(Path(args.fit).read_text()) - -fig, ax = plt.subplots() -ax.scatter(x, y, s=12) -xs = np.linspace(x.min(), x.max(), 2) -ax.plot(xs, fit["slope"] * xs + fit["intercept"], color="C1") -ax.set_xlabel("x") -ax.set_ylabel("y") -ax.set_title(f"slope = {fit['slope']:.3f}") -fig.savefig(args.output, dpi=150) -``` - -Both scripts import from the project's locked environment, so declare -what they need: - -```bash -uv add numpy matplotlib -``` - -That one command updates `pyproject.toml`, re-locks `uv.lock`, and syncs -`.venv`. It's the only way packages reach a recipe — recipes run -sandboxed in the locked environment, so a stray `pip install` on your -machine changes nothing they can see. That's a feature: the lock *is* -the record of what your results were computed with. - -## 5. Materialize - -Launch the built-in local offer; no compute configuration is needed. It provides -all usable CPUs and RAM, and stops once it has had no work for 30 minutes. Keep -the returned ID in `CLUSTER` for this walkthrough. Configured remote offers coexist with that default. A catalog can -override the local budget, disable local compute, or provide explicit local offers; -see [Running on a Cluster](cluster.md). NERSC login nodes automatically refuse local -compute; use a compute node in an interactive allocation or a configured Slurm offer. - -```bash -CLUSTER=$(lc compute launch --wait --json | python -c 'import json,sys; print(json.load(sys.stdin)["id"])') -``` - -Execution always requires this cluster ID. `lc materialize --check` can inspect -what needs rebuilding without allocating compute. If the allocation expires -during the walkthrough, launch another one and replace `CLUSTER` with its new ID. - -Commit, then build: - -```bash -git add -A && git commit -m "Line-fit analysis" -lc materialize "$CLUSTER" -``` - -The commit isn't ceremony — every output is committed together with the -code that produced it, so a build refuses to start from a tree with -uncommitted edits (it wouldn't be able to say what code ran). Then: - -``` - ✓ made baseline/fit - ✓ made baseline/fit_plot - ! no [project].license in pyproject.toml, so no RO-Crate publication - view is maintained — declare one to enable it - -✓ Made 2 output(s) in /home/you/line-fit-demo -``` - -(We'll come back to that license line in step 7.) Each output landed in -`results/baseline/.` next to a -`..manifest.json` — -a manifest recording the recipe, the decisions, the input hashes, the -environment, and the commit — and was committed with a run record that -`datalad rerun` can replay. Look at `git log`: the build wrote history, -not just files. - -Check where things stand any time: - -```bash -lc status -``` - -``` - mode: direct - sandbox: landlock (fs: declared, network: allowed) - crate: not maintained — declare [project].license to enable it - - · current baseline/fit a3f1f11 - · current baseline/fit_plot a3f1f11 - -2 current -``` - -The commit column is the answer to "which code made this?" — for every -output, current or not. And `lc materialize` is idempotent: run it again -and it reports the project is up to date without executing anything. - -## 6. Sweep the decision - -Add the second universe — `universes/robust.yaml`: - -```yaml -id: robust -description: "Points beyond 3 sigma of an initial fit are dropped." -decisions: - outliers: clip -``` - -Commit and materialize again: - -```bash -git add -A && git commit -m "Add the robust universe" -lc materialize "$CLUSTER" -``` - -``` - ✓ made robust/fit - ✓ made robust/fit_plot - · up to date baseline/fit - · up to date baseline/fit_plot - -✓ Made 2 output(s) in /home/you/line-fit-demo -``` - -Only the new universe's outputs ran — `baseline` was already exactly -what the spec asks for, so it wasn't touched. Your comparison is on -disk: with this guide's synthetic dataset, clipping drops 4 points and -moves the slope from 2.414 to 2.450 — visibly closer to the true 2.5 -the data was generated with. - -If a recipe fails, `lc materialize` reports which output failed and why, -and leaves the tree as clean as it found it; fix the script or the spec, -commit, and rerun — only the affected outputs re-execute. - -## 7. Publish - -RO-Crate requires a license, so declaring one is how you tell `lc` the -project is meant for the outside world. Add one line under `[project]` -in `pyproject.toml`: - -```toml -license = "CC-BY-4.0" -``` - -then commit and materialize once more: - -```bash -git add -A && git commit -m "Declare a license" -lc materialize "$CLUSTER" -``` - -Nothing is rebuilt — but `ro-crate-metadata.json` appears at the project -root and is committed automatically. From here on, every materialize -keeps it in line with the repository: the project *is* the crate, and -depositing it is just `git archive` (or `datalad export-archive`) on a -repository you already have. - -## What just happened - -- `astra.yaml` was the only place your analysis was *described* — - inputs, outputs, the decision, and the recipes all live there. -- The scripts take decision values as plain command-line arguments, so - nothing methodological is hardcoded. -- `lc materialize` ran each recipe in the project's locked environment, - sandboxed — free to write the directory its output lands in, and - nothing else — - and committed every output with a manifest and a re-runnable run - record. -- `lc status` and `lc materialize --check` read those manifests — they - don't re-execute anything; they just classify. An output is remade - when the spec defines it differently than it was made, or when its - declared inputs changed; an output whose *environment* has since - moved is reported as `behind` and deliberately left alone — the - manifest records exactly which environment and commit produced it. - -Clone this repository on a fresh machine, run `lc init` (it rebuilds -the two pieces of local state git doesn't carry — the `.venv` and the -annex), then `lc materialize --check`: it reports up to date without fetching a -single data byte, because the provenance travels in git. The bytes -themselves follow with `git annex get` whenever you actually need them. - -## Where to next - -- [Core Concepts](concepts.md) — the model behind what you just did: - the three states, the commit discipline, the two execution modes. -- [Running on a Cluster](cluster.md) — take the same project to SLURM. -- [Troubleshooting](troubleshooting.md) — when something goes sideways. -- [Glossary](glossary.md) — terms like universe, decision, and manifest - in plain language. -- The [ASTRA docs](https://astra-spec.org/latest/) — the full spec: - sub-analyses, prior insights, findings, and evidence. - -Release the allocation when finished: `lc compute down "$CLUSTER"`. diff --git a/docs/user/glossary.md b/docs/user/glossary.md deleted file mode 100644 index 537ba98f..00000000 --- a/docs/user/glossary.md +++ /dev/null @@ -1,186 +0,0 @@ -# Glossary - -The terms you'll see all over the docs and the `lc` command output, in -plain language. - -## ASTRA - -**A**gentic **S**chema for **T**ransparent **R**esearch **A**nalysis. -The schema lightcone-cli is built around. ASTRA's job is to capture an -analysis's inputs, outputs, and methodological decisions in a single -file (`astra.yaml`); lightcone-cli's job is to execute that spec -reproducibly. ASTRA ships separately as the `astra-tools` package, and -its `astra` CLI handles the spec itself (validation, universe -management, evidence verification). - -## astra.yaml - -Your project's spec file. The single source of truth — every input, -output, recipe, and decision is declared here. Sub-analyses can be -nested via `analyses:` references. - -## Recipe - -A short shell command that produces an output. Lives inside an output's -`recipe:` block in `astra.yaml`. Outputs declare what they depend on, -and the recipe references those dependencies through placeholders: - -```yaml -outputs: - - id: fit - inputs: [points] - decisions: [outliers] - recipe: - command: python src/fit.py --points {inputs.points} --outliers {decisions.outliers} --output {output} -``` - -## Decision - -A methodological choice with multiple defensible options (e.g. -"standardize features?", "what outlier threshold?"). Decisions live -in the `decisions:` section of `astra.yaml` along with their `default`, -their `options`, and their `rationale`. - -## Universe - -One specific selection of decision values. Universes live as YAML -files in `universes/` (e.g. `universes/baseline.yaml`, -`universes/robust.yaml`). Each universe materializes its results -to its own directory: `results//.`. - -## Sub-analysis - -A nested ASTRA analysis with its own inputs, outputs, and decisions, -referenced from a parent's `analyses:` section. `lc` materializes a flat -analysis: an output id it cannot name a file from is refused, so a -nested spec is not buildable today. - -## Materialize - -Making the outputs the spec declares: `lc materialize` runs each recipe -in dependency order and commits every result as it lands. Idempotent — -a second run remakes only what is `stale`, and a run with nothing to do -says so and touches nothing. - -## Manifest - -The per-output sidecar JSON file, `..manifest.json` beside -the output itself, recording what produced the -output: the recipe, the decisions, `definition_version`, -`env_version`, `data_version`, `input_versions`, the git commit the -run started at, the engine version, what enforcement actually ran -(`hermeticity`), and — for containerized runs — the image. Written by -the run, read by `lc status` and `lc materialize --check`; kept in -plain git so a clone can classify a whole project without fetching any -data. - -## definition_version - -A hash of an output's recipe and decision values — the fingerprint of -"what is this output?". When it drifts, the output is `stale` and the -next run remakes it. - -## env_version - -A hash of the environment — the lock file's bytes, the pinned -interpreter, the install settings, and the image declaration if any. -Deliberately *not* part of an output's definition: when it drifts, the -output is `behind`, reported and left alone. - -## data_version - -A content hash of an output's bytes (or of a declared input). For a -file it is a plain sha256 — the number `sha256sum` prints, and the one -the RO-Crate publishes; a directory-valued declared input is hashed -tree-wise and framed, so the two can never collide. This is what flows -downstream: a dependent is remade when an input's `data_version` -changed, and a rebuild that comes out byte-identical stops the cascade -right there. - -## input_versions - -Inside a manifest, a map from each declared input to the -`data_version` it had when the output was made. Comparing it against -the present is how a change to an input cascades. - -## current / behind / stale - -The three states an output can be in: - -- `current` — exactly what the spec asks for, made from these inputs, - under this environment. Nothing to do. -- `behind` — still what the spec asks for; only the environment moved - since it was made. Reported, left alone; `--refresh` remakes. -- `stale` — contradicts the project: the spec defines it differently, - an input changed, or the output was edited by hand. Remade on the - next run. - -## Direct mode / containerized mode - -How recipes execute, derived from the project rather than configured. -Direct mode (the default): the project's `.venv`, under the OS sandbox. -Containerized mode: declaring `[tool.lightcone.image]` in -`pyproject.toml` switches the project over — recipes run inside a -content-addressed image built from that declaration. - -## Image - -Containerized mode's execution world: a base (digest-pinned), optional -apt packages, and the pinned interpreter. Built by `lc build` and saved -**into the repository** as versioned content, so clones obtain the -exact bytes through git-annex with no registry involved. Execution pins -the image's content id, never a tag. - -## Runtime - -The OCI tool that executes containers. Detected, never configured: -`podman-hpc`, then `podman`, then `docker` (skipped if its daemon is -down). - -## Sandbox - -The isolation every recipe and every `lc run` command executes under — -Landlock on Linux, Seatbelt on macOS, the container boundary in -containerized mode. The project tree is read-only apart from the -directory the output being made lands in; undeclared tools don't -execute. Each -manifest's `hermeticity` field records what was actually enforced, and -a host with no mechanism says so rather than pretending. - -## git-annex - -How the repository carries data: git holds history and small files, -git-annex holds the bytes of `data/` and `results/` behind ordinary -git commands. You never run git-annex yourself except to fetch bytes -on a clone (`git annex get`), and `lc materialize` even does that for -declared inputs it needs. - -## Run record - -The commit message a materialized output is saved under — a -machine-readable record of the exact command that made it, in a format -`datalad rerun` can replay: it reconstructs the engine, the project -environment, and the sandbox, and remakes the output from its spec. -Your `git log` is the build log. - -## RO-Crate - -The publication view. Declare a `license` under `[project]` in -`pyproject.toml` and every materialize maintains -`ro-crate-metadata.json` — a machine-readable description of the -project, its outputs, and the runs that produced them, following the -Provenance Run Crate profile. The repository is the crate; deposit is -`git archive`. - -## Prior insight - -A piece of evidence from the literature that informs a decision. -Lives in the `prior_insights:` section of `astra.yaml`, with a `claim` -and verifiable `evidence` (DOI plus exact quote). - -## Finding - -A conclusion drawn *from* the analysis (as opposed to a prior insight, -which comes *into* it). Findings live in the `findings:` section and -cite specific outputs as evidence — the bridge between materialized -results and the eventual paper. diff --git a/docs/user/index.md b/docs/user/index.md deleted file mode 100644 index 7fbee559..00000000 --- a/docs/user/index.md +++ /dev/null @@ -1,73 +0,0 @@ -# Welcome to the user guide - -`lightcone-cli` is a small toolchain that turns a research question into -a reproducible analysis. You describe what you're trying to learn as a -precise specification — an `astra.yaml` file following the -[**ASTRA**][astra] schema — and the `lc` command line keeps the -resulting code, environments, decisions, and outputs in sync. - -ASTRA specs are plain YAML, designed to be easy for both humans and AI -assistants to write. However the spec gets written, **you stay in charge -of the scientific choices** — every methodological decision is declared -in the open, and `lc` records exactly what produced every result: the -recipe, the decisions, the input data, the environment, and the commit. - -## What this guide covers - -- [Install](install.md) — get the `lc` command line running on your - machine or on a cluster. -- [Getting Started](getting-started.md) — create your first project, - build it end-to-end, and understand what each piece does. -- [Core Concepts](concepts.md) — the model behind the tool: what the - states mean, why everything is committed, and how the two execution - modes differ. -- [Running on a Cluster](cluster.md) — taking your analysis to a SLURM - HPC system. -- [Troubleshooting](troubleshooting.md) — common issues and how to - unstick them. -- [Glossary](glossary.md) — the terms that show up everywhere - (universe, decision, manifest, …) explained in plain language. - -## What you'll do, in a handful of lines - -!!! tip "Quick start" - - === "uv" - ```bash - uv tool install lightcone-cli - lc init my-analysis && cd my-analysis - # describe your analysis in astra.yaml, write your scripts, - # declare what they import (uv add numpy ...), then: - git add -A && git commit -m "First analysis" - lc materialize "$CLUSTER" - ``` - - === "pip" - ```bash - pip install lightcone-cli - lc init my-analysis && cd my-analysis - # describe your analysis in astra.yaml, write your scripts, - # declare what they import (uv add numpy ...), then: - git add -A && git commit -m "First analysis" - lc materialize "$CLUSTER" - ``` - -That's the shortest possible path. The rest of the guide is the -unhurried version — and the commit is not ceremony: every output is -committed together with the code that produced it, which is why a build -starts from a clean tree. - -## What lightcone-cli is *not* - -- **A statistics package.** It runs your code; it doesn't compute - things itself. -- **A workflow language.** Recipes in `astra.yaml` are short shell - commands, not a DSL. There's no learning curve beyond what's in - [Getting Started](getting-started.md). -- **An IDE.** `lc` is a command-line tool; write `astra.yaml` and your - analysis code with whatever editor or tooling you prefer. - -If you'd rather skim the design and architecture, the -[maintainer docs](../maintainer.md) are the other half of this site. - -[astra]: https://astra-spec.org/latest/ diff --git a/docs/user/install.md b/docs/user/install.md deleted file mode 100644 index 48f13ed0..00000000 --- a/docs/user/install.md +++ /dev/null @@ -1,124 +0,0 @@ -# Install - -To work on a lightcone project you need two things on your machine: -[uv](https://docs.astral.sh/uv/) and git. Everything else — Python -itself included — is installed by uv or ships with `lc`. - -!!! note "Supported platforms" - Linux (glibc 2.34+, x86_64 or aarch64) and macOS (14+ on Apple - silicon, 15+ on Intel). On Windows, use WSL. - -## 1. uv and git - -`lc` uses uv as its only environment substrate — projects are -`pyproject.toml` + `uv.lock`, and uv manages the Python interpreters -too, so there is no separate Python install step. - -=== "macOS / Linux" - ```bash - curl -LsSf https://astral.sh/uv/install.sh | sh - ``` - - git is preinstalled on macOS; on Linux use your package manager - (`apt install git`, `dnf install git`, …). - -=== "NERSC Perlmutter" - NERSC doesn't ship `uv`, but it installs into your home directory - with a single curl: - - ```bash - curl -LsSf https://astral.sh/uv/install.sh | sh - ``` - - `uv` lands under `~/.local/bin` — make sure it's on your `PATH`. - git is already on the system. - -## 2. lightcone-cli - -The published name on PyPI is `lightcone-cli`; the command it provides -is `lc`. - -=== "uv" - ```bash - uv tool install lightcone-cli - ``` - -=== "pip" - ```bash - python -m pip install lightcone-cli - ``` - -Get a confirmation of the proper installation by running - - lc --version # → lc, version ... - -> **Note** Some people may have already set a personal shell alias -> `lc='ls --color'`. If that's you, installing lightcone-cli will shadow -> the alias — make sure to rebind it (e.g. `alias l='ls --color'`). - -## 3. Tell git who you are - -Every output `lc` makes is committed, so git needs an identity before -the first build — `lc materialize` checks up front rather than failing -after your recipes have run: - -```bash -git config --global user.name "Ada Lovelace" -git config --global user.email "ada@example.org" -``` - -If you already commit from this machine, you're done. - -## 4. (Optional) Podman or Docker - -Only *containerized* projects need a container runtime — a project opts -in by declaring `[tool.lightcone.image]` in its `pyproject.toml`, and -until it does, recipes run directly on your machine in the project's -own locked environment. - -- Local machine: install [Podman](https://podman.io/) (rootless, no - daemon) or [Docker](https://docs.docker.com/get-docker/). -- HPC login node: see [Running on a Cluster](cluster.md). - -There is nothing to configure: `lc` detects whichever runtime is -available (`podman-hpc`, then `podman`, then `docker` — skipping docker -if its daemon isn't running). - -## Sanity check - - lc --help - lc init --help - -Both should print help text. If `lc` is shadowed by an `ls` alias, -unset it (`unalias lc`) or use the full path (`$(which lc) --version`). - -## Updating - -=== "uv tool" - ```bash - uv tool upgrade lightcone-cli - ``` - -=== "pip" - ```bash - pip install -U lightcone-cli - ``` - -An upgrade never invalidates your results: the engine's version is -recorded in every output's manifest, but it is not part of any output's -identity, so nothing gets rebuilt just because `lc` moved. - -## Uninstalling - -=== "uv tool" - ```bash - uv tool uninstall lightcone-cli - ``` - -=== "pip" - ```bash - pip uninstall lightcone-cli - ``` - -Your projects are untouched — everything `lc` knows about an analysis -lives in the project's own repository, not in any global state. diff --git a/docs/user/troubleshooting.md b/docs/user/troubleshooting.md deleted file mode 100644 index 9ba673a0..00000000 --- a/docs/user/troubleshooting.md +++ /dev/null @@ -1,239 +0,0 @@ -# Troubleshooting - -Common situations and how to unstick them, roughly ordered by how often -they come up. `lc`'s refusals try to carry their own remedy — this page -adds the context around them. - -## "uncommitted changes in …" - -``` -Error: uncommitted changes in /home/you/my-analysis — every materialization is committed with the code that produced it, so a run cannot start from a tree that does not say what that code is. - - commit these: git add -A . && git commit -m "…" - ?? src/ - - if a cluster run was interrupted, stop its allocation first: - lc compute down - confirm its recipes have stopped, then discard these (lc writes results/): - git restore --staged --worktree results/ && git clean -fd results/ - ?? results/baseline/ -``` - -Usually not an error in your project — just the order of operations: -commit, then materialize. The refusal sorts the paths it found: work you -own gets the `commit these` line, while files under `results/` are listed -as wreckage to discard instead — `results/` is `lc`'s to write, and -committing hand-placed files there defeats the provenance the tool -exists for. - -Files under `results/` usually come from an interrupted run. `lc` leaves -an interrupted run's unreported outputs in place on purpose, because the -recipes writing them may still be running on the cluster. Stop the -allocation first, by full ID, then discard them. - -## "… is not a Lightcone project" - -You're outside a project. The current directory *is* the project — `lc` -never walks up to find one, by design — so: - -```bash -cd path/to/your/project -``` - -or, starting fresh, `lc init my-analysis && cd my-analysis`. If you're -in a fresh clone, run `lc init` once — it rebuilds the `.venv` and the -annex, the two pieces of local state git doesn't carry. - -## "lc: command not found" or `lc` prints a directory listing - -Two possibilities: - -1. The tool isn't on `PATH` — with `uv tool install`, that's - `~/.local/bin`; `uv tool update-shell` fixes the profile. -2. Your shell has a personal alias `lc='ls --color'` shadowing the - real command. Run `type lc` to see; `unalias lc` to remove. - -## A recipe fails with "Permission denied" or "No module named …" - -Every sandboxed failure ends with this trailer: - -``` -this ran under the lc sandbox (landlock) — a permissions or missing-file -error can mean the command reached for something outside the declared -environment -``` - -Recipes run in the project's locked environment, with the tree -read-only apart from the directory their output lands in. The common cases: - -- **`ModuleNotFoundError`** — the package isn't in the project's lock. - `uv add `, commit, re-run. (Installing it on the host with - `pip` changes nothing a recipe sees — that's the point.) -- **Reading a file outside the project** — declare it as an ASTRA - input; declared inputs are readable and their content becomes part - of the output's provenance. -- **Writing outside that directory** — a recipe's product - belongs in `{output}`; for true scratch files, use - `tempfile.mkdtemp()`, which lands in the writable temp area. - -To probe interactively, `lc run CLUSTER_ID -- ` runs any command under -exactly the isolation a recipe gets — if it works there, it works as a -recipe. - -## Everything shows `behind` after a `uv add` - -Not a problem, and nothing was invalidated. `behind` means: the output -is still exactly what the spec asks for, but the environment has moved -since it was made. Environment changes deliberately don't trigger -rebuilds — the manifest records which environment and commit produced -each output, so nothing is lost by leaving it. When you do want them -remade under the current environment: - -```bash -lc materialize "$CLUSTER" --refresh -``` - -See [Core Concepts](concepts.md) for the `stale` / `behind` -distinction. - -## Everything shows `stale` after a spec edit - -`stale` means the spec now defines the output differently than it was -made — you edited its recipe, a decision, or a declared input's -content changed. That's the invalidation model working; the next -`lc materialize` remakes exactly those outputs. - -One edit that deliberately does *not* invalidate: changing your -analysis code (`src/…`). The recipe *string* is the identity, so if -you want code changes to cascade, declare the source file as an ASTRA -input of the outputs it shapes — that choice is yours to make per -output. - -## "the content is not in this clone" - -``` -data/points.csv: the content is not in this clone — git-annex holds a -reference to it, not the data. Fetch it with `git annex get data/points.csv`. -``` - -The clone has the *pointer* to an annexed file but not its bytes. -`lc materialize` fetches the declared inputs it needs by itself; the -read-only verbs (`lc status`, `--check`) never transfer data, so they -report the fact instead. Fetch by hand only when you want the bytes -for your own inspection. - -## "fatal: … clean filter 'annex' failed" - -``` -git-annex filter-process: line 1: git-annex: command not found -error: could not read greeting from subprocess 'git-annex filter-process' -error: initialization for subprocess 'git-annex filter-process' failed -fatal: data/catalog.fits: clean filter 'annex' failed -``` - -Your shell's `PATH` has no `git-annex`, so git could not run the filter -that turns a large file into an annex pointer. **Nothing was staged**, -which is the point: without `filter.annex.required=true` — which -`lc init` sets — git would have exited 0 and committed the raw bytes -into history instead. - -Once a project holds committed annexed content, this is not limited to -`git add`. Any command that has to run the filter over that content -stops the same way, `git status`, `git diff` and `git checkout` -included — so the whole project reads as broken until git-annex is back -on your `PATH`. That is the intended shape of the failure: a repository -you cannot use is recoverable in one command, and one that quietly -absorbed a multi-gigabyte file is not. - -`git-annex` ships with `lc`, so a tool install puts both on your `PATH`: - -```bash -uv tool install lightcone-cli -git-annex version -``` - -If `lc` runs but `git-annex` does not, uv's tool directory is not on -your `PATH` — run `uv tool update-shell` and open a new shell. Running -`lc` through `uvx` puts nothing on your `PATH` at all, so a plain -`git add` cannot work that way. - -This failure is deliberately loud. `lc init` sets -`filter.annex.required=true` in every project precisely because -without it git handles the same situation by printing the error, -**exiting 0, and staging your data's raw bytes into git history** — -committing a multi-gigabyte dataset into git proper, silently, where -every clone carries it forever. A refused `git add` costs you one -`lc init`; the silent version costs you the repository. - -## "… has not started yet" or "does not have its expected workers" - -`lc run` and `lc materialize` never wait for a cluster: they refuse one -that is not active with a worker connected for every node. Right after a -launch, wait for readiness first: - -```bash -lc compute status "$CLUSTER" --wait -``` - -If a Slurm allocation stays active but never becomes ready, its startup -failed. Its submission log, under `submissions//` in the -connection root (`~/.lightcone/compute` by default), records why, on a line -starting `Slurm Dask startup failed:` — for example a node shape that -does not match the offer, or a scheduler that did not start within 120 -seconds. A local allocation's startup failure is shown as the reason by -`lc compute status CLUSTER`. - -## uv: "Could not acquire lock" (os error 524) on a cluster - -A recipe or `lc run` probe on a Slurm compute node fails because uv -cannot lock its cache: the cache is on a filesystem the compute nodes -mount without file locking, which is the case for `$HOME` at NERSC. Put -the cache on one that supports locks, then launch a new allocation, -since workers keep the environment they were launched with: - -```bash -export UV_CACHE_DIR=$PSCRATCH/uv-cache -lc compute down "$OLD_CLUSTER_ID" -CLUSTER=$(lc compute launch --cpus 256 --memory 480) -``` - -See [Configure Slurm](cluster.md#configure-slurm). - -## Selecting compute from a login shell - -Use `lc compute resources` and `lc compute launch`, then pass the returned cluster -ID to `run` or `materialize`. Lightcone does not inspect hostnames or site markers -to reject local compute. The configured catalog and native backend permissions -determine what can be allocated. See [Running on a Cluster](cluster.md). - -## git doesn't know who you are - -Every output is committed, so a machine that has never committed needs -an identity before the first run — `lc materialize` checks up front, -before any recipe spends time: - -```bash -git config --global user.name "Ada Lovelace" -git config --global user.email "ada@example.org" -``` - -## Containerized projects - -- **"image absent"** — the declared image hasn't been built and - committed yet: `lc build` (announced by materialize too, which - builds it as a preflight when missing). -- **No runtime found** — install [Podman](https://podman.io/) or - [Docker](https://docs.docker.com/get-docker/); detection is - automatic and there is nothing to configure. -- **Architecture mismatch** — the committed archive records the - architecture it was built for, and a host that can't execute it is - refused before the recipe would have died mid-run. Build on a - matching host (on NERSC, a login node), commit, push, and pull on - the other side. - -## Filing a bug - -Open an issue at -[github.com/LightconeResearch/lightcone-cli/issues](https://github.com/LightconeResearch/lightcone-cli/issues). -Include the output of `lc --version`, the command you ran, and the -full message — the refusals are designed to be pasted. diff --git a/pyproject.toml b/pyproject.toml index 6a1206af..d01fe6a0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -47,12 +47,6 @@ dev = [ "types-pyyaml>=6.0.12.20260906", "types-psutil>=7.2.2.20260906", ] -docs = [ - "zensical>=0.0.33", - # squidfunk's mike fork — required by zensical's versioning provider. - # Not on PyPI; install from git. - "mike @ git+https://github.com/squidfunk/mike.git ; python_version >= '3.10'", -] [project.urls] diff --git a/zensical.toml b/zensical.toml deleted file mode 100644 index fc1ed491..00000000 --- a/zensical.toml +++ /dev/null @@ -1,88 +0,0 @@ -[project] -site_name = "lightcone-cli" -site_description = "Execution layer for ASTRA research pipelines" -site_author = "Lightcone Research Team" -repo_url = "https://github.com/LightconeResearch/lightcone-cli" -repo_name = "LightconeResearch/lightcone-cli" -copyright = "© 2026 Lightcone Research" -docs_dir = "docs" -extra_css = ["stylesheets/extra.css"] - -nav = [ - {"Home" = "index.md"}, - {"User Guide" = [ - {"Welcome" = "user/index.md"}, - {"Install" = "user/install.md"}, - {"Getting Started" = "user/getting-started.md"}, - {"Core Concepts" = "user/concepts.md"}, - {"Running on a Cluster" = "user/cluster.md"}, - {"Troubleshooting" = "user/troubleshooting.md"}, - {"Glossary" = "user/glossary.md"}, - ]}, - {"Developer Corner" = [ - {"Welcome" = "maintainer.md"}, - {"Architecture" = "architecture.md"}, - {"CLI Reference" = [ - {"Overview" = "cli/index.md"}, - {"lc init" = "cli/init.md"}, - {"lc materialize" = "cli/materialize.md"}, - {"lc status" = "cli/status.md"}, - {"lc run" = "cli/run.md"}, - {"lc compute" = "cli/compute.md"}, - {"lc build" = "cli/build.md"}, - ]}, - {"Engine Internals" = [ - {"Overview" = "api/index.md"}, - {"project" = "api/project.md"}, - {"dataset" = "api/dataset.md"}, - {"identity" = "api/identity.md"}, - {"plan" = "api/plan.md"}, - {"assets" = "api/assets.md"}, - {"worker" = "api/worker.md"}, - {"materialize" = "api/materialize.md"}, - {"compute" = "api/compute.md"}, - {"sandbox" = "api/sandbox.md"}, - {"image & container" = "api/container.md"}, - {"crate" = "api/crate.md"}, - ]}, - {"Contributing" = [ - {"Development Setup" = "contributing/setup.md"}, - {"Testing" = "contributing/testing.md"}, - {"Extending" = "contributing/extending.md"}, - ]}, - ]}, - {"ASTRA docs" = "https://astra-spec.org/latest/"}, -] - -# Versioning is handled by mike (squidfunk's fork; see the docs -# dependency group). Each version of the site is deployed as a -# subdirectory of the gh-pages branch (e.g. /0.4.1/, /latest/). -# The version picker is rendered natively in the header. -[project.extra.version] -provider = "mike" - -[project.theme] -variant = "modern" -logo = "assets/logo.svg" -favicon = "assets/favicon.svg" -features = [ - "navigation.tabs", - "navigation.sections", - "navigation.top", - "search.highlight", - "content.code.copy", -] - -[[project.theme.palette]] -scheme = "default" -primary = "custom" -accent = "custom" -toggle.icon = "lucide/sun" -toggle.name = "Switch to dark mode" - -[[project.theme.palette]] -scheme = "slate" -primary = "custom" -accent = "custom" -toggle.icon = "lucide/moon" -toggle.name = "Switch to light mode"