From cc121424e2f3695163207b911b017ec4ff7756f1 Mon Sep 17 00:00:00 2001 From: Christopher Haar Date: Thu, 1 Oct 2026 19:35:13 +0200 Subject: [PATCH] feat(serving-stack): add tests around provided serving stack, checking if a cluster mets requirements Signed-off-by: Christopher Haar --- .github/workflows/e2e.yml | 65 ++ apis/inferenceclusters/definition.yaml | 26 + apis/servingstacks/definition.yaml | 18 + docs/content/platform/inference-cluster.md | 12 + .../platform/serving-stack-requirements.md | 238 ++++ .../concepts/inference-cluster-provided.yaml | 28 + .../config/vocabularies/Modelplane/accept.txt | 4 + e2e/README.md | 7 + e2e/provided/charts.tsv | 12 + e2e/provided/manifests.yaml | 1019 +++++++++++++++++ e2e/provided/values/ai-gateway.yaml | 5 + e2e/provided/values/cert-manager.yaml | 5 + e2e/provided/values/envoy-gateway.yaml | 31 + .../values/kube-prometheus-stack.yaml | 31 + .../values/node-feature-discovery.yaml | 8 + .../values/nvidia-dra-driver-gpu.yaml | 7 + e2e/provided/values/trust-manager.yaml | 11 + e2e/run.sh | 140 ++- flake.nix | 1 + .../compose-inference-cluster/function/fn.py | 59 +- .../tests/test_fn.py | 87 ++ .../compose-serving-stack/function/fn.py | 268 ++++- .../function/stacks/__init__.py | 55 +- .../function/stacks/clouds/existing.py | 64 +- .../function/stacks/common.py | 71 +- .../function/stacks/components.py | 93 ++ .../function/stacks/dynamo.py | 33 +- .../function/stacks/standard.py | 8 +- .../compose-serving-stack/requirements_doc.py | 222 ++++ .../compose-serving-stack/tests/test_fn.py | 400 ++++++- .../tests/test_stacks.py | 23 + nix/apps.nix | 25 + nix/checks.nix | 27 + pyproject.toml | 2 + schemas/.lock.json | 2 +- .../modelplane/inferencecluster/v1alpha1.py | 4 + .../infrastructure/servingstack/v1alpha1.py | 4 + 37 files changed, 3062 insertions(+), 53 deletions(-) create mode 100644 docs/content/platform/serving-stack-requirements.md create mode 100644 docs/manifests/concepts/inference-cluster-provided.yaml create mode 100644 e2e/provided/charts.tsv create mode 100644 e2e/provided/manifests.yaml create mode 100644 e2e/provided/values/ai-gateway.yaml create mode 100644 e2e/provided/values/cert-manager.yaml create mode 100644 e2e/provided/values/envoy-gateway.yaml create mode 100644 e2e/provided/values/kube-prometheus-stack.yaml create mode 100644 e2e/provided/values/node-feature-discovery.yaml create mode 100644 e2e/provided/values/nvidia-dra-driver-gpu.yaml create mode 100644 e2e/provided/values/trust-manager.yaml create mode 100644 functions/compose-serving-stack/requirements_doc.py diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index dae613934..b699bc423 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -97,3 +97,68 @@ jobs: - name: Teardown if: always() run: nix run .#e2e -- --clean + + # The Provided-mode variant: the substrate is pre-installed on the + # workload cluster and Modelplane only checks it (see e2e/run.sh + # --provided). Its own label, because it roughly doubles the e2e + # minutes and the RequirementsMet flow only needs exercising on + # changes near the serving stack. + e2e-provided: + if: >- + github.event_name == 'workflow_dispatch' + || (github.event.action == 'labeled' && github.event.label.name == 'test-e2e-provided') + || (github.event.action != 'labeled' && contains(github.event.pull_request.labels.*.name, 'test-e2e-provided')) + runs-on: ubuntu-24.04 + timeout-minutes: 60 + + steps: + - name: Cleanup Disk + uses: jlumbroso/free-disk-space@54081f138730dfa15788a46383842cd2f914a1be # v1.3.1 + with: + android: true + dotnet: true + haskell: true + tool-cache: true + swap-storage: false + large-packages: false + docker-images: false + + - name: Checkout + uses: actions/checkout@v4 + + - name: Install Nix + uses: cachix/install-nix-action@v31 + with: + extra_nix_config: log-lines = 500 + + - name: Cache Nix store + uses: DeterminateSystems/magic-nix-cache-action@908b263ff629f4cc17666315b7fd3ec127c6244d # v14 + with: + use-flakehub: false + + - name: Run e2e (provided) + run: nix run .#e2e --print-build-logs -- --provided --verify + + - name: Collect diagnostics + if: failure() + run: | + mkdir -p e2e-logs + nix develop --command bash -c ' + kind export logs e2e-logs/control-plane --name modelplane-e2e-local || true + kind export logs e2e-logs/workload --name modelplane-e2e-workload || true + kubectl --context kind-modelplane-e2e-local get composite -o wide > e2e-logs/composites.txt 2>&1 || true + kubectl --context kind-modelplane-e2e-local get inferencecluster local -o yaml > e2e-logs/inferencecluster.yaml 2>&1 || true + kubectl --context kind-modelplane-e2e-local -n ml-team get modeldeployment,modelreplica,modelendpoint,modelservice -o yaml > e2e-logs/model.yaml 2>&1 || true + ' + + - name: Upload diagnostics + if: failure() + uses: actions/upload-artifact@v4 + with: + name: e2e-provided-diagnostics + path: e2e-logs + if-no-files-found: ignore + + - name: Teardown + if: always() + run: nix run .#e2e -- --clean diff --git a/apis/inferenceclusters/definition.yaml b/apis/inferenceclusters/definition.yaml index a3b6d897f..bc40ce6d5 100644 --- a/apis/inferenceclusters/definition.yaml +++ b/apis/inferenceclusters/definition.yaml @@ -128,6 +128,32 @@ spec: ReadWriteMany dynamic provisioning). minLength: 1 maxLength: 253 + components: + type: string + default: Managed + x-kubernetes-validations: + - rule: self == oldSelf + message: >- + spec.cluster.existing.components is immutable; + recreate the cluster to change who supplies the + serving substrate. + description: >- + Who supplies the serving substrate on this cluster. + Managed (the default) has Modelplane install every + serving stack component. Provided installs no + substrate: the cluster already runs cert-manager, + the gateway stack, Prometheus, the GPU DRA driver + and the stack's workload controller, and Modelplane + verifies they are present (the RequirementsMet + condition reports what's missing) and composes only + its own configuration on top. All or nothing; the + cluster provides the whole substrate or none of it. + Immutable because flipping it would uninstall a + live cluster's substrate or adopt one Modelplane + doesn't own. + enum: + - Managed + - Provided gke: type: object description: >- diff --git a/apis/servingstacks/definition.yaml b/apis/servingstacks/definition.yaml index 65722e89a..68d78cc50 100644 --- a/apis/servingstacks/definition.yaml +++ b/apis/servingstacks/definition.yaml @@ -45,7 +45,25 @@ spec: x-kubernetes-validations: - rule: "self.secrets.exists(s, s.type == 'Kubeconfig')" message: spec.secrets must include a Kubeconfig entry. + - rule: "!has(self.components) || self.components != 'Provided' || self.cloud == 'Existing'" + message: spec.components Provided is only supported when cloud is Existing. properties: + components: + type: string + default: Managed + description: >- + Who supplies the serving substrate. Managed (the default) + has Modelplane install every component. Provided, valid + only on an Existing cluster, installs no substrate charts + or vendored CRDs: Modelplane verifies the cluster supplies + them - reported through the RequirementsMet condition - and + composes only its own configuration on top. All or + nothing. Mirrors + InferenceCluster.spec.cluster.existing.components; the + cluster composition sets it. + enum: + - Managed + - Provided cloud: type: string description: >- diff --git a/docs/content/platform/inference-cluster.md b/docs/content/platform/inference-cluster.md index 652ade6e9..e7708b2ea 100644 --- a/docs/content/platform/inference-cluster.md +++ b/docs/content/platform/inference-cluster.md @@ -50,6 +50,11 @@ The `cluster.source` discriminator picks one of two models: provision infrastructure, and each pool's `InferenceClass` provides hardware capabilities for scheduling only. You're responsible for the cluster meeting [Modelplane's requirements](#requirements-for-an-existing-cluster). + If the cluster already runs the serving substrate - cert-manager, a gateway + stack, Prometheus - set `existing.components: Provided` and Modelplane + installs none of it, verifying the cluster meets the + [serving stack requirements]({{< ref "/platform/serving-stack-requirements.md" >}}) + instead and reporting what's missing through the `RequirementsMet` condition. ## Requirements for an existing cluster @@ -77,12 +82,19 @@ An existing cluster must meet what Modelplane would otherwise set up for you: address. - **No conflicting Gateway controller.** Modelplane installs Envoy Gateway and owns its `GatewayClass`. Don't run another controller claiming the same class. + With `components: Provided` this inverts: the cluster runs its own Envoy + Gateway, which serves the `GatewayClass` Modelplane composes. - **A `ReadWriteMany` StorageClass**, if you use a `ModelCache`. See [Cache storage](#cache-storage). - **Any multi-node fabric you need.** For multi-node serving you provide and configure the RDMA or InfiniBand fabric and its drivers. Modelplane installs those only on the clouds it provisions. +With `components: Provided` the cluster provides the serving stack itself on +top of all this; the +[serving stack requirements]({{< ref "/platform/serving-stack-requirements.md" >}}) +page lists what that adds, per component. + ## Serving stack `spec.stack` selects the serving layer the cluster runs: `Standard` (the default) diff --git a/docs/content/platform/serving-stack-requirements.md b/docs/content/platform/serving-stack-requirements.md new file mode 100644 index 000000000..164eeaa34 --- /dev/null +++ b/docs/content/platform/serving-stack-requirements.md @@ -0,0 +1,238 @@ +--- +title: Serving Stack Requirements +weight: 35 +description: What an existing cluster must provide to run the serving stack itself. +--- + +An `InferenceCluster` with `spec.cluster.existing.components: Provided` +installs no serving stack components. The cluster provides the whole +substrate itself, and Modelplane only verifies it and composes its own +configuration on top. This page lists what that cluster must provide, +per component Modelplane would otherwise install. It applies on top of +the general [requirements for an existing +cluster]({{< ref "/platform/inference-cluster.md" >}}). + +Modelplane checks the **checked** entries continuously and reports what +is missing through the `RequirementsMet` condition on the +`InferenceCluster`. A check verifies presence and served API versions, +not the installed release: the **not checked** entries, including +component versions, are yours to meet. The version listed per component +is the one Modelplane installs in Managed mode and tests against; stay +close to it. + +## On every stack + +### `cert-manager` + +Managed mode installs chart `cert-manager` `v1.20.2`. + +Checked: + +- CRD `certificates.cert-manager.io` serving `v1` + +Not checked: + +- The cert-manager controller and webhook are running and issue Certificates. + +### `kube-prometheus-stack` + +Managed mode installs chart `kube-prometheus-stack` `84.4.0`. + +Checked: + +- CRD `podmonitors.monitoring.coreos.com` serving `v1` +- CRD `servicemonitors.monitoring.coreos.com` serving `v1` + +Not checked: + +- Prometheus discovers `PodMonitor` objects in every namespace: with the chart, set `podMonitorSelectorNilUsesHelmValues` to false and `podMonitorNamespaceSelector` to empty. Modelplane's scrape targets are `PodMonitor` objects in workload namespaces, and the chart's default release label selector never matches them. +- A scrape job for the Envoy Gateway proxy pods' stats endpoint, if you want request metrics at the proxy level. + +Not needed: + +- Modelplane disables Grafana and Alertmanager. + +### `node-feature-discovery` + +Managed mode installs chart `node-feature-discovery` `0.19.0`. + +Checked: + +- CRD `nodefeatures.nfd.k8s-sigs.io` serving `v1alpha1` + +Not checked: + +- The NFD worker runs on the GPU nodes and labels them with `feature.node.kubernetes.io/pci-10de` and friends. If GPU nodes are tainted, the worker must tolerate the taint, or the DRA driver's `kubelet` plugin never schedules there and every GPU ResourceClaim stays pending with all components looking healthy. + +### `nvidia-dra-driver-gpu` + +Managed mode installs chart `dra-driver-nvidia-gpu` `0.4.1`. + +Checked: + +- `DeviceClass` `gpu.nvidia.com` (`resource.k8s.io/v1`) + +Not checked: + +- The NVIDIA kernel driver and Container Toolkit on every GPU node, from the node image or the GPU Operator. The requirements for an existing cluster state the versions. +- The driver's `kubelet` plugin publishes each GPU node's devices as ResourceSlices. +- No device plugin advertising `nvidia.com/gpu`: a second allocator would hand out the same GPUs behind DRA's back. + +Not needed: + +- Modelplane disables `ComputeDomains` (multi-node NVLink) and their prerequisites. + +### `envoy-gateway` + +Managed mode installs chart `gateway-helm` `v1.8.4`. + +Checked: + +- CRD `gatewayclasses.gateway.networking.k8s.io` serving `v1` +- CRD `gateways.gateway.networking.k8s.io` serving `v1` +- CRD `httproutes.gateway.networking.k8s.io` serving `v1` +- CRD `envoyproxies.gateway.envoyproxy.io` serving `v1alpha1` +- CRD `backends.gateway.envoyproxy.io` serving `v1alpha1` +- CRD `clienttrafficpolicies.gateway.envoyproxy.io` serving `v1alpha1` + +Not checked: + +- The Envoy Gateway controller runs with the `extensionManager` wired exactly as the values above: external processing delegated to the AI Gateway controller's Service, with the Backend API enabled and InferencePool declared a backend resource. Without it, HTTPRoute to InferencePool `backendRefs` never route, with every component looking healthy. + +The exact values Modelplane installs the chart with. The `extensionManager` wiring is the one coupling no check can verify. Without it, routes to an `InferencePool` never route while every component looks healthy: + +```yaml +config: + envoyGateway: + extensionApis: + enableBackend: true + extensionManager: + hooks: + xdsTranslator: + translation: + listener: + includeAll: true + route: + includeAll: true + cluster: + includeAll: true + secret: + includeAll: true + post: + - Translation + - Cluster + - Route + service: + fqdn: + hostname: ai-gateway-controller.envoy-ai-gateway-system.svc.cluster.local + port: 1063 + backendResources: + - group: inference.networking.k8s.io + kind: InferencePool + version: v1 +``` + +### `ai-gateway-crds` + +Managed mode installs chart `ai-gateway-crds-helm` `v1.1.0`. + +Checked: + +- CRD `aigatewayroutes.aigateway.envoyproxy.io` serving `v1alpha1` + +Not checked: + +- The AI Gateway APIs are v1alpha1 and move with the controller. A provided install tracks the pinned v1.1.0 release. + +### `ai-gateway` + +Managed mode installs chart `ai-gateway-helm` `v1.1.0`. + +Not checked: + +- The AI Gateway controller is reachable at `ai-gateway-controller.envoy-ai-gateway-system.svc.cluster.local:1063`, the address Modelplane's Envoy Gateway `extensionManager` values point at. A controller installed elsewhere never receives the external processing traffic. + +### `gaie-crds` + +Managed mode applies manifests Modelplane vendors from the upstream release. + +Checked: + +- CRD `inferencepools.inference.networking.k8s.io` serving `v1` + +### `trust-manager` + +Managed mode installs chart `trust-manager` `v0.25.0`. + +Checked: + +- CRD `bundles.trust.cert-manager.io` serving `v1alpha1` + +Not checked: + +- trust-manager watches modelplane-system as its trust namespace, where the gateway PKI composes its Bundle. An install watching another namespace never syncs the Bundle, and the control plane can't read the cluster CA. + +### `dra-driver-critical-pods-quota` + +Managed mode applies manifests Modelplane vendors from the upstream release. + +Not checked: + +- On clusters that restrict `system-node-critical` pods by namespace quota (GKE does), the DRA driver's namespace needs a `ResourceQuota` admitting them, or the `kubelet` plugin `DaemonSet` never starts. + +## Standard stack + +### `leader-worker-set` + +Managed mode installs chart `lws` `v0.8.0`. + +Checked: + +- CRD `leaderworkersets.leaderworkerset.x-k8s.io` serving `v1` + +Not checked: + +- The LeaderWorkerSet controller is running. Modelplane composes against the v0.8 line. + +## Dynamo stack + +### `grove` + +Managed mode installs chart `grove-charts` `v0.1.0-alpha.12-rc2`. + +Checked: + +- CRD `podcliquesets.grove.io` serving `v1alpha1` + +Not checked: + +- Grove's API is v1alpha1 with no compatibility promise between versions, so a provided install must run the exact pinned version. A CRD check can't tell alpha revisions apart. Prefer Managed for the Dynamo stack until Grove stabilizes. + +### `kai-scheduler` + +Managed mode installs chart `kai-scheduler` `v0.16.8`. + +Checked: + +- CRD `queues.scheduling.run.ai` serving `v2` + +Not checked: + +- The scheduler answers to the `schedulerName` value `kai-scheduler`, the name Modelplane's engine pods request. Its Queue webhook must be serving. Modelplane composes against the v0.16 line. + +## What Modelplane still installs + +Provided mode only skips the substrate. Modelplane still composes its +own configuration and workloads: the `modelplane-system` namespace, the +`EnvoyProxy`, `GatewayClass` and `Gateway` for the inference gateway, +and on the Dynamo stack its KAI `Queue` hierarchy and the ModelExpress +server with its CRDs. Their substrate dependencies gate on the checks +above, so none of them is applied before the cluster serves the APIs +they need. + +During deletion, Modelplane can't order its configuration ahead of a +substrate it doesn't own. Keep your controllers (the gateway +controller, KAI) running while an `InferenceCluster` deletes, so they +can process finalizers on Modelplane's configuration. diff --git a/docs/manifests/concepts/inference-cluster-provided.yaml b/docs/manifests/concepts/inference-cluster-provided.yaml new file mode 100644 index 000000000..27f746aa0 --- /dev/null +++ b/docs/manifests/concepts/inference-cluster-provided.yaml @@ -0,0 +1,28 @@ +# An InferenceCluster on an existing cluster that already runs the +# serving substrate: cert-manager, the Envoy and AI gateway stack, +# kube-prometheus-stack, the NVIDIA DRA driver, and the stack's +# workload controller. +# +# components: Provided keeps Modelplane from installing any of them, so +# nothing conflicts with the cluster's own installs. Modelplane checks +# the cluster supplies what the stack needs and reports what's missing +# through the RequirementsMet condition; see the serving stack +# requirements page for the full list. +apiVersion: modelplane.ai/v1alpha1 +kind: InferenceCluster +metadata: + name: byo-provided + labels: + modelplane.ai/region: us-east +spec: + cluster: + source: Existing + existing: + components: Provided + secretRef: + name: byo-cluster-kubeconfig + key: kubeconfig + nodePools: + - name: gpu-h100 + className: h100-8x-byo + nodeCount: 2 diff --git a/docs/utils/vale/styles/config/vocabularies/Modelplane/accept.txt b/docs/utils/vale/styles/config/vocabularies/Modelplane/accept.txt index c0efe996f..0138ca77a 100644 --- a/docs/utils/vale/styles/config/vocabularies/Modelplane/accept.txt +++ b/docs/utils/vale/styles/config/vocabularies/Modelplane/accept.txt @@ -118,6 +118,10 @@ NIXL NixlConnector GPUDirect Prometheus +Grafana +Alertmanager +NFD +CDI ArgoCD FluxCD GitHub diff --git a/e2e/README.md b/e2e/README.md index d8bac4ee5..afd16c8e7 100644 --- a/e2e/README.md +++ b/e2e/README.md @@ -97,6 +97,13 @@ nix run .#e2e -- --verify # same, then wait for readiness and assert a live 200 nix run .#e2e -- --clean # tear both clusters down ``` +`--provided` (combinable with `--verify`) registers the workload cluster with +`components: Provided`: the substrate is pre-installed from the generated +`e2e/provided/` inputs under non-`mp-` release names, one chart is held back to +assert the `RequirementsMet` condition names it and flips once installed, and +Modelplane must compose no Helm release of its own. The `E2E` workflow runs it +under the `test-e2e-provided` label. + `crossplane project run` installs the config and applies the resources, then returns; the serving-stack install and model rollout reconcile in the background. So wait for the `ModelService` to become ready before curling. The gateway's diff --git a/e2e/provided/charts.tsv b/e2e/provided/charts.tsv new file mode 100644 index 000000000..e56c010d9 --- /dev/null +++ b/e2e/provided/charts.tsv @@ -0,0 +1,12 @@ +# Generated by functions/compose-serving-stack/requirements_doc.py +# from the serving stack component lists. Do not edit; run +# `nix run .#requirements-doc` after changing the stack data. +cert-manager cert-manager https://charts.jetstack.io v1.20.2 cert-manager true +kube-prometheus-stack kube-prometheus-stack https://prometheus-community.github.io/helm-charts 84.4.0 monitoring false +node-feature-discovery node-feature-discovery https://kubernetes-sigs.github.io/node-feature-discovery/charts 0.19.0 node-feature-discovery false +nvidia-dra-driver-gpu dra-driver-nvidia-gpu oci://registry.k8s.io/dra-driver-nvidia/charts 0.4.1 nvidia-dra-driver false +envoy-gateway gateway-helm oci://docker.io/envoyproxy v1.8.4 envoy-gateway-system true +ai-gateway-crds ai-gateway-crds-helm oci://docker.io/envoyproxy v1.1.0 envoy-ai-gateway-system true +ai-gateway ai-gateway-helm oci://docker.io/envoyproxy v1.1.0 envoy-ai-gateway-system false +trust-manager trust-manager oci://quay.io/jetstack/charts v0.25.0 modelplane-system false +leader-worker-set lws oci://registry.k8s.io/lws/charts v0.8.0 lws-system false diff --git a/e2e/provided/manifests.yaml b/e2e/provided/manifests.yaml new file mode 100644 index 000000000..dde37a1b5 --- /dev/null +++ b/e2e/provided/manifests.yaml @@ -0,0 +1,1019 @@ +# Generated by functions/compose-serving-stack/requirements_doc.py +# from the serving stack component lists. Do not edit; run +# `nix run .#requirements-doc` after changing the stack data. +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + inference.networking.k8s.io/bundle-version: v1.0.2 + name: inferenceobjectives.inference.networking.x-k8s.io +spec: + group: inference.networking.x-k8s.io + names: + kind: InferenceObjective + listKind: InferenceObjectiveList + plural: inferenceobjectives + singular: inferenceobjective + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.poolRef.name + name: Inference Pool + type: string + - jsonPath: .spec.priority + name: Priority + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: InferenceObjective is the Schema for the InferenceObjectives + API. + properties: + apiVersion: + description: 'APIVersion defines the versioned schema of this representation + of an object. + + Servers should convert recognized schemas to the latest internal value, + and + + may reject unrecognized values. + + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources' + type: string + kind: + description: 'Kind is a string value representing the REST resource this + object represents. + + Servers may infer this from the endpoint the client submits requests + to. + + Cannot be updated. + + In CamelCase. + + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds' + type: string + metadata: + type: object + spec: + description: 'InferenceObjectiveSpec represents the desired state of a + specific model use case. This resource is + + managed by the "Inference Workload Owner" persona. + + + The Inference Workload Owner persona is someone that trains, verifies, + and + + leverages a large language model from a model frontend, drives the lifecycle + + and rollout of new versions of those models, and defines the specific + + performance and latency goals for the model. These workloads are + + expected to operate within an InferencePool sharing compute capacity + with other + + InferenceObjectives, defined by the Inference Platform Admin.' + properties: + poolRef: + description: PoolRef is a reference to the inference pool, the pool + must exist in the same namespace. + properties: + group: + default: inference.networking.k8s.io + description: Group is the group of the referent. + maxLength: 253 + pattern: ^$|^[a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*$ + type: string + kind: + default: InferencePool + description: Kind is kind of the referent. For example "InferencePool". + maxLength: 63 + minLength: 1 + pattern: ^[a-zA-Z]([-a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + name: + description: Name is the name of the referent. + maxLength: 253 + minLength: 1 + type: string + required: + - name + type: object + priority: + description: 'Priority defines how important it is to serve the request + compared to other requests in the same pool. + + Priority is an integer value that defines the priority of the request. + + The higher the value, the more critical the request is; negative + values _are_ allowed. + + No default value is set for this field, allowing for future additions + of new fields that may ''one of'' with this field. + + However, implementations that consume this field (such as the Endpoint + Picker) will treat an unset value as ''0''. + + Priority is used in flow control, primarily in the event of resource + scarcity(requests need to be queued). + + All requests will be queued, and flow control will _always_ allow + requests of higher priority to be served first. + + Fairness is only enforced and tracked between requests of the same + priority. + + + Example: requests with Priority 10 will always be served before + + requests with Priority of 0 (the value used if Priority is unset + or no InfereneceObjective is specified). + + Similarly requests with a Priority of -10 will always be served + after requests with Priority of 0.' + type: integer + required: + - poolRef + type: object + status: + description: InferenceObjectiveStatus defines the observed state of InferenceObjective + properties: + conditions: + default: + - lastTransitionTime: '1970-01-01T00:00:00Z' + message: Waiting for controller + reason: Pending + status: Unknown + type: Ready + description: 'Conditions track the state of the InferenceObjective. + + + Known condition types are: + + + * "Accepted"' + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: 'lastTransitionTime is the last time the condition + transitioned from one status to another. + + This should be when the underlying condition changed. If + that is not known, then using the time when the API field + changed is acceptable.' + format: date-time + type: string + message: + description: 'message is a human readable message indicating + details about the transition. + + This may be an empty string.' + maxLength: 32768 + type: string + observedGeneration: + description: 'observedGeneration represents the .metadata.generation + that the condition was set based upon. + + For instance, if .metadata.generation is currently 12, but + the .status.conditions[x].observedGeneration is 9, the condition + is out of date + + with respect to the current state of the instance.' + format: int64 + minimum: 0 + type: integer + reason: + description: 'reason contains a programmatic identifier indicating + the reason for the condition''s last transition. + + Producers of specific condition types may define expected + values and meanings for this field, + + and whether the values are considered a guaranteed API. + + The value should be a CamelCase string. + + This field may not be empty.' + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - 'True' + - 'False' + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + maxItems: 8 + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + type: object + type: object + served: true + storage: true + subresources: + status: {} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + api-approved.kubernetes.io: https://github.com/kubernetes-sigs/gateway-api-inference-extension/pull/1173 + inference.networking.k8s.io/bundle-version: v1.0.2 + name: inferencepools.inference.networking.k8s.io +spec: + group: inference.networking.k8s.io + names: + kind: InferencePool + listKind: InferencePoolList + plural: inferencepools + shortNames: + - infpool + singular: inferencepool + scope: Namespaced + versions: + - name: v1 + schema: + openAPIV3Schema: + description: 'InferencePool is the Schema for the InferencePools API. + + ' + properties: + apiVersion: + description: 'APIVersion defines the versioned schema of this representation + of an object. + + Servers should convert recognized schemas to the latest internal value, + and + + may reject unrecognized values. + + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources' + type: string + kind: + description: 'Kind is a string value representing the REST resource this + object represents. + + Servers may infer this from the endpoint the client submits requests + to. + + Cannot be updated. + + In CamelCase. + + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds' + type: string + metadata: + type: object + spec: + description: Spec defines the desired state of the InferencePool. + properties: + endpointPickerRef: + description: 'EndpointPickerRef is a reference to the Endpoint Picker + extension and its + + associated configuration.' + properties: + failureMode: + default: FailClose + description: 'FailureMode configures how the parent handles the + case when the Endpoint Picker extension + + is non-responsive. When unspecified, defaults to "FailClose".' + enum: + - FailOpen + - FailClose + type: string + group: + default: '' + description: 'Group is the group of the referent API object. When + unspecified, the default value + + is "", representing the Core API group.' + maxLength: 253 + minLength: 0 + pattern: ^$|^[a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*$ + type: string + kind: + default: Service + description: 'Kind is the Kubernetes resource kind of the referent. + + + Required if the referent is ambiguous, e.g. service with multiple + ports. + + + Defaults to "Service" when not specified. + + + ExternalName services can refer to CNAME DNS records that may + live + + outside of the cluster and as such are difficult to reason about + in + + terms of conformance. They also may not be safe to forward to + (see + + CVE-2021-25740 for more information). Implementations MUST NOT + + support ExternalName Services.' + maxLength: 63 + minLength: 1 + pattern: ^[a-zA-Z]([-a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + name: + description: Name is the name of the referent API object. + maxLength: 253 + minLength: 1 + type: string + port: + description: 'Port is the port of the Endpoint Picker extension + service. + + + Port is required when the referent is a Kubernetes Service. + In this + + case, the port number is the service port number, not the target + port. + + For other resources, destination port might be derived from + the referent + + resource or this field.' + properties: + number: + description: 'Number defines the port number to access the + selected model server Pods. + + The number must be in the range 1 to 65535.' + format: int32 + maximum: 65535 + minimum: 1 + type: integer + required: + - number + type: object + required: + - name + type: object + x-kubernetes-validations: + - message: port is required when kind is 'Service' or unspecified + (defaults to 'Service') + rule: self.kind != 'Service' || has(self.port) + selector: + description: 'Selector determines which Pods are members of this inference + pool. + + It matches Pods by their labels only within the same namespace; + cross-namespace + + selection is not supported. + + + The structure of this LabelSelector is intentionally simple to be + compatible + + with Kubernetes Service selectors, as some implementations may translate + + this configuration into a Service resource.' + properties: + matchLabels: + additionalProperties: + description: 'LabelValue is the value of a label. This is used + for validation + + of maps. This matches the Kubernetes label validation rules: + + * must be 63 characters or less (can be empty), + + * unless empty, must begin and end with an alphanumeric character + ([a-z0-9A-Z]), + + * could contain dashes (-), underscores (_), dots (.), and + alphanumerics between. + + + Valid values include: + + + * MyValue + + * my.name + + * 123-my-value' + maxLength: 63 + minLength: 0 + pattern: ^(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])?$ + type: string + description: 'MatchLabels contains a set of required {key,value} + pairs. + + An object must match every label in this map to be selected. + + The matching logic is an AND operation on all entries.' + maxProperties: 64 + minProperties: 1 + type: object + required: + - matchLabels + type: object + targetPorts: + description: 'TargetPorts defines a list of ports that are exposed + by this InferencePool. + + Currently, the list may only include a single port definition.' + items: + description: Port defines the network port that will be exposed + by this InferencePool. + properties: + number: + description: 'Number defines the port number to access the selected + model server Pods. + + The number must be in the range 1 to 65535.' + format: int32 + maximum: 65535 + minimum: 1 + type: integer + required: + - number + type: object + maxItems: 1 + minItems: 1 + type: array + x-kubernetes-list-type: atomic + required: + - endpointPickerRef + - selector + - targetPorts + type: object + status: + description: Status defines the observed state of the InferencePool. + properties: + parents: + description: 'Parents is a list of parent resources, typically Gateways, + that are associated with + + the InferencePool, and the status of the InferencePool with respect + to each parent. + + + A controller that manages the InferencePool, must add an entry for + each parent it manages + + and remove the parent entry when the controller no longer considers + the InferencePool to + + be associated with that parent. + + + A maximum of 32 parents will be represented in this list. When the + list is empty, + + it indicates that the InferencePool is not associated with any parents.' + items: + description: ParentStatus defines the observed state of InferencePool + from a Parent, i.e. Gateway. + properties: + conditions: + description: 'Conditions is a list of status conditions that + provide information about the observed + + state of the InferencePool. This field is required to be set + by the controller that + + manages the InferencePool. + + + Supported condition types are: + + + * "Accepted" + + * "ResolvedRefs"' + items: + description: Condition contains details for one aspect of + the current state of this API Resource. + properties: + lastTransitionTime: + description: 'lastTransitionTime is the last time the + condition transitioned from one status to another. + + This should be when the underlying condition changed. If + that is not known, then using the time when the API + field changed is acceptable.' + format: date-time + type: string + message: + description: 'message is a human readable message indicating + details about the transition. + + This may be an empty string.' + maxLength: 32768 + type: string + observedGeneration: + description: 'observedGeneration represents the .metadata.generation + that the condition was set based upon. + + For instance, if .metadata.generation is currently 12, + but the .status.conditions[x].observedGeneration is + 9, the condition is out of date + + with respect to the current state of the instance.' + format: int64 + minimum: 0 + type: integer + reason: + description: 'reason contains a programmatic identifier + indicating the reason for the condition''s last transition. + + Producers of specific condition types may define expected + values and meanings for this field, + + and whether the values are considered a guaranteed API. + + The value should be a CamelCase string. + + This field may not be empty.' + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, + Unknown. + enum: + - 'True' + - 'False' + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + maxItems: 8 + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + parentRef: + description: 'ParentRef is used to identify the parent resource + that this status + + is associated with. It is used to match the InferencePool + with the parent + + resource, such as a Gateway.' + properties: + group: + default: gateway.networking.k8s.io + description: 'Group is the group of the referent API object. + When unspecified, the referent is assumed + + to be in the "gateway.networking.k8s.io" API group.' + maxLength: 253 + minLength: 0 + pattern: ^$|^[a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*$ + type: string + kind: + default: Gateway + description: 'Kind is the kind of the referent API object. + When unspecified, the referent is assumed + + to be a "Gateway" kind.' + maxLength: 63 + minLength: 1 + pattern: ^[a-zA-Z]([-a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + name: + description: Name is the name of the referent API object. + maxLength: 253 + minLength: 1 + type: string + namespace: + description: 'Namespace is the namespace of the referenced + object. When unspecified, the local + + namespace is inferred. + + + Note that when a namespace different than the local namespace + is specified, + + a ReferenceGrant object is required in the referent namespace + to allow that + + namespace''s owner to accept the reference. See the ReferenceGrant + + documentation for details: https://gateway-api.sigs.k8s.io/api-types/referencegrant/' + maxLength: 63 + minLength: 1 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + required: + - name + type: object + required: + - parentRef + type: object + maxItems: 32 + type: array + x-kubernetes-list-type: atomic + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + api-approved.kubernetes.io: unapproved, experimental-only + inference.networking.k8s.io/bundle-version: v1.0.2 + name: inferencepools.inference.networking.x-k8s.io +spec: + group: inference.networking.x-k8s.io + names: + kind: InferencePool + listKind: InferencePoolList + plural: inferencepools + shortNames: + - xinfpool + singular: inferencepool + scope: Namespaced + versions: + - name: v1alpha2 + schema: + openAPIV3Schema: + description: InferencePool is the Schema for the InferencePools API. + properties: + apiVersion: + description: 'APIVersion defines the versioned schema of this representation + of an object. + + Servers should convert recognized schemas to the latest internal value, + and + + may reject unrecognized values. + + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources' + type: string + kind: + description: 'Kind is a string value representing the REST resource this + object represents. + + Servers may infer this from the endpoint the client submits requests + to. + + Cannot be updated. + + In CamelCase. + + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds' + type: string + metadata: + type: object + spec: + description: InferencePoolSpec defines the desired state of InferencePool + properties: + extensionRef: + description: Extension configures an endpoint picker as an extension + service. + properties: + failureMode: + default: FailClose + description: 'Configures how the gateway handles the case when + the extension is not responsive. + + Defaults to failClose.' + enum: + - FailOpen + - FailClose + type: string + group: + default: '' + description: 'Group is the group of the referent. + + The default value is "", representing the Core API group.' + maxLength: 253 + pattern: ^$|^[a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*$ + type: string + kind: + default: Service + description: 'Kind is the Kubernetes resource kind of the referent. + + + Defaults to "Service" when not specified. + + + ExternalName services can refer to CNAME DNS records that may + live + + outside of the cluster and as such are difficult to reason about + in + + terms of conformance. They also may not be safe to forward to + (see + + CVE-2021-25740 for more information). Implementations MUST NOT + + support ExternalName Services.' + maxLength: 63 + minLength: 1 + pattern: ^[a-zA-Z]([-a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + name: + description: Name is the name of the referent. + maxLength: 253 + minLength: 1 + type: string + portNumber: + description: 'The port number on the service running the extension. + When unspecified, + + implementations SHOULD infer a default value of 9002 when the + Kind is + + Service.' + format: int32 + maximum: 65535 + minimum: 1 + type: integer + required: + - name + type: object + selector: + additionalProperties: + description: 'LabelValue is the value of a label. This is used for + validation + + of maps. This matches the Kubernetes label validation rules: + + * must be 63 characters or less (can be empty), + + * unless empty, must begin and end with an alphanumeric character + ([a-z0-9A-Z]), + + * could contain dashes (-), underscores (_), dots (.), and alphanumerics + between. + + + Valid values include: + + + * MyValue + + * my.name + + * 123-my-value' + maxLength: 63 + minLength: 0 + pattern: ^(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])?$ + type: string + description: 'Selector defines a map of labels to watch model server + Pods + + that should be included in the InferencePool. + + In some cases, implementations may translate this field to a Service + selector, so this matches the simple + + map used for Service selectors instead of the full Kubernetes LabelSelector + type. + + If specified, it will be applied to match the model server pods + in the same namespace as the InferencePool. + + Cross namesoace selector is not supported.' + type: object + targetPortNumber: + description: 'TargetPortNumber defines the port number to access the + selected model server Pods. + + The number must be in the range 1 to 65535.' + format: int32 + maximum: 65535 + minimum: 1 + type: integer + required: + - extensionRef + - selector + - targetPortNumber + type: object + status: + default: + parent: + - conditions: + - lastTransitionTime: '1970-01-01T00:00:00Z' + message: Waiting for controller + reason: Pending + status: Unknown + type: Accepted + parentRef: + kind: Status + name: default + description: Status defines the observed state of InferencePool. + properties: + parent: + description: "Parents is a list of parent resources (usually Gateways)\ + \ that are\nassociated with the InferencePool, and the status of\ + \ the InferencePool with respect to\neach parent.\n\nA maximum of\ + \ 32 Gateways will be represented in this list. When the list contains\n\ + `kind: Status, name: default`, it indicates that the InferencePool\ + \ is not\nassociated with any Gateway and a controller must perform\ + \ the following:\n\n - Remove the parent when setting the \"Accepted\"\ + \ condition.\n - Add the parent when the controller will no longer\ + \ manage the InferencePool\n and no other parents exist." + items: + description: PoolStatus defines the observed state of InferencePool + from a Gateway. + properties: + conditions: + default: + - lastTransitionTime: '1970-01-01T00:00:00Z' + message: Waiting for controller + reason: Pending + status: Unknown + type: Accepted + description: 'Conditions track the state of the InferencePool. + + + Known condition types are: + + + * "Accepted" + + * "ResolvedRefs"' + items: + description: Condition contains details for one aspect of + the current state of this API Resource. + properties: + lastTransitionTime: + description: 'lastTransitionTime is the last time the + condition transitioned from one status to another. + + This should be when the underlying condition changed. If + that is not known, then using the time when the API + field changed is acceptable.' + format: date-time + type: string + message: + description: 'message is a human readable message indicating + details about the transition. + + This may be an empty string.' + maxLength: 32768 + type: string + observedGeneration: + description: 'observedGeneration represents the .metadata.generation + that the condition was set based upon. + + For instance, if .metadata.generation is currently 12, + but the .status.conditions[x].observedGeneration is + 9, the condition is out of date + + with respect to the current state of the instance.' + format: int64 + minimum: 0 + type: integer + reason: + description: 'reason contains a programmatic identifier + indicating the reason for the condition''s last transition. + + Producers of specific condition types may define expected + values and meanings for this field, + + and whether the values are considered a guaranteed API. + + The value should be a CamelCase string. + + This field may not be empty.' + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, + Unknown. + enum: + - 'True' + - 'False' + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + maxItems: 8 + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + parentRef: + description: GatewayRef indicates the gateway that observed + state of InferencePool. + properties: + group: + default: gateway.networking.k8s.io + description: Group is the group of the referent. + maxLength: 253 + pattern: ^$|^[a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*$ + type: string + kind: + default: Gateway + description: Kind is kind of the referent. For example "Gateway". + maxLength: 63 + minLength: 1 + pattern: ^[a-zA-Z]([-a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + name: + description: Name is the name of the referent. + maxLength: 253 + minLength: 1 + type: string + namespace: + description: 'Namespace is the namespace of the referent. If + not present, + + the namespace of the referent is assumed to be the same + as + + the namespace of the referring object.' + maxLength: 63 + minLength: 1 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + required: + - name + type: object + required: + - parentRef + type: object + maxItems: 32 + type: array + type: object + type: object + served: true + storage: true + subresources: + status: {} +--- +apiVersion: v1 +kind: ResourceQuota +metadata: + name: allow-critical-pods + namespace: nvidia-dra-driver +spec: + hard: + pods: '1000' + scopeSelector: + matchExpressions: + - operator: In + scopeName: PriorityClass + values: + - system-node-critical + - system-cluster-critical diff --git a/e2e/provided/values/ai-gateway.yaml b/e2e/provided/values/ai-gateway.yaml new file mode 100644 index 000000000..91943f29c --- /dev/null +++ b/e2e/provided/values/ai-gateway.yaml @@ -0,0 +1,5 @@ +# Generated by functions/compose-serving-stack/requirements_doc.py +# from the serving stack component lists. Do not edit; run +# `nix run .#requirements-doc` after changing the stack data. +controller: + logRequestHeaderAttributes: x-modelplane-caller:caller diff --git a/e2e/provided/values/cert-manager.yaml b/e2e/provided/values/cert-manager.yaml new file mode 100644 index 000000000..434464e70 --- /dev/null +++ b/e2e/provided/values/cert-manager.yaml @@ -0,0 +1,5 @@ +# Generated by functions/compose-serving-stack/requirements_doc.py +# from the serving stack component lists. Do not edit; run +# `nix run .#requirements-doc` after changing the stack data. +crds: + enabled: true diff --git a/e2e/provided/values/envoy-gateway.yaml b/e2e/provided/values/envoy-gateway.yaml new file mode 100644 index 000000000..f83555b1c --- /dev/null +++ b/e2e/provided/values/envoy-gateway.yaml @@ -0,0 +1,31 @@ +# Generated by functions/compose-serving-stack/requirements_doc.py +# from the serving stack component lists. Do not edit; run +# `nix run .#requirements-doc` after changing the stack data. +config: + envoyGateway: + extensionApis: + enableBackend: true + extensionManager: + hooks: + xdsTranslator: + translation: + listener: + includeAll: true + route: + includeAll: true + cluster: + includeAll: true + secret: + includeAll: true + post: + - Translation + - Cluster + - Route + service: + fqdn: + hostname: ai-gateway-controller.envoy-ai-gateway-system.svc.cluster.local + port: 1063 + backendResources: + - group: inference.networking.k8s.io + kind: InferencePool + version: v1 diff --git a/e2e/provided/values/kube-prometheus-stack.yaml b/e2e/provided/values/kube-prometheus-stack.yaml new file mode 100644 index 000000000..482272dde --- /dev/null +++ b/e2e/provided/values/kube-prometheus-stack.yaml @@ -0,0 +1,31 @@ +# Generated by functions/compose-serving-stack/requirements_doc.py +# from the serving stack component lists. Do not edit; run +# `nix run .#requirements-doc` after changing the stack data. +fullnameOverride: prometheus +prometheus: + prometheusSpec: + podMonitorSelectorNilUsesHelmValues: false + podMonitorNamespaceSelector: {} + additionalScrapeConfigs: + - job_name: envoy-gateway-proxy + kubernetes_sd_configs: + - role: pod + namespaces: + names: + - envoy-gateway-system + relabel_configs: + - source_labels: + - __meta_kubernetes_pod_label_app_kubernetes_io_component + action: keep + regex: proxy + - source_labels: + - __address__ + action: replace + regex: ([^:]+)(?::\d+)? + replacement: $1:19001 + target_label: __address__ + metrics_path: /stats/prometheus +grafana: + enabled: false +alertmanager: + enabled: false diff --git a/e2e/provided/values/node-feature-discovery.yaml b/e2e/provided/values/node-feature-discovery.yaml new file mode 100644 index 000000000..51e3fe4dd --- /dev/null +++ b/e2e/provided/values/node-feature-discovery.yaml @@ -0,0 +1,8 @@ +# Generated by functions/compose-serving-stack/requirements_doc.py +# from the serving stack component lists. Do not edit; run +# `nix run .#requirements-doc` after changing the stack data. +worker: + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule diff --git a/e2e/provided/values/nvidia-dra-driver-gpu.yaml b/e2e/provided/values/nvidia-dra-driver-gpu.yaml new file mode 100644 index 000000000..dd37d3609 --- /dev/null +++ b/e2e/provided/values/nvidia-dra-driver-gpu.yaml @@ -0,0 +1,7 @@ +# Generated by functions/compose-serving-stack/requirements_doc.py +# from the serving stack component lists. Do not edit; run +# `nix run .#requirements-doc` after changing the stack data. +gpuResourcesEnabledOverride: true +resources: + computeDomains: + enabled: false diff --git a/e2e/provided/values/trust-manager.yaml b/e2e/provided/values/trust-manager.yaml new file mode 100644 index 000000000..d46b06e83 --- /dev/null +++ b/e2e/provided/values/trust-manager.yaml @@ -0,0 +1,11 @@ +# Generated by functions/compose-serving-stack/requirements_doc.py +# from the serving stack component lists. Do not edit; run +# `nix run .#requirements-doc` after changing the stack data. +crds: + enabled: true + keep: true +app: + trust: + namespace: modelplane-system +defaultPackage: + enabled: false diff --git a/e2e/run.sh b/e2e/run.sh index 84662ebbb..1f07b5a37 100644 --- a/e2e/run.sh +++ b/e2e/run.sh @@ -100,6 +100,57 @@ metadata: { name: kind-l2, namespace: metallb-system } spec: { ipAddressPools: [kind-pool] } POOL +apply_manifests=1 +verify=0 +provided=0 +for arg in "$@"; do + case "$arg" in + --no-apply) apply_manifests=0 ;; + --verify) verify=1 ;; + --provided) provided=1 ;; + esac +done + +# Provided mode pre-installs the serving substrate on the workload +# cluster from the generated e2e/provided/ inputs — the exact charts, +# versions and values Modelplane would install, under non-mp release +# names, proving Modelplane composes against a substrate it doesn't +# own. install_chart reads one chart's line from charts.tsv. +TAB="$(printf '\t')" +install_chart() { + local key="$1" k chart repo version namespace wait ref + local args=() + while IFS="$TAB" read -r k chart repo version namespace wait; do + [ "$k" = "$key" ] || continue + args=(--namespace "$namespace" --create-namespace --version "$version") + ref="$chart" + case "$repo" in + oci://*) ref="$repo/$chart" ;; + *) args+=(--repo "$repo") ;; + esac + [ -f "$ROOT/e2e/provided/values/$key.yaml" ] && args+=(-f "$ROOT/e2e/provided/values/$key.yaml") + [ "$wait" = "true" ] && args+=(--wait --timeout 10m) + helm --kube-context "$WLCTX" upgrade --install "byo-$key" "$ref" "${args[@]}" + return 0 + done <"$ROOT/e2e/provided/charts.tsv" + echo "chart $key not found in e2e/provided/charts.tsv" >&2 + return 1 +} + +# kube-prometheus-stack is held back: after the manifests apply, the run +# asserts RequirementsMet=False names it, installs it, and asserts the +# condition flips — the continuous re-check the observe Objects buy. +HOLDBACK=kube-prometheus-stack +if [ "$provided" = 1 ]; then + log "Provided mode: pre-installing the serving substrate (holding back $HOLDBACK)" + while IFS="$TAB" read -r key _ _ _ _ _; do + case "$key" in '' | \#*) continue ;; esac + [ "$key" = "$HOLDBACK" ] && continue + install_chart "$key" + done <"$ROOT/e2e/provided/charts.tsv" + kubectl --context "$WLCTX" apply -f "$ROOT/e2e/provided/manifests.yaml" +fi + # Fake DRA GPUs so a `claim: DRA` engine's ResourceClaim binds on this GPU-less # node (vendored dra-example-driver — see dra-example-driver.yaml). Without a DRA # driver the ResourceClaim stays Pending and the engine pod never schedules; the @@ -125,14 +176,21 @@ export DOCKER_CONFIG="$docker_config" # --verify after apply, wait for the ModelService and assert a live 200, # exiting non-zero on failure. This is exactly what CI runs, so # running it locally gives the same pass/fail signal (dev/CI parity). +# --provided register the workload cluster with components: Provided — the +# substrate pre-installed above, Modelplane only checking it — and +# assert the RequirementsMet flow before the usual verify. manifests="$ROOT/e2e/manifests" +if [ "$provided" = 1 ]; then + # The cluster supplies the substrate this run pre-installed above; + # patch the InferenceCluster in a rendered copy. + rendered="$work/rendered" + mkdir -p "$rendered" + cp "$ROOT/e2e/manifests/"*.yaml "$rendered/" + sed -i.bak 's/^ existing:$/ existing:\n components: Provided/' \ + "$rendered/30-inference-cluster.yaml" && rm -f "$rendered/"*.bak + manifests="$rendered" +fi cpctx="kind-$CP" -apply_manifests=1 -verify=0 -case "${1:-}" in ---no-apply) apply_manifests=0 ;; ---verify) verify=1 ;; -esac log "Building + running the control plane" cd "$ROOT" @@ -176,6 +234,76 @@ fi # model manifests. kubectl --context "$cpctx" apply -f "$manifests/" +if [ "$provided" = 1 ]; then + # The InferenceCluster mirrors the backend's RequirementsMet. First it + # must go False naming the held-back chart, then flip once the chart + # installs, then the serving stack must have composed no Helm release. + requirements() { + kubectl --context "$cpctx" get inferencecluster local \ + -o jsonpath="{.status.conditions[?(@.type=='RequirementsMet')].$1}" 2>/dev/null || true + } + + # On a fresh cluster the condition must go False naming the held-back + # chart; a rerun against a surviving cluster already has the holdback + # installed and settles at True without ever naming it, so accept + # either and only run the flip where it exists. CI always runs fresh. + log "Provided: waiting for RequirementsMet (False naming $HOLDBACK, or True on a rerun)" + state="" + for _ in $(seq 1 60); do + st="$(requirements status)" + if [ "$st" = "True" ]; then + state=met + break + fi + if [ "$st" = "False" ] && requirements message | grep -q "$HOLDBACK"; then + state=missing + break + fi + sleep 10 + done + case "$state" in + met) + log "Provided: requirements already met; skipping the holdback flip" + ;; + missing) + log "Provided: installing $HOLDBACK and waiting for RequirementsMet to flip" + install_chart "$HOLDBACK" + ok=0 + for _ in $(seq 1 60); do + if [ "$(requirements status)" = "True" ]; then + ok=1 + break + fi + sleep 10 + done + [ "$ok" = 1 ] || { + echo "verify: RequirementsMet never flipped to True after installing $HOLDBACK" >&2 + kubectl --context "$cpctx" get inferencecluster local -o yaml >&2 || true + exit 1 + } + ;; + *) + echo "verify: RequirementsMet never reported $HOLDBACK missing nor settled True" >&2 + kubectl --context "$cpctx" get inferencecluster local -o yaml >&2 || true + exit 1 + ;; + esac + + # Scoped to the ServingStack's own composed resources: the control + # plane's InferenceGateway legitimately composes Releases of its own + # (Traefik, MetalLB), which are not this assertion's business. + log "Provided: asserting the serving stack composed no Helm releases" + ss_name="$(kubectl --context "$cpctx" -n modelplane-system get servingstacks.infrastructure.modelplane.ai \ + -o jsonpath='{.items[0].metadata.name}')" + releases="$(kubectl --context "$cpctx" get releases.helm.m.crossplane.io -A \ + -l "crossplane.io/composite=$ss_name" --no-headers 2>/dev/null | grep -c . || true)" + [ "$releases" = "0" ] || { + echo "verify: expected the serving stack to compose no Release managed resources, found $releases" >&2 + kubectl --context "$cpctx" get releases.helm.m.crossplane.io -A -l "crossplane.io/composite=$ss_name" >&2 || true + exit 1 + } +fi + if [ "$verify" = 0 ]; then log "Done. Curl the ModelService per the README; clean up with: nix run .#e2e -- --clean" exit 0 diff --git a/flake.nix b/flake.nix index 56c8b85ab..edb73c72d 100644 --- a/flake.nix +++ b/flake.nix @@ -174,6 +174,7 @@ stop = apps.stop { inherit crossplane; }; e2e = apps.e2e { inherit crossplane functionsPkg; }; stacks = apps.stacks { inherit (pkgs) aicr; }; + requirements-doc = apps.requirements-doc { }; } ); diff --git a/functions/compose-inference-cluster/function/fn.py b/functions/compose-inference-cluster/function/fn.py index e5d1dd4ee..09fa3800c 100644 --- a/functions/compose-inference-cluster/function/fn.py +++ b/functions/compose-inference-cluster/function/fn.py @@ -82,6 +82,19 @@ CONDITION_REASON_INSTALLING = "Installing" CONDITION_REASON_INVALID_NODE_POOL = "InvalidNodePool" +# Mirrored from the backend ServingStack on Provided-mode Existing +# clusters: whether the cluster supplies the serving substrate the +# stack would otherwise install. compose-serving-stack derives it from +# its requirement checks; this function re-emits it here so the user +# reads what's missing off the InferenceCluster they created, not a +# composed XR they'd have to find. Checking covers the window before +# the backend first reports. +CONDITION_TYPE_REQUIREMENTS_MET = "RequirementsMet" +CONDITION_REASON_CHECKING = "Checking" + +# spec.cluster.existing.components value that selects Provided mode. +_COMPONENTS_PROVIDED = "Provided" + # Composed resource key for the backend XR. BACKEND_RESOURCE_KEY = "serving-stack" @@ -848,10 +861,46 @@ def compose_existing(self, existing: v1alpha1.Existing | None) -> None: # type defaults to GCP in the XRD; coalesce so it's never None. ssv1alpha1.Secret(type=identity.type or _IDENTITY_TYPE_GCP, name=identity.name, key=identity.key), ) - self.compose_serving_stack(backend_secrets, CLUSTER_SOURCE_EXISTING) + provided = existing.components == _COMPONENTS_PROVIDED + self.compose_serving_stack(backend_secrets, CLUSTER_SOURCE_EXISTING, provided=provided) self.write_status(self.gpu_pools()) self.derive_conditions(cluster_ready=True) + if provided: + self.mirror_requirements_condition() + + def mirror_requirements_condition(self) -> None: + """Re-emit the backend's RequirementsMet condition on this XR. + + The backend ServingStack derives it from its requirement checks + on the target cluster; mirroring status, reason and message + verbatim keeps the single source of truth there while surfacing + it where the user looks. Until the backend first reports - it + composes a reconcile after the ServingStack does - the mirror + says Checking. + """ + observed = self.req.observed.resources.get(BACKEND_RESOURCE_KEY) + condition = resource.get_condition(observed, CONDITION_TYPE_REQUIREMENTS_MET) + if condition.status not in ("True", "False"): + response.set_conditions( + self.rsp, + resource.Condition( + typ=CONDITION_TYPE_REQUIREMENTS_MET, + status="False", + reason=CONDITION_REASON_CHECKING, + message="Waiting for the serving stack to check the cluster's substrate", + ), + ) + return + response.set_conditions( + self.rsp, + resource.Condition( + typ=CONDITION_TYPE_REQUIREMENTS_MET, + status=condition.status, + reason=condition.reason, + message=condition.message, + ), + ) def compose_serving_stack( self, @@ -859,6 +908,7 @@ def compose_serving_stack( cloud: Cloud, *, gpu: ssv1alpha1.Gpu | None = None, + provided: bool = False, ) -> None: """Compose a ServingStack XR with the given secrets. @@ -867,7 +917,10 @@ def compose_serving_stack( including cloud specifics like where the node image puts the NVIDIA driver. gpu carries per-pool driver configuration the component list can't know at build time (see - civo_nvlink_disabled_pools). + civo_nvlink_disabled_pools). provided (Existing clusters only) + selects Provided mode, where the cluster supplies the substrate + and the stack only checks it; spec.components stays unset + otherwise so every other path composes byte-identically. """ # The gateway's name and the CAs it should accept client certificates # from. The name is Modelplane's own, derived from this cluster's name; @@ -887,6 +940,8 @@ def compose_serving_stack( ) if gpu is not None: spec.gpu = gpu + if provided: + spec.components = _COMPONENTS_PROVIDED resource.update( self.rsp.desired.resources[BACKEND_RESOURCE_KEY], ssv1alpha1.ServingStack( diff --git a/functions/compose-inference-cluster/tests/test_fn.py b/functions/compose-inference-cluster/tests/test_fn.py index 69ed95122..cd99f8d96 100644 --- a/functions/compose-inference-cluster/tests/test_fn.py +++ b/functions/compose-inference-cluster/tests/test_fn.py @@ -940,6 +940,83 @@ async def test_compose(self) -> None: # noqa: PLR0915 ) want3.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + # --- Case 3b: Provided mode threads spec.components to the + # ServingStack and mirrors RequirementsMet as Checking until the + # backend first reports. Derived from Case 1: same cluster, the + # existing block gains components: Provided. --- + req_provided = fnv1.RunFunctionRequest() + req_provided.CopyFrom(req1) + xr = resource.struct_to_dict(req_provided.observed.composite.resource) + xr["spec"]["cluster"]["existing"]["components"] = "Provided" + req_provided.observed.composite.resource.CopyFrom(resource.dict_to_struct(xr)) + + want_provided = fnv1.RunFunctionResponse() + want_provided.CopyFrom(want1) + ss = resource.struct_to_dict(want_provided.desired.resources["serving-stack"].resource) + ss["spec"]["components"] = "Provided" + want_provided.desired.resources["serving-stack"].resource.CopyFrom(resource.dict_to_struct(ss)) + want_provided.conditions.append( + fnv1.Condition( + type="RequirementsMet", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Checking", + message="Waiting for the serving stack to check the cluster's substrate", + ) + ) + + # --- Case 3c: the backend's RequirementsMet is mirrored verbatim + # once it reports. The backend observed unready with the condition + # naming what the cluster is missing. --- + req_mirror = fnv1.RunFunctionRequest() + req_mirror.CopyFrom(req_provided) + req_mirror.observed.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": {"name": "test-cluster-serving-stack-fd00b"}, + "status": { + "conditions": [ + {"type": "Ready", "status": "False"}, + { + "type": "RequirementsMet", + "status": "False", + "reason": "MissingRequirements", + "message": "The cluster is missing: cert-manager" + " (API v1.cert-manager.io not served)", + }, + ], + }, + } + ) + ) + ) + + want_mirror = fnv1.RunFunctionResponse() + want_mirror.CopyFrom(want_provided) + del want_mirror.conditions[:] + want_mirror.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + fnv1.Condition( + type="RequirementsMet", + status=fnv1.STATUS_CONDITION_FALSE, + reason="MissingRequirements", + message="The cluster is missing: cert-manager (API v1.cert-manager.io not served)", + ), + ] + ) + # --- Case 4: EKS cluster first pass - no observed EKS, classes resolved. --- inference_class_l4_eks = { "apiVersion": "modelplane.ai/v1alpha1", @@ -3332,6 +3409,8 @@ async def test_compose(self) -> None: # noqa: PLR0915 want1, want2, want3, + want_provided, + want_mirror, want4, want5, want6, @@ -3549,6 +3628,14 @@ async def test_compose(self) -> None: # noqa: PLR0915 Case(name="GKE cluster first pass composes GKECluster XR only", req=req2, want=want2), Case(name="GKE credentials pass through to GKECluster spec", req=req_creds, want=want_creds), Case(name="existing cluster second pass with backend ready", req=req3, want=want3), + Case( + name="provided components thread to the backend and mirror as checking", + req=req_provided, + want=want_provided, + ), + Case( + name="provided components mirror the backend's missing requirements", req=req_mirror, want=want_mirror + ), Case(name="EKS cluster first pass composes EKSCluster XR only", req=req4, want=want4), Case(name="EKS cluster not ready re-emits existing CPC unchanged", req=req5, want=want5), Case(name="GKE cluster ready composes CPC, backend, usage, and RWX StorageClass", req=req6, want=want6), diff --git a/functions/compose-serving-stack/function/fn.py b/functions/compose-serving-stack/function/fn.py index 9f77439e0..91ebc5528 100644 --- a/functions/compose-serving-stack/function/fn.py +++ b/functions/compose-serving-stack/function/fn.py @@ -32,6 +32,14 @@ spec.gateway, which is per-cluster configuration rather than stack data: the gateway pair, its PKI, and the Usages sequencing the pair's teardown ahead of the Envoy Gateway release. + +In Provided mode (spec.components, Existing clusters only) the cluster +supplies the substrate itself. No substrate component renders; each +one's requires entries render instead, as observe-only Objects on the +target cluster, and the RequirementsMet condition reports what the +cluster is missing. Modelplane's own config components still compose, +their substrate depends_on edges gated on those checks, so nothing is +ever applied into a cluster whose API can't accept it. """ import grpc @@ -137,6 +145,63 @@ "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" ) +# The Provided-mode condition: does the cluster supply what the skipped +# substrate components would have installed? Reported on the +# ServingStack and mirrored onto the InferenceCluster. False with +# MissingRequirements names exactly what's missing; Checking means the +# observe Objects haven't reported yet. +CONDITION_TYPE_REQUIREMENTS_MET = "RequirementsMet" +CONDITION_REASON_REQUIREMENTS_MET = "RequirementsMet" +CONDITION_REASON_MISSING = "MissingRequirements" +CONDITION_REASON_CHECKING = "Checking" + +# CEL readiness for a RequiredObject with no query of its own: Ready +# once the object is observed at all. DeriveFromCelQuery rather than +# SuccessfulCreate because an observe-only Object never creates +# anything - readiness must derive from what was observed. +_OBSERVED_CEL = "has(object.metadata.name)" + +# CEL readiness for a RequiredCRD's APIService: the aggregator reports +# Available once the group-version actually serves. +_APISERVICE_AVAILABLE_CEL = ( + 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")' +) + + +def _apiservice_name(r: stacks.RequiredCRD) -> str: + """The aggregated APIService name for a RequiredCRD: ..""" + return f"{r.versions[0]}.{r.name.split('.', 1)[1]}" + + +def _requirement_manifest(r: stacks.RequiredCRD | stacks.RequiredObject) -> tuple[dict, str]: + """The manifest an observe-only requirement check points at, and the + CEL query that makes it Ready. + + A RequiredCRD is checked through the aggregated APIService the API + server auto-registers for every served group-version + (.), not the CRD itself. Observing the CRD would + copy its whole OpenAPI schema into the Object's status and back + into every RunFunctionRequest - hundreds of kilobytes, nested past + the protobuf decoder's depth limit. The APIService is a few hundred + bytes and proves the same thing at group-version granularity, with + an Available condition to gate on; the generated requirements docs + still name the CRD itself. A RequiredObject observes the named + cluster-scoped object with its own query, if any. + """ + if isinstance(r, stacks.RequiredCRD): + return ( + { + "apiVersion": "apiregistration.k8s.io/v1", + "kind": "APIService", + "metadata": {"name": _apiservice_name(r)}, + }, + _APISERVICE_AVAILABLE_CEL, + ) + return ( + {"apiVersion": r.api_version, "kind": r.kind, "metadata": {"name": r.name}}, + r.ready or _OBSERVED_CEL, + ) + def _name(meta: metav1.ObjectMeta | None) -> str: """The object's name, always set on resources read from the API server.""" @@ -317,13 +382,46 @@ def compose(self) -> None: if nvlink_pools: components = stacks.civo.with_nvlink_disabled(components, nvlink_pools) - rendered = self.compose_components(components) + if self.xr.spec.components != "Provided": + rendered = self.compose_components(components) + rendered += self.compose_gateway() + rendered += self.compose_gateway_pki() + self.compose_component_usages(components) + self.compose_gateway_usages() + self.write_status() + self.mark_readiness(rendered) + return + + # Provided: the cluster supplies the substrate. Substrate + # components render as their requirement checks instead, and + # only Modelplane's config components (plus the gateway pair) + # compose - each substrate depends_on edge gated on the checks + # standing in for it. The checks count toward readiness, so the + # composite isn't Ready until the cluster meets every + # requirement, and RequirementsMet says what's missing. + checks = self.compose_requirements(components) + gates = { + c.key: (stacks.components.requirement_keys(c) if c.role == "substrate" else stacks.components.doc_keys(c)) + for c in components + } + config = [c for c in components if c.role == "config"] + rendered = self.compose_components(config, gates=gates) rendered += self.compose_gateway() - rendered += self.compose_gateway_pki() - self.compose_component_usages(components) - self.compose_gateway_usages() + # The PKI composes cert-manager and trust-manager CRs, so in + # Provided mode it waits for the checks standing in for both - + # the same gate a config component's depends_on edge gets. + pki_deps = [ + key + for c in components + if c.key in ("cert-manager", "trust-manager") + for key in stacks.components.requirement_keys(c) + ] + rendered += self.compose_gateway_pki(require=pki_deps) + self.compose_component_usages(config) + self.compose_gateway_usages(provided=True) + self.set_requirements_condition(components) self.write_status() - self.mark_readiness(rendered) + self.mark_readiness(rendered + checks) def compose_provider_configs(self) -> None: """Build ProviderConfigs from the XR's secrets. @@ -403,7 +501,11 @@ def compose_provider_configs(self) -> None: ), ) - def compose_components(self, components: list[stacks.Component]) -> list[str]: + def compose_components( + self, + components: list[stacks.Component], + gates: dict[str, list[str]] | None = None, + ) -> list[str]: """Render every component of the joined stack. A Chart renders as one provider-helm Release under the entry's @@ -422,17 +524,26 @@ def compose_components(self, components: list[stacks.Component]) -> list[str]: Release reports Ready when Helm deploys it, not when its workloads run, so this is deploy-order, not health-order. + `gates` overrides which observed keys a dependency's Ready is + read from. Provided mode passes the full joined map with each + substrate component standing behind its requirement checks, so + a config component's edge onto skipped substrate gates on the + cluster actually serving the API (an empty list gates on + nothing). Default: the components' own rendered docs. + Returns the composed-resource keys it rendered, for readiness. """ pc_observed = self.provider_configs_observed() pc = _pc_name(self.xr) - docs = {c.key: stacks.components.doc_keys(c) for c in components} + # A new name rather than narrowing the parameter: deps_ready + # closes over it, and a closed-over variable doesn't narrow. + gate_keys = gates if gates is not None else {c.key: stacks.components.doc_keys(c) for c in components} def deps_ready(c: stacks.Component) -> bool: return all( resource.get_condition(self.req.observed.resources.get(key), "Ready").status == "True" for dep in c.depends_on - for key in docs[dep] + for key in gate_keys[dep] ) rendered: list[str] = [] @@ -470,6 +581,13 @@ def compose_component_usages(self, components: list[stacks.Component]) -> None: outlives the Envoy Gateway release whose webhooks need it, and so on. Usages reference nothing on the remote cluster, so they compose ungated and are ready on arrival. + + An edge onto a component outside the given list derives no + Usage. That's the Provided path, where only config components + are passed: their substrate dependencies are the cluster's own + installs, which no Usage on the control plane can hold - the + user's controllers must outlive Modelplane's config on + teardown, which the requirements docs state. """ refs: dict[str, tuple[str, str]] = {} docs: dict[str, list[str]] = {} @@ -481,6 +599,8 @@ def compose_component_usages(self, components: list[stacks.Component]) -> None: for c in components: for dep in c.depends_on: + if dep not in docs: + continue for of_key in docs[dep]: for by_key in docs[c.key]: key = f"usage-{of_key}-by-{by_key}" @@ -547,7 +667,7 @@ def compose_gateway(self) -> list[str]: rendered.append(key) return rendered - def compose_gateway_pki(self) -> list[str]: + def compose_gateway_pki(self, require: list[str] | None = None) -> list[str]: """Compose the cluster gateway's certificate, and the requirement that a caller present one of its own. @@ -561,9 +681,18 @@ def compose_gateway_pki(self) -> list[str]: so the requirement waits for the first one, and serves_gateway withholds the listener it would have governed until then. + `require` names observed keys whose Ready gates first creation, + the same gate a config component's depends_on edge gets: Provided + mode passes the cert-manager and trust-manager checks, so no + Certificate or Bundle is applied into a cluster that doesn't + serve their APIs. + Returns the composed-resource keys it rendered, for readiness. """ - pc_observed = self.provider_configs_observed() + gate = self.provider_configs_observed() and all( + resource.get_condition(self.req.observed.resources.get(key), "Ready").status == "True" + for key in require or [] + ) pc = _pc_name(self.xr) gw = self.xr.spec.gateway @@ -626,7 +755,7 @@ def compose_gateway_pki(self) -> list[str]: ), ] for key, manifest in certs: - if not (pc_observed or key in self.req.observed.resources): + if not (gate or key in self.req.observed.resources): continue cel = _CERTIFICATE_READY_CEL if manifest["kind"] == "Certificate" else None resource.update(self.rsp.desired.resources[key], _k8s_object(pc, manifest, ready_when=cel)) @@ -644,7 +773,7 @@ def compose_gateway_pki(self) -> list[str]: # so the key is read once, in-cluster, by a controller already entitled to # it. trust-manager also rejects any PEM block that isn't a CERTIFICATE, # so it can't be made to republish a key by naming the wrong source key. - if pc_observed or "gateway-ca-bundle" in self.req.observed.resources: + if gate or "gateway-ca-bundle" in self.req.observed.resources: resource.update( self.rsp.desired.resources["gateway-ca-bundle"], _k8s_object( @@ -673,7 +802,7 @@ def compose_gateway_pki(self) -> list[str]: # Observed, not managed: trust-manager owns this ConfigMap, and this only # needs to read the certificate back out so status can publish it. - if pc_observed or "gateway-ca-configmap" in self.req.observed.resources: + if gate or "gateway-ca-configmap" in self.req.observed.resources: resource.update( self.rsp.desired.resources["gateway-ca-configmap"], _k8s_object( @@ -693,7 +822,7 @@ def compose_gateway_pki(self) -> list[str]: client_cas = gw.clientCAs if not client_cas: return rendered - if not (pc_observed or "gateway-client-ca-bundle" in self.req.observed.resources): + if not (gate or "gateway-client-ca-bundle" in self.req.observed.resources): return rendered # One ConfigMap holding every InferenceGateway's CA, concatenated, which # is what a PEM trust bundle is. @@ -744,7 +873,7 @@ def compose_gateway_pki(self) -> list[str]: rendered.append("gateway-client-auth") return rendered - def compose_gateway_usages(self) -> None: + def compose_gateway_usages(self, *, provided: bool = False) -> None: """Compose Usages ordering the hand-rendered gateway teardown. The Envoy Gateway controller must outlive the Gateway and @@ -758,10 +887,17 @@ def compose_gateway_usages(self) -> None: Gateway to be protected by. A Usage whose "by" selector matches nothing errors on every reconcile, and compose_gateway withholds the Gateway until the cluster has an InferenceGateway CA to trust. + + In Provided mode the envoy-gateway Release doesn't exist - the + cluster runs its own controller - so that edge is skipped, and + processing the GatewayClass finalizer on teardown relies on the + user keeping that controller alive. """ - usages = [ - ("usage-envoy-gateway-by-gateway-class", _RELEASE_REF, "envoy-gateway", _OBJECT_REF, "gateway-class"), - ] + usages = [] + if not provided: + usages.append( + ("usage-envoy-gateway-by-gateway-class", _RELEASE_REF, "envoy-gateway", _OBJECT_REF, "gateway-class") + ) if self.serves_gateway(): usages.insert( 0, @@ -790,6 +926,102 @@ def observed_ca_certificate(self) -> str | None: data = d.get("status", {}).get("atProvider", {}).get("manifest", {}).get("data", {}) return data.get("ca.crt") or None + def compose_requirements(self, components: list[stacks.Component]) -> list[str]: + """Render every substrate component's requirement checks. + + One observe-only Object per requirement, keyed + require-- (stacks.components. + requirement_keys), observing the served API or cluster-scoped + object the provided cluster must supply in its place. The + Observe management policy means creating one touches nothing + remote and neither does deleting it; provider-kubernetes keeps + re-observing, so a check flips Ready when the admin installs + the missing piece - and flips back if it's removed. Gated on + the ProviderConfigs like every remote-cluster resource. + + Returns the composed-resource keys it rendered; they count + toward readiness so the composite can't be Ready with an unmet + requirement. + """ + pc_observed = self.provider_configs_observed() + pc = _pc_name(self.xr) + rendered: list[str] = [] + for c in components: + if c.role != "substrate": + continue + for key, r in zip(stacks.components.requirement_keys(c), c.requires, strict=True): + if not (pc_observed or key in self.req.observed.resources): + continue + manifest, cel = _requirement_manifest(r) + obj = _k8s_object( + pc, + manifest, + metadata=metav1.ObjectMeta(labels={_LABEL_RESOURCE: key}), + ready_when=cel, + management_policies=["Observe"], + ) + resource.update(self.rsp.desired.resources[key], obj) + rendered.append(key) + return rendered + + def set_requirements_condition(self, components: list[stacks.Component]) -> None: + """Derive the RequirementsMet condition from the observed checks. + + The condition is the primary UX for a provided cluster: the + message names every unmet requirement in one place, instead of + the user chasing per-Object errors. A check whose Synced is + False was unobservable (the object doesn't exist, or the + kubeconfig can't read it); Synced True with Ready False means + the object exists but fails its query (a CRD not serving an + accepted version). Checks not yet observed report Checking. + """ + missing: list[str] = [] + checking = False + for c in components: + if c.role != "substrate": + continue + for key, r in zip(stacks.components.requirement_keys(c), c.requires, strict=True): + observed = self.req.observed.resources.get(key) + if observed is None: + checking = True + continue + if resource.get_condition(observed, "Ready").status == "True": + continue + synced = resource.get_condition(observed, "Synced").status + if synced == "False": + detail = "not served" if isinstance(r, stacks.RequiredCRD) else "not found" + elif synced != "True": + checking = True + continue + elif isinstance(r, stacks.RequiredCRD): + detail = "not available" + else: + detail = "not ready" + what = f"API {_apiservice_name(r)}" if isinstance(r, stacks.RequiredCRD) else f"{r.kind} {r.name}" + missing.append(f"{c.key} ({what} {detail})") + + if missing: + condition = resource.Condition( + typ=CONDITION_TYPE_REQUIREMENTS_MET, + status="False", + reason=CONDITION_REASON_MISSING, + message=f"The cluster is missing: {'; '.join(missing)}", + ) + elif checking: + condition = resource.Condition( + typ=CONDITION_TYPE_REQUIREMENTS_MET, + status="False", + reason=CONDITION_REASON_CHECKING, + message="Checking the cluster provides the serving substrate", + ) + else: + condition = resource.Condition( + typ=CONDITION_TYPE_REQUIREMENTS_MET, + status="True", + reason=CONDITION_REASON_REQUIREMENTS_MET, + ) + response.set_conditions(self.rsp, condition) + def write_status(self) -> None: """Extract the gateway address from the observed Gateway Object and write it to the XR's status.""" diff --git a/functions/compose-serving-stack/function/stacks/__init__.py b/functions/compose-serving-stack/function/stacks/__init__.py index 6e8d240a6..318303e1c 100644 --- a/functions/compose-serving-stack/function/stacks/__init__.py +++ b/functions/compose-serving-stack/function/stacks/__init__.py @@ -24,13 +24,23 @@ from function.stacks import common, components, dynamo, standard from function.stacks.clouds import civo, existing, nebius, vultr from function.stacks.clouds.generated.aicr import aks, eks, gke -from function.stacks.components import Chart, Cloud, Component, Manifests, Stack +from function.stacks.components import ( + Chart, + Cloud, + Component, + Manifests, + RequiredCRD, + RequiredObject, + Stack, +) __all__ = [ "Chart", "Cloud", "Component", "Manifests", + "RequiredCRD", + "RequiredObject", "Stack", "clouds", "components", @@ -38,6 +48,11 @@ "stacks", ] +# Kubernetes label values cap at 63 characters, and fn.py labels every +# composed resource with its key (_LABEL_RESOURCE), so a requirement +# key that renders longer than this couldn't be composed. +_MAX_KEY = 63 + # The cloud halves, keyed by the InferenceCluster's source values. EKS, # AKS and GKE come from clouds/generated/aicr/, written by # `nix run .#stacks`; the rest are hand-written in clouds/. @@ -74,7 +89,13 @@ def join(cloud: Cloud, stack: Stack) -> list[Component]: unknown cloud or stack, on a key two lists both produce, and on a depends_on edge naming a component the join didn't produce - which catches a generator allowlist that dropped something another - component needs. + component needs. The Provided-mode requirement data is held to the + same bar: requirement keys must be unique and composable, only a + substrate component may carry them, and on Existing - the one cloud + Provided mode can select - every substrate component must say what + a provided cluster supplies in its place (`requires`) or state why + nothing is checkable (`unchecked`), so no component can silently + become uncheckable. """ if cloud not in _CLOUDS: raise ValueError(f"unknown cloud {cloud!r}; known: {', '.join(_CLOUDS)}") @@ -89,9 +110,11 @@ def join(cloud: Cloud, stack: Stack) -> list[Component]: raise ValueError(f"{cloud}/{stack}: duplicate component keys {duplicates}") # The composed-resource keys a component renders under (one per - # manifest for a multi-doc bundle) must be unique across the join + # manifest for a multi-doc bundle, plus one observe Object per + # requirement in Provided mode) must be unique across the join # too, or two components would fight over one desired resource. rendered = [k for c in joined for k in components.doc_keys(c)] + rendered += [k for c in joined for k in components.requirement_keys(c)] duplicates = sorted({k for k in rendered if rendered.count(k) > 1}) if duplicates: raise ValueError(f"{cloud}/{stack}: duplicate composed-resource keys {duplicates}") @@ -102,4 +125,30 @@ def join(cloud: Cloud, stack: Stack) -> list[Component]: if dep not in known: raise ValueError(f"{cloud}/{stack}: {c.key} depends on {dep!r}, which the join did not produce") + _validate_requirements(cloud, stack, joined) + return joined + + +def _validate_requirements(cloud: Cloud, stack: Stack, joined: list[Component]) -> None: + """Validate the joined components' Provided-mode requirement data. + + Split from join() only to keep it under the branch-count lint + threshold; the failure semantics are join()'s. + """ + for c in joined: + if c.role == "config" and (c.requires or c.unchecked or c.not_needed): + raise ValueError(f"{cloud}/{stack}: {c.key} is config; requirement data belongs on substrate components") + for key in components.requirement_keys(c): + if len(key) > _MAX_KEY: + raise ValueError(f"{cloud}/{stack}: requirement key {key!r} exceeds {_MAX_KEY} characters") + for r in c.requires: + # The APIService check encodes one group-version; any-of + # waits until a requirement actually needs it. + if isinstance(r, components.RequiredCRD) and len(r.versions) != 1: + raise ValueError(f"{cloud}/{stack}: {c.key} requirement {r.key!r} must accept exactly one version") + + if cloud == "Existing": + silent = sorted(c.key for c in joined if c.role == "substrate" and not (c.requires or c.unchecked)) + if silent: + raise ValueError(f"{cloud}/{stack}: substrate components without requires or unchecked notes: {silent}") diff --git a/functions/compose-serving-stack/function/stacks/clouds/existing.py b/functions/compose-serving-stack/function/stacks/clouds/existing.py index 6dae5b824..169314a4f 100644 --- a/functions/compose-serving-stack/function/stacks/clouds/existing.py +++ b/functions/compose-serving-stack/function/stacks/clouds/existing.py @@ -20,9 +20,15 @@ component also appears on a generated cloud, this file states the same pin, so one review moves both halves when a version changes - regenerate, then mirror the shared pins here. + +This half also carries requirement data (`requires`, `unchecked`, +`not_needed`): what a cluster must already supply per component when it +provides the substrate itself (spec.components: Provided). Existing is +the only cloud that mode can select, so only this half and the shared +halves it joins with state it. """ -from function.stacks.components import Chart, Component +from function.stacks.components import Chart, Component, RequiredCRD, RequiredObject COMPONENTS: list[Component] = [ Chart( @@ -40,6 +46,12 @@ # Objects, and removing their CRDs would stop provider-kubernetes # observing them to release their finalizers. values={"crds": {"enabled": True}}, + requires=[ + RequiredCRD(key="crds", name="certificates.cert-manager.io", versions=["v1"]), + ], + unchecked=[ + "The cert-manager controller and webhook are running and issue Certificates.", + ], ), Chart( key="kube-prometheus-stack", @@ -94,6 +106,23 @@ "grafana": {"enabled": False}, "alertmanager": {"enabled": False}, }, + requires=[ + RequiredCRD(key="podmonitors", name="podmonitors.monitoring.coreos.com", versions=["v1"]), + RequiredCRD(key="servicemonitors", name="servicemonitors.monitoring.coreos.com", versions=["v1"]), + ], + unchecked=[ + "Prometheus discovers `PodMonitor` objects in every namespace:" + " with the chart, set `podMonitorSelectorNilUsesHelmValues` to" + " false and `podMonitorNamespaceSelector` to empty." + " Modelplane's scrape targets are `PodMonitor` objects in" + " workload namespaces, and the chart's default release label" + " selector never matches them.", + "A scrape job for the Envoy Gateway proxy pods' stats" + " endpoint, if you want request metrics at the proxy level.", + ], + not_needed=[ + "Modelplane disables Grafana and Alertmanager.", + ], ), Chart( key="node-feature-discovery", @@ -123,6 +152,17 @@ ], }, }, + requires=[ + RequiredCRD(key="nodefeatures", name="nodefeatures.nfd.k8s-sigs.io", versions=["v1alpha1"]), + ], + unchecked=[ + "The NFD worker runs on the GPU nodes and labels them with" + " `feature.node.kubernetes.io/pci-10de` and friends. If GPU" + " nodes are tainted, the worker must tolerate the taint, or" + " the DRA driver's `kubelet` plugin never schedules there and" + " every GPU ResourceClaim stays pending with all components" + " looking healthy.", + ], ), # Publishes each GPU node's devices as DRA ResourceSlices and # registers the gpu.nvidia.com DeviceClass ModelReplica @@ -141,5 +181,27 @@ "gpuResourcesEnabledOverride": True, "resources": {"computeDomains": {"enabled": False}}, }, + requires=[ + # Observing the DeviceClass at resource.k8s.io/v1 also + # proves the cluster serves GA DRA, which means Kubernetes + # 1.34 or newer. + RequiredObject( + key="deviceclass", + api_version="resource.k8s.io/v1", + kind="DeviceClass", + name="gpu.nvidia.com", + ), + ], + unchecked=[ + "The NVIDIA kernel driver and Container Toolkit on every" + " GPU node, from the node image or the GPU Operator. The" + " requirements for an existing cluster state the versions.", + "The driver's `kubelet` plugin publishes each GPU node's devices as ResourceSlices.", + "No device plugin advertising `nvidia.com/gpu`: a second" + " allocator would hand out the same GPUs behind DRA's back.", + ], + not_needed=[ + "Modelplane disables `ComputeDomains` (multi-node NVLink) and their prerequisites.", + ], ), ] diff --git a/functions/compose-serving-stack/function/stacks/common.py b/functions/compose-serving-stack/function/stacks/common.py index 10c5ec3ee..22a6f111b 100644 --- a/functions/compose-serving-stack/function/stacks/common.py +++ b/functions/compose-serving-stack/function/stacks/common.py @@ -35,7 +35,7 @@ import yaml -from function.stacks.components import Chart, Component, Manifests +from function.stacks.components import Chart, Component, Manifests, RequiredCRD # The AI Gateway controller supplies the ext-proc extension server that # Envoy Gateway delegates InferencePool backend resolution to, so @@ -96,6 +96,9 @@ def _crds(filename: str) -> list[dict[str, Any]]: # against the joined list. Envoy Gateway needs it for its # webhooks. depends_on=["cert-manager"], + # gateway-proxy below depends on this chart, so its Ready must + # mean healthy, not just deployed. + wait=True, # The extensionManager block points Envoy Gateway at the Envoy AI # Gateway controller's ext-proc server and declares InferencePool # a backend resource, so HTTPRoute -> InferencePool backendRefs @@ -134,6 +137,26 @@ def _crds(filename: str) -> list[dict[str, Any]]: }, }, }, + requires=[ + RequiredCRD(key="gatewayclasses", name="gatewayclasses.gateway.networking.k8s.io", versions=["v1"]), + RequiredCRD(key="gateways", name="gateways.gateway.networking.k8s.io", versions=["v1"]), + RequiredCRD(key="httproutes", name="httproutes.gateway.networking.k8s.io", versions=["v1"]), + RequiredCRD(key="envoyproxies", name="envoyproxies.gateway.envoyproxy.io", versions=["v1alpha1"]), + RequiredCRD(key="backends", name="backends.gateway.envoyproxy.io", versions=["v1alpha1"]), + # The gateway PKI composes a ClientTrafficPolicy to require + # client certificates (fn.py's compose_gateway_pki), so a + # provided install must serve it too. + RequiredCRD(key="ctp", name="clienttrafficpolicies.gateway.envoyproxy.io", versions=["v1alpha1"]), + ], + unchecked=[ + "The Envoy Gateway controller runs with the" + " `extensionManager` wired exactly as the values above:" + " external processing delegated to the AI Gateway" + " controller's Service, with the Backend API enabled and" + " InferencePool declared a backend resource. Without it," + " HTTPRoute to InferencePool `backendRefs` never route," + " with every component looking healthy.", + ], ), Chart( key="ai-gateway-crds", @@ -144,6 +167,14 @@ def _crds(filename: str) -> list[dict[str, Any]]: version=_AI_GATEWAY_VERSION, # ai-gateway depends on this chart. wait=True, + requires=[ + RequiredCRD(key="routes", name="aigatewayroutes.aigateway.envoyproxy.io", versions=["v1alpha1"]), + ], + unchecked=[ + "The AI Gateway APIs are v1alpha1 and move with the" + " controller. A provided install tracks the pinned" + f" {_AI_GATEWAY_VERSION} release.", + ], ), Chart( key="ai-gateway", @@ -169,6 +200,13 @@ def _crds(filename: str) -> list[dict[str, Any]]: # fix in flight as #2601). That failure is visible, where a lost caller # isn't. values={"controller": {"logRequestHeaderAttributes": f"{_CALLER_HEADER}:caller"}}, + unchecked=[ + "The AI Gateway controller is reachable at" + f" `ai-gateway-controller.{_AI_GATEWAY_NAMESPACE}.svc.cluster.local:1063`," + " the address Modelplane's Envoy Gateway `extensionManager`" + " values point at. A controller installed elsewhere never" + " receives the external processing traffic.", + ], ), # Gateway API Inference Extension CRDs, providing the InferencePool # that disaggregated replicas front their decode endpoints with. @@ -176,6 +214,9 @@ def _crds(filename: str) -> list[dict[str, Any]]: Manifests( key="gaie-crds", manifests=_crds("gaie.yaml"), + requires=[ + RequiredCRD(key="pools", name="inferencepools.inference.networking.k8s.io", versions=["v1"]), + ], ), # Both gateways live here, with the fleet gateway's own healthz and redirect # routes, and nothing else provisions the namespace. The gateways' listeners @@ -184,6 +225,7 @@ def _crds(filename: str) -> list[dict[str, Any]]: # Without it the gateways' own routes wouldn't attach. Manifests( key="gateway-namespace", + role="config", manifests=[ { "apiVersion": "v1", @@ -204,7 +246,12 @@ def _crds(filename: str) -> list[dict[str, Any]]: # GatewayClass references it via parametersRef. Manifests( key="gateway-proxy", - depends_on=["gateway-namespace"], + role="config", + # envoy-gateway serves the EnvoyProxy CRD this CR instantiates. + # The edge orders install on the API existing - in Provided mode + # it maps to the envoy-gateway requirement checks - and holds + # the release on teardown until the CR is gone. + depends_on=["gateway-namespace", "envoy-gateway"], manifests=[ { "apiVersion": "gateway.envoyproxy.io/v1alpha1", @@ -228,6 +275,7 @@ def _crds(filename: str) -> list[dict[str, Any]]: # namespace it lives in. Manifests( key="gateway-selfsigned-issuer", + role="config", depends_on=["cert-manager", "gateway-namespace"], manifests=[ { @@ -269,6 +317,15 @@ def _crds(filename: str) -> list[dict[str, Any]]: # it drops an init container and the image pull it waits on. "defaultPackage": {"enabled": False}, }, + requires=[ + RequiredCRD(key="bundles", name="bundles.trust.cert-manager.io", versions=["v1alpha1"]), + ], + unchecked=[ + "trust-manager watches modelplane-system as its trust" + " namespace, where the gateway PKI composes its Bundle." + " An install watching another namespace never syncs the" + " Bundle, and the control plane can't read the cluster CA.", + ], ), # The DRA driver's kubelet plugin runs at system-node-critical # priority. GKE only admits such pods in a namespace whose @@ -276,8 +333,18 @@ def _crds(filename: str) -> list[dict[str, Any]]: # daemonset gets FailedCreate and never publishes ResourceSlices. # Laid down everywhere: it only grants headroom, so it's harmless on # clusters that don't restrict them. + # Substrate, not config: it targets the namespace the DRA driver + # chart installs into, which a provided cluster's driver may not + # even use, and granting the headroom belongs to whoever installed + # the driver there. Manifests( key="dra-driver-critical-pods-quota", + unchecked=[ + "On clusters that restrict `system-node-critical` pods by" + " namespace quota (GKE does), the DRA driver's namespace" + " needs a `ResourceQuota` admitting them, or the `kubelet`" + " plugin `DaemonSet` never starts.", + ], manifests=[ { "apiVersion": "v1", diff --git a/functions/compose-serving-stack/function/stacks/components.py b/functions/compose-serving-stack/function/stacks/components.py index cc6267681..11314110e 100644 --- a/functions/compose-serving-stack/function/stacks/components.py +++ b/functions/compose-serving-stack/function/stacks/components.py @@ -32,6 +32,63 @@ Cloud = Literal["GKE", "EKS", "AKS", "Nebius", "Vultr", "Civo", "Existing"] Stack = Literal["Standard", "Dynamo"] +# Who a component belongs to when the cluster provides the substrate +# (ServingStack spec.components: Provided). A substrate component is +# skipped there - the cluster already runs it - and its `requires` +# entries are checked in its place. A config component is Modelplane's +# own configuration on top of the substrate (the gateway namespace and +# EnvoyProxy, the KAI Queues, the ModelExpress server) and composes in +# every mode. +Role = Literal["substrate", "config"] + + +@dataclass +class RequiredCRD: + """A CRD a provided cluster must serve in a substrate component's place. + + Checked through the aggregated APIService the API server registers + for every served group-version (.), not the CRD + itself: a CRD manifest carries its whole OpenAPI schema, too large + and too deeply nested to haul through the composition pipeline on + every reconcile. Served-API granularity is the whole checkable + surface anyway - chart and controller versions aren't recoverable + from a CRD either, so they belong in the component's `unchecked` + notes. The generated docs still name the CRD itself. + + `versions` holds exactly one entry today: the APIService check + encodes one group-version, and join() fails closed on more until a + requirement actually needs any-of semantics. + + `key` suffixes the composed-resource key (see requirement_keys), so + it only needs to be unique within one component's requires list. + """ + + key: str + name: str # ., the CRD's metadata.name + versions: list[str] # served versions accepted + + +@dataclass +class RequiredObject: + """A cluster-scoped object a provided cluster must already carry. + + For substrate whose footprint isn't a CRD: the NVIDIA DRA driver is + checked through the gpu.nvidia.com DeviceClass it registers. + Observing it also proves the cluster serves the object's API group, + so a RequiredObject on a versioned core API doubles as a floor check + (resource.k8s.io/v1 means Kubernetes 1.34). `ready` is an optional + CEL query over the observed manifest, as on Manifests. + """ + + key: str + api_version: str + kind: str + name: str + ready: str | None = None + + +Requirement = RequiredCRD | RequiredObject + @dataclass class Chart: @@ -56,6 +113,18 @@ class Chart: depends on, so the install gate orders on health rather than deploy - the generator derives it from the dependency edges, and the hand-written files state it where a cross-half edge lands on them. + + `role`, `requires`, `unchecked` and `not_needed` describe the + component when the cluster provides the substrate instead of + Modelplane installing it (spec.components: Provided, Existing + clusters only). `requires` is what gets checked in the component's + place; `unchecked` is what a provided cluster must also supply but + no observe Object can verify (controllers running, node drivers, + values wiring), stated for the generated requirements docs; and + `not_needed` is what the component would normally bring that + Modelplane doesn't use, so users know what they can skip. Only the + Existing halves carry them - the generated clouds never join in + Provided mode. """ key: str @@ -67,6 +136,10 @@ class Chart: wait: bool = False depends_on: list[str] = field(default_factory=list) values: dict[str, Any] | None = None + role: Role = "substrate" + requires: list[Requirement] = field(default_factory=list) + unchecked: list[str] = field(default_factory=list) + not_needed: list[str] = field(default_factory=list) @dataclass @@ -82,12 +155,21 @@ class Manifests: applied to every doc in the entry (see fn.py's _k8s_object): use it when readiness must reflect a controller-populated status field, and keep an entry to one doc when only that doc has one. + + `role`, `requires`, `unchecked` and `not_needed` behave as on + Chart. Most Manifests entries are Modelplane's own configuration + (role config); the substrate ones are the vendored CRD bundles a + provided cluster brings itself. """ key: str manifests: list[dict[str, Any]] depends_on: list[str] = field(default_factory=list) ready: str | None = None + role: Role = "substrate" + requires: list[Requirement] = field(default_factory=list) + unchecked: list[str] = field(default_factory=list) + not_needed: list[str] = field(default_factory=list) # A plain assignment rather than a `type` statement: the packages @@ -109,3 +191,14 @@ def doc_keys(component: Component) -> list[str]: if isinstance(component, Chart) or len(component.manifests) == 1: return [component.key] return [f"{component.key}-{doc['metadata']['name']}" for doc in component.manifests] + + +def requirement_keys(component: Component) -> list[str]: + """Composed-resource keys of a component's checks in Provided mode. + + One observe Object per requirement, keyed + `require--`. The same rename caveat + as doc_keys applies in the harmless direction: renaming recreates + the Object, but an observe-only Object touches nothing remote. + """ + return [f"require-{component.key}-{r.key}" for r in component.requires] diff --git a/functions/compose-serving-stack/function/stacks/dynamo.py b/functions/compose-serving-stack/function/stacks/dynamo.py index 646f7d86d..2199f0173 100644 --- a/functions/compose-serving-stack/function/stacks/dynamo.py +++ b/functions/compose-serving-stack/function/stacks/dynamo.py @@ -32,7 +32,7 @@ import yaml -from function.stacks.components import Chart, Component, Manifests +from function.stacks.components import Chart, Component, Manifests, RequiredCRD # The name and the `default` namespace are a cross-function contract: # compose-model-replica points engine pods, which run in their team's @@ -97,6 +97,16 @@ def _kai_queue(name: str, parent: str | None) -> dict[str, Any]: chart="grove-charts", repository="oci://ghcr.io/ai-dynamo/grove", version="v0.1.0-alpha.12-rc2", + requires=[ + RequiredCRD(key="podcliquesets", name="podcliquesets.grove.io", versions=["v1alpha1"]), + ], + unchecked=[ + "Grove's API is v1alpha1 with no compatibility promise" + " between versions, so a provided install must run the exact" + " pinned version. A CRD check can't tell alpha revisions" + " apart. Prefer Managed for the Dynamo stack until Grove" + " stabilizes.", + ], ), Chart( key="kai-scheduler", @@ -108,6 +118,15 @@ def _kai_queue(name: str, parent: str | None) -> dict[str, Any]: # The Queue CRs below depend on this chart: KAI must serve the # Queue CRD and its webhook before they are first applied. wait=True, + requires=[ + RequiredCRD(key="queues", name="queues.scheduling.run.ai", versions=["v2"]), + ], + unchecked=[ + "The scheduler answers to the `schedulerName` value" + " `kai-scheduler`, the name Modelplane's engine pods" + " request. Its Queue webhook must be serving. Modelplane" + " composes against the v0.16 line.", + ], ), # KAI refuses to schedule a pod whose queue doesn't exist. Its chart # installs a default hierarchy, but nothing ties Modelplane's @@ -118,19 +137,24 @@ def _kai_queue(name: str, parent: str | None) -> dict[str, Any]: # first leaves the CRs hanging with no controller to finalize them. Manifests( key="kai-queue-root", + role="config", depends_on=["kai-scheduler"], manifests=[_kai_queue("modelplane-root", None)], ), Manifests( key="kai-queue", + role="config", depends_on=["kai-scheduler"], manifests=[_kai_queue("modelplane", "modelplane-root")], ), # ModelExpress CRDs (ModelMetadata, ModelCacheEntry), the metadata # backend the shared server uses. Vendored from the upstream - # release. + # release. Config, not substrate: they pair 1:1 with the server + # below, which Modelplane runs in every mode, and nothing else on a + # cluster brings them. Manifests( key="modelexpress-crds", + role="config", manifests=_crds("modelexpress.yaml"), ), # The shared ModelExpress server, one per Dynamo cluster. It's @@ -142,6 +166,7 @@ def _kai_queue(name: str, parent: str | None) -> dict[str, Any]: # and the depends_on edge, and a single-doc entry keeps its key. Manifests( key="modelexpress-server-sa", + role="config", manifests=[ { "apiVersion": "v1", @@ -157,6 +182,7 @@ def _kai_queue(name: str, parent: str | None) -> dict[str, Any]: # own Helm chart Role. Manifests( key="modelexpress-server-role", + role="config", manifests=[ { "apiVersion": "rbac.authorization.k8s.io/v1", @@ -184,6 +210,7 @@ def _kai_queue(name: str, parent: str | None) -> dict[str, Any]: ), Manifests( key="modelexpress-server-rolebinding", + role="config", manifests=[ { "apiVersion": "rbac.authorization.k8s.io/v1", @@ -206,6 +233,7 @@ def _kai_queue(name: str, parent: str | None) -> dict[str, Any]: ), Manifests( key="modelexpress-server-svc", + role="config", manifests=[ { "apiVersion": "v1", @@ -222,6 +250,7 @@ def _kai_queue(name: str, parent: str | None) -> dict[str, Any]: # outlive it for cleanup to resolve. Manifests( key="modelexpress-server", + role="config", depends_on=["modelexpress-crds"], ready=_MODELEXPRESS_SERVER_READY_CEL, manifests=[ diff --git a/functions/compose-serving-stack/function/stacks/standard.py b/functions/compose-serving-stack/function/stacks/standard.py index 4b815571c..d651d024b 100644 --- a/functions/compose-serving-stack/function/stacks/standard.py +++ b/functions/compose-serving-stack/function/stacks/standard.py @@ -20,7 +20,7 @@ carries an lws component (NVIDIA/aicr#2500 tracks the aicr one). """ -from function.stacks.components import Chart, Component +from function.stacks.components import Chart, Component, RequiredCRD COMPONENTS: list[Component] = [ Chart( @@ -30,5 +30,11 @@ chart="lws", repository="oci://registry.k8s.io/lws/charts", version="v0.8.0", + requires=[ + RequiredCRD(key="lws", name="leaderworkersets.leaderworkerset.x-k8s.io", versions=["v1"]), + ], + unchecked=[ + "The LeaderWorkerSet controller is running. Modelplane composes against the v0.8 line.", + ], ), ] diff --git a/functions/compose-serving-stack/requirements_doc.py b/functions/compose-serving-stack/requirements_doc.py new file mode 100644 index 000000000..e60b18e63 --- /dev/null +++ b/functions/compose-serving-stack/requirements_doc.py @@ -0,0 +1,222 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Generate the serving stack requirements docs page and e2e inputs. + +Renders, from the requirement data the component lists carry +(function/stacks/): + +- docs/content/platform/serving-stack-requirements.md: per substrate + component, what Modelplane would install in Managed mode, what a + Provided cluster must supply in its place, what nothing can check, + and what users can skip. +- e2e/provided/: the Standard substrate as helm inputs (charts.tsv, + values/.yaml) and raw manifests, so the Provided-mode e2e + (`nix run .#e2e -- --provided`) pre-installs exactly the versions and + values Modelplane would, without a second copy of the pins. + +The docs, the e2e inputs, and the Provided-mode checks in +compose-serving-stack all read the same data, so they can't drift; the +requirements-doc-current flake check regenerates these outputs and +fails on a stale or hand-edited copy. Run via +`nix run .#requirements-doc`. +""" + +import pathlib +import sys + +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent)) + +import yaml +from function import stacks +from function.stacks.components import ( + Chart, + Component, + RequiredCRD, + RequiredObject, +) + +ROOT = pathlib.Path(__file__).resolve().parents[2] +OUT = ROOT / "docs" / "content" / "platform" / "serving-stack-requirements.md" +E2E = ROOT / "e2e" / "provided" + +GENERATED_BANNER = ( + "# Generated by functions/compose-serving-stack/requirements_doc.py\n" + "# from the serving stack component lists. Do not edit; run\n" + "# `nix run .#requirements-doc` after changing the stack data.\n" +) + +HEADER = """\ +--- +title: Serving Stack Requirements +weight: 35 +description: What an existing cluster must provide to run the serving stack itself. +--- + +An `InferenceCluster` with `spec.cluster.existing.components: Provided` +installs no serving stack components. The cluster provides the whole +substrate itself, and Modelplane only verifies it and composes its own +configuration on top. This page lists what that cluster must provide, +per component Modelplane would otherwise install. It applies on top of +the general [requirements for an existing +cluster]({{< ref "/platform/inference-cluster.md" >}}). + +Modelplane checks the **checked** entries continuously and reports what +is missing through the `RequirementsMet` condition on the +`InferenceCluster`. A check verifies presence and served API versions, +not the installed release: the **not checked** entries, including +component versions, are yours to meet. The version listed per component +is the one Modelplane installs in Managed mode and tests against; stay +close to it. +""" + +FOOTER = """\ +## What Modelplane still installs + +Provided mode only skips the substrate. Modelplane still composes its +own configuration and workloads: the `modelplane-system` namespace, the +`EnvoyProxy`, `GatewayClass` and `Gateway` for the inference gateway, +and on the Dynamo stack its KAI `Queue` hierarchy and the ModelExpress +server with its CRDs. Their substrate dependencies gate on the checks +above, so none of them is applied before the cluster serves the APIs +they need. + +During deletion, Modelplane can't order its configuration ahead of a +substrate it doesn't own. Keep your controllers (the gateway +controller, KAI) running while an `InferenceCluster` deletes, so they +can process finalizers on Modelplane's configuration. +""" + + +def _requirement_line(r: RequiredCRD | RequiredObject) -> str: + if isinstance(r, RequiredCRD): + versions = " or ".join(f"`{v}`" for v in r.versions) + return f"CRD `{r.name}` serving {versions}" + return f"`{r.kind}` `{r.name}` (`{r.api_version}`)" + + +def _component_section(c: Component) -> list[str]: + # The heading is the component key in a code span: keys are stable + # identity (and stable anchors), and Vale skips code spans, so + # lowercase keys like kai-scheduler don't trip its brand-name rules. + lines = [f"### `{c.key}`", ""] + if isinstance(c, Chart): + lines.append(f"Managed mode installs chart `{c.chart}` `{c.version}`.") + else: + lines.append("Managed mode applies manifests Modelplane vendors from the upstream release.") + lines.append("") + if c.requires: + lines.append("Checked:") + lines.append("") + lines.extend(f"- {_requirement_line(r)}" for r in c.requires) + lines.append("") + if c.unchecked: + lines.append("Not checked:") + lines.append("") + lines.extend(f"- {note}" for note in c.unchecked) + lines.append("") + if c.not_needed: + lines.append("Not needed:") + lines.append("") + lines.extend(f"- {note}" for note in c.not_needed) + lines.append("") + if isinstance(c, Chart) and c.key == "envoy-gateway" and c.values: + lines.append( + "The exact values Modelplane installs the chart with. The" + " `extensionManager` wiring is the one coupling no check can" + " verify. Without it, routes to an `InferencePool` never" + " route while every component looks healthy:" + ) + lines.append("") + lines.append("```yaml") + lines.append(yaml.safe_dump(c.values, default_flow_style=False, sort_keys=False).rstrip()) + lines.append("```") + lines.append("") + return lines + + +def render() -> str: + standard = stacks.join("Existing", "Standard") + dynamo = stacks.join("Existing", "Dynamo") + standard_keys = {c.key for c in standard} + dynamo_keys = {c.key for c in dynamo} + + shared = [c for c in standard if c.role == "substrate" and c.key in dynamo_keys] + standard_only = [c for c in standard if c.role == "substrate" and c.key not in dynamo_keys] + dynamo_only = [c for c in dynamo if c.role == "substrate" and c.key not in standard_keys] + + lines = [HEADER] + lines.append("## On every stack") + lines.append("") + for c in shared: + lines.extend(_component_section(c)) + lines.append("## Standard stack") + lines.append("") + for c in standard_only: + lines.extend(_component_section(c)) + lines.append("## Dynamo stack") + lines.append("") + for c in dynamo_only: + lines.extend(_component_section(c)) + lines.append(FOOTER) + return "\n".join(lines) + + +def write_e2e() -> list[pathlib.Path]: + """Write the Standard substrate as e2e pre-install inputs. + + charts.tsv carries one substrate chart per line, in stack data + order (which is dependency order: cert-manager before the gateway + that needs its webhooks), values/.yaml the chart's values + verbatim, and manifests.yaml the substrate's raw manifests (the + vendored GAIE CRDs). run.sh loops helm over them with non-mp + release names, proving Modelplane composes against a substrate it + doesn't own. + """ + written: list[pathlib.Path] = [] + values_dir = E2E / "values" + values_dir.mkdir(parents=True, exist_ok=True) + for stale in values_dir.glob("*.yaml"): + stale.unlink() + + rows: list[str] = [] + manifests: list[dict] = [] + for c in stacks.join("Existing", "Standard"): + if c.role != "substrate": + continue + if isinstance(c, Chart): + rows.append("\t".join((c.key, c.chart, c.repository, c.version, c.namespace, str(c.wait).lower()))) + if c.values: + out = values_dir / f"{c.key}.yaml" + out.write_text(GENERATED_BANNER + yaml.safe_dump(c.values, default_flow_style=False, sort_keys=False)) + written.append(out) + else: + manifests.extend(c.manifests) + + charts = E2E / "charts.tsv" + charts.write_text(GENERATED_BANNER + "\n".join(rows) + "\n") + written.append(charts) + out = E2E / "manifests.yaml" + out.write_text(GENERATED_BANNER + yaml.safe_dump_all(manifests, default_flow_style=False, sort_keys=False)) + written.append(out) + return written + + +if __name__ == "__main__": + OUT.write_text(render()) + print(f"wrote {OUT.relative_to(ROOT)}") + for path in write_e2e(): + print(f"wrote {path.relative_to(ROOT)}") diff --git a/functions/compose-serving-stack/tests/test_fn.py b/functions/compose-serving-stack/tests/test_fn.py index f041b2edb..d87e858d4 100644 --- a/functions/compose-serving-stack/tests/test_fn.py +++ b/functions/compose-serving-stack/tests/test_fn.py @@ -93,28 +93,31 @@ def _crds(filename: str) -> list[dict]: ] -def _request(cloud: str, stack: str, observed: dict | None = None) -> fnv1.RunFunctionRequest: +def _request( + cloud: str, stack: str, observed: dict | None = None, components: str | None = None +) -> fnv1.RunFunctionRequest: """Build a RunFunctionRequest for a test-backend ServingStack.""" + spec = v1alpha1.Spec( + cloud=cloud, # ty: ignore[invalid-argument-type] # cases pass values of the literal + stack=stack, # ty: ignore[invalid-argument-type] + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname=_GATEWAY_HOSTNAME, + clientCAs=[v1alpha1.ClientCA(name="eu", certificate=_CLIENT_CA)], + ), + ) + if components is not None: + spec.components = components # ty: ignore[invalid-assignment] return fnv1.RunFunctionRequest( observed=fnv1.State( composite=fnv1.Resource( resource=resource.dict_to_struct( v1alpha1.ServingStack( metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud=cloud, # ty: ignore[invalid-argument-type] # cases pass values of the literal - stack=stack, # ty: ignore[invalid-argument-type] - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - v1alpha1.Secret( - type="GoogleApplicationCredentials", name="sa-secret", key="private_key" - ), - ], - gateway=v1alpha1.Gateway( - hostname=_GATEWAY_HOSTNAME, - clientCAs=[v1alpha1.ClientCA(name="eu", certificate=_CLIENT_CA)], - ), - ), + spec=spec, ).model_dump(exclude_none=True, mode="json") ), ), @@ -303,6 +306,7 @@ def _observed_pcs() -> dict[str, fnv1.Resource]: "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( _OBJECT_REF, "gateway-selfsigned-issuer", _RELEASE_REF, "trust-manager" ), + "usage-envoy-gateway-by-gateway-proxy": _usage(_RELEASE_REF, "envoy-gateway", _OBJECT_REF, "gateway-proxy"), "usage-kai-scheduler-by-kai-queue-root": _usage(_RELEASE_REF, "kai-scheduler", _OBJECT_REF, "kai-queue-root"), "usage-kai-scheduler-by-kai-queue": _usage(_RELEASE_REF, "kai-scheduler", _OBJECT_REF, "kai-queue"), "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage( @@ -432,6 +436,7 @@ def _existing_dynamo_stack() -> dict[str, fnv1.Resource]: chart="gateway-helm", repository="oci://docker.io/envoyproxy", version="v1.8.4", + wait=True, values={ "config": { "envoyGateway": { @@ -828,7 +833,11 @@ def _existing_dynamo_stack() -> dict[str, fnv1.Resource]: return out -def _response(resources: dict[str, fnv1.Resource], status: dict | None = None) -> fnv1.RunFunctionResponse: +def _response( + resources: dict[str, fnv1.Resource], + status: dict | None = None, + conditions: list[fnv1.Condition] | None = None, +) -> fnv1.RunFunctionResponse: """A whole expected response: 60s TTL, empty context, the XR status.""" return fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), @@ -837,6 +846,7 @@ def _response(resources: dict[str, fnv1.Resource], status: dict | None = None) - resources=resources, ), context=structpb.Struct(), + conditions=conditions or [], ) @@ -863,7 +873,7 @@ async def test_compose(self) -> None: dep_gated = { "envoy-gateway", # -> cert-manager "ai-gateway", # -> ai-gateway-crds - "gateway-proxy", # -> gateway-namespace + "gateway-proxy", # -> gateway-namespace, envoy-gateway "kai-queue-root", # -> kai-scheduler "kai-queue", # -> kai-scheduler "modelexpress-server", # -> modelexpress-crds @@ -1272,6 +1282,280 @@ async def test_cluster_gateway_without_ca_serves_nothing(self) -> None: ) +# --- Provided mode (spec.components: Provided, Existing clusters) --- +# +# The substrate isn't composed; its requirement checks are, as +# observe-only Objects. Expectations are literals typed here, same as +# the Managed cases above. + + +_APISERVICE_AVAILABLE_CEL = ( + 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")' +) + + +def _require_api(key: str, apiservice: str) -> fnv1.Resource: + """The expected observe Object for a RequiredCRD, from literals. + + apiservice is the aggregated APIService name, .. + """ + return _object( + key, + {"apiVersion": "apiregistration.k8s.io/v1", "kind": "APIService", "metadata": {"name": apiservice}}, + cel=_APISERVICE_AVAILABLE_CEL, + management_policies=["Observe"], + ) + + +def _provided_checks() -> dict[str, fnv1.Resource]: + """Every requirement check the Existing/Dynamo stack renders.""" + return { + "require-cert-manager-crds": _require_api("require-cert-manager-crds", "v1.cert-manager.io"), + "require-kube-prometheus-stack-podmonitors": _require_api( + "require-kube-prometheus-stack-podmonitors", "v1.monitoring.coreos.com" + ), + "require-kube-prometheus-stack-servicemonitors": _require_api( + "require-kube-prometheus-stack-servicemonitors", "v1.monitoring.coreos.com" + ), + "require-node-feature-discovery-nodefeatures": _require_api( + "require-node-feature-discovery-nodefeatures", "v1alpha1.nfd.k8s-sigs.io" + ), + "require-nvidia-dra-driver-gpu-deviceclass": _object( + "require-nvidia-dra-driver-gpu-deviceclass", + {"apiVersion": "resource.k8s.io/v1", "kind": "DeviceClass", "metadata": {"name": "gpu.nvidia.com"}}, + cel="has(object.metadata.name)", + management_policies=["Observe"], + ), + "require-envoy-gateway-gatewayclasses": _require_api( + "require-envoy-gateway-gatewayclasses", "v1.gateway.networking.k8s.io" + ), + "require-envoy-gateway-gateways": _require_api( + "require-envoy-gateway-gateways", "v1.gateway.networking.k8s.io" + ), + "require-envoy-gateway-httproutes": _require_api( + "require-envoy-gateway-httproutes", "v1.gateway.networking.k8s.io" + ), + "require-envoy-gateway-envoyproxies": _require_api( + "require-envoy-gateway-envoyproxies", "v1alpha1.gateway.envoyproxy.io" + ), + "require-envoy-gateway-backends": _require_api( + "require-envoy-gateway-backends", "v1alpha1.gateway.envoyproxy.io" + ), + "require-envoy-gateway-ctp": _require_api("require-envoy-gateway-ctp", "v1alpha1.gateway.envoyproxy.io"), + "require-ai-gateway-crds-routes": _require_api( + "require-ai-gateway-crds-routes", "v1alpha1.aigateway.envoyproxy.io" + ), + "require-gaie-crds-pools": _require_api("require-gaie-crds-pools", "v1.inference.networking.k8s.io"), + "require-grove-podcliquesets": _require_api("require-grove-podcliquesets", "v1alpha1.grove.io"), + "require-kai-scheduler-queues": _require_api("require-kai-scheduler-queues", "v2.scheduling.run.ai"), + "require-trust-manager-bundles": _require_api( + "require-trust-manager-bundles", "v1alpha1.trust.cert-manager.io" + ), + } + + +# The config components Modelplane still composes in Provided mode, by +# composed-resource key, plus the gateway pair. +_PROVIDED_CONFIG_KEYS = frozenset( + { + "gateway-namespace", + "gateway-selfsigned-issuer", + "gateway-proxy", + "kai-queue-root", + "kai-queue", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-server-sa", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-svc", + "modelexpress-server", + "gateway-class", + "gateway", + } +) + +# The Usages that survive in Provided mode: config-to-config edges and +# the hand-written Gateway -> GatewayClass edge. Every edge onto skipped +# substrate (kai-scheduler, envoy-gateway, cert-manager, the AI gateway +# CRDs) is gone - there is nothing composed to hold. +# The gateway PKI is hand-rendered, not stack data; in Provided mode it +# waits for the cert-manager and trust-manager checks, so it renders in +# the same wave as their dependents. +_PROVIDED_PKI_KEYS = frozenset( + { + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + } +) + +_PROVIDED_USAGES = { + "usage-gateway-class-by-gateway": _usage(_OBJECT_REF, "gateway-class", _OBJECT_REF, "gateway"), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + _OBJECT_REF, "gateway-namespace", _OBJECT_REF, "gateway-selfsigned-issuer" + ), + "usage-gateway-namespace-by-gateway-proxy": _usage(_OBJECT_REF, "gateway-namespace", _OBJECT_REF, "gateway-proxy"), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage( + _OBJECT_REF, "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", _OBJECT_REF, "modelexpress-server" + ), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _usage( + _OBJECT_REF, "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", _OBJECT_REF, "modelexpress-server" + ), +} + + +def _requirements_condition(status: "fnv1.Status", reason: str, message: str = "") -> fnv1.Condition: + c = fnv1.Condition(type="RequirementsMet", status=status, reason=reason) + if message: + c.message = message + return c + + +class TestProvided(unittest.IsolatedAsyncioTestCase): + maxDiff = None + + @classmethod + def setUpClass(cls) -> None: + cls.runner = fn.FunctionRunner() + + async def test_compose(self) -> None: + stack = _existing_dynamo_stack() + config = {k: v for k, v in stack.items() if k in _PROVIDED_CONFIG_KEYS} + pki = {k: v for k, v in stack.items() if k in _PROVIDED_PKI_KEYS} + + # Second pass: PCs observed. The checks and the dependency-free + # config wave render; gateway-proxy waits on the envoy-gateway + # checks, the Queues on the kai-scheduler check, the + # ModelExpress server on its CRDs, the self-signed Issuer and + # the PKI on the cert-manager (and trust-manager) checks. + dep_gated = { + "gateway-proxy", # -> gateway-namespace, envoy-gateway checks + "gateway-selfsigned-issuer", # -> cert-manager check, gateway-namespace + "kai-queue-root", # -> kai-scheduler check + "kai-queue", # -> kai-scheduler check + "modelexpress-server", # -> modelexpress-crds + } + first_wave = {k: v for k, v in config.items() if k not in dep_gated} + + # Third pass: every check and every config resource observed + # Ready (the gateway with its address), so the whole desired + # state is marked ready and the condition is met. + observed_ready = _observed_pcs() + for key in list(_provided_checks()) + [k for k in (config | pki) if k != "gateway"]: + observed_ready[key] = fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ) + observed_ready["gateway"] = fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "conditions": [{"type": "Ready", "status": "True"}], + "atProvider": { + "manifest": {"status": {"addresses": [{"type": "IPAddress", "value": "203.0.113.7"}]}}, + }, + }, + } + ) + ) + all_ready = copy.deepcopy(_provider_configs() | _PROVIDED_USAGES | _provided_checks() | config | pki) + for res in all_ready.values(): + res.ready = fnv1.READY_TRUE + + # Failing pass: every check observed Ready except the KAI Queue + # CRD, whose observe failed (Synced False: the CRD doesn't + # exist). The Queues never compose and the condition names it. + observed_missing = _observed_pcs() + for key in _provided_checks(): + observed_missing[key] = fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ) + observed_missing["require-kai-scheduler-queues"] = fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "conditions": [ + {"type": "Synced", "status": "False", "reason": "ReconcileError"}, + {"type": "Ready", "status": "False"}, + ], + }, + } + ) + ) + checks_missing = copy.deepcopy(_provided_checks()) + for key, res in checks_missing.items(): + if key != "require-kai-scheduler-queues": + res.ready = fnv1.READY_TRUE + + cases = [ + Case( + name="first pass composes only the provider configs and config usages, checking", + req=_request("Existing", "Dynamo", components="Provided"), + want=_response( + _provider_configs(ready=False) | _PROVIDED_USAGES, + conditions=[ + _requirements_condition( + fnv1.STATUS_CONDITION_FALSE, + "Checking", + "Checking the cluster provides the serving substrate", + ), + ], + ), + ), + Case( + name="second pass renders the checks and the dependency-free config wave", + req=_request("Existing", "Dynamo", observed=_observed_pcs(), components="Provided"), + want=_response( + _provider_configs() | _PROVIDED_USAGES | _provided_checks() | first_wave, + conditions=[ + _requirements_condition( + fnv1.STATUS_CONDITION_FALSE, + "Checking", + "Checking the cluster provides the serving substrate", + ), + ], + ), + ), + Case( + name="a failed check blocks its dependents and names what's missing", + req=_request("Existing", "Dynamo", observed=observed_missing, components="Provided"), + # The cert-manager and trust-manager checks are Ready, so + # the PKI renders alongside the still-gated first wave. + want=_response( + _provider_configs() | _PROVIDED_USAGES | checks_missing | first_wave | pki, + conditions=[ + _requirements_condition( + fnv1.STATUS_CONDITION_FALSE, + "MissingRequirements", + "The cluster is missing: kai-scheduler (API v2.scheduling.run.ai not served)", + ), + ], + ), + ), + Case( + name="all checks and config ready marks everything ready and the requirements met", + req=_request("Existing", "Dynamo", observed=observed_ready, components="Provided"), + want=_response( + all_ready, + status={"gateway": {"address": "203.0.113.7"}}, + conditions=[_requirements_condition(fnv1.STATUS_CONDITION_TRUE, "RequirementsMet")], + ), + ), + ] + for case in cases: + with self.subTest(case.name): + got = await self.runner.RunFunction(case.req, None) + self.assertEqual( + json_format.MessageToDict(case.want), + json_format.MessageToDict(got), + "-want, +got", + ) + + # The composed-resource key a component renders under is its identity: # renaming one deletes and recreates the remote resource (for an Object # holding a CRD, the CRD and its CRs). This pins the full key set per @@ -1316,6 +1600,7 @@ async def test_cluster_gateway_without_ca_serves_nothing(self) -> None: "usage-ai-gateway-crds-by-ai-gateway", "usage-cert-manager-by-envoy-gateway", "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-envoy-gateway-by-gateway-proxy", "usage-gateway-namespace-by-gateway-proxy", "usage-gateway-namespace-by-gateway-selfsigned-issuer", "usage-gateway-selfsigned-issuer-by-trust-manager", @@ -1429,6 +1714,71 @@ async def test_cluster_gateway_without_ca_serves_nothing(self) -> None: "Existing": _HAND_WRITTEN, } +# Provided-mode inventories (Existing only): the substrate keys are +# replaced by their require-* checks, whose keys are identity too - +# renaming one only churns control-plane Objects, but keep it reviewed. +# Config components and the gateway pair stay; the usage-envoy-gateway +# edges and every edge onto skipped substrate are gone. +_PROVIDED_COMMON = frozenset( + { + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "gateway-selfsigned-issuer", + "require-trust-manager-bundles", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-class-by-gateway", + "require-cert-manager-crds", + "require-kube-prometheus-stack-podmonitors", + "require-kube-prometheus-stack-servicemonitors", + "require-node-feature-discovery-nodefeatures", + "require-nvidia-dra-driver-gpu-deviceclass", + "require-envoy-gateway-gatewayclasses", + "require-envoy-gateway-gateways", + "require-envoy-gateway-httproutes", + "require-envoy-gateway-envoyproxies", + "require-envoy-gateway-backends", + "require-envoy-gateway-ctp", + "require-ai-gateway-crds-routes", + "require-gaie-crds-pools", + "gateway-namespace", + "gateway-proxy", + "usage-gateway-namespace-by-gateway-proxy", + } +) + +_PROVIDED_STANDARD = frozenset( + { + "require-leader-worker-set-lws", + } +) + +_PROVIDED_DYNAMO = frozenset( + { + "require-grove-podcliquesets", + "require-kai-scheduler-queues", + "kai-queue", + "kai-queue-root", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + } +) + class TestKeyInventory(unittest.IsolatedAsyncioTestCase): maxDiff = None @@ -1455,3 +1805,19 @@ async def test_composed_resource_keys(self) -> None: ) got = await self.runner.RunFunction(_request(cloud, stack, observed=observed), None) self.assertEqual(expected, set(got.desired.resources.keys())) + + async def test_provided_composed_resource_keys(self) -> None: + for stack, stack_keys in (("Standard", _PROVIDED_STANDARD), ("Dynamo", _PROVIDED_DYNAMO)): + with self.subTest(stack=stack): + expected = _PROVIDED_COMMON | stack_keys + observed = _observed_pcs() + for key in expected: + observed[key] = fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"conditions": [{"type": "Ready", "status": "True"}]}} + ) + ) + got = await self.runner.RunFunction( + _request("Existing", stack, observed=observed, components="Provided"), None + ) + self.assertEqual(expected, set(got.desired.resources.keys())) diff --git a/functions/compose-serving-stack/tests/test_stacks.py b/functions/compose-serving-stack/tests/test_stacks.py index f17464637..9381d8730 100644 --- a/functions/compose-serving-stack/tests/test_stacks.py +++ b/functions/compose-serving-stack/tests/test_stacks.py @@ -124,6 +124,29 @@ def check(node: object, where: str) -> None: with self.subTest(cloud=cloud, stack=stack, key=c.key): check(c.values if isinstance(c, stacks.Chart) else c.manifests, c.key) + def test_required_crd_versions_are_singular(self) -> None: + # A RequiredCRD renders as one APIService observe Object, which + # encodes exactly one group-version. join() fails closed on any + # other count; this documents it against the real lists. + for cloud in stacks.clouds(): + for stack in stacks.stacks(): + for c in stacks.join(cloud, stack): + for r in c.requires: + if isinstance(r, stacks.RequiredCRD): + with self.subTest(cloud=cloud, stack=stack, key=c.key, req=r.key): + self.assertEqual(1, len(r.versions)) + + def test_existing_substrate_components_state_requirements(self) -> None: + # Provided mode can only select Existing, and there every + # substrate component must say what the cluster supplies in its + # place, or state why nothing is checkable. join() fails closed + # on this; the test documents it against the real lists. + for stack in stacks.stacks(): + for c in stacks.join("Existing", stack): + if c.role == "substrate": + with self.subTest(stack=stack, key=c.key): + self.assertTrue(c.requires or c.unchecked) + def test_unknown_cloud_and_stack_fail_closed(self) -> None: # The Literal types reject these at type-checking time; this # exercises the runtime guard behind them, which catches the API diff --git a/nix/apps.nix b/nix/apps.nix index 48c62ff1f..0e5c52eac 100644 --- a/nix/apps.nix +++ b/nix/apps.nix @@ -325,6 +325,9 @@ pkgs.gawk pkgs.kind pkgs.kubectl + # --provided pre-installs the substrate charts on the + # workload cluster (see e2e/provided/). + pkgs.kubernetes-helm pkgs.curl pkgs.docker-client pkgs.git @@ -350,6 +353,28 @@ # PATH matches its pin, so bumping aicr means updating nix/aicr.nix and # generate.py together. Extra args name the clouds to regenerate, e.g.: # nix run .#stacks -- gke + # Regenerate the serving stack requirements docs page + # (docs/content/platform/serving-stack-requirements.md) from the + # requirement data the component lists carry. The + # requirements-doc-current flake check fails CI when the page is + # stale, so run this after changing the stack data. + requirements-doc = _: { + type = "app"; + meta.description = "Regenerate the serving stack requirements docs page"; + program = pkgs.lib.getExe ( + pkgs.writeShellApplication { + name = "modelplane-requirements-doc"; + runtimeInputs = [ + (pkgs.python312.withPackages (ps: [ ps.pyyaml ])) + ]; + inheritPath = false; + text = '' + python3 functions/compose-serving-stack/requirements_doc.py + ''; + } + ); + }; + stacks = { aicr }: { diff --git a/nix/checks.nix b/nix/checks.nix index 29993c8af..e6ba083ea 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -142,6 +142,33 @@ in touch $out/.stacks-current ''; + # Fail if the generated serving stack requirements outputs are + # stale: regenerate the docs page and the e2e substrate inputs from + # the component lists' requirement data and diff against what's + # checked in. Catches a stack-data change committed without + # `nix run .#requirements-doc` and a hand-edit to a generated file. + # Runs the generator twice so non-determinism fails here rather than + # flapping CI. + requirements-doc-current = + pkgs.runCommand "modelplane-requirements-doc-current" + { + nativeBuildInputs = [ + (pkgs.python312.withPackages (ps: [ ps.pyyaml ])) + ]; + } + '' + cp -r ${self} src + chmod -R u+w src + cd src + python3 functions/compose-serving-stack/requirements_doc.py + python3 functions/compose-serving-stack/requirements_doc.py + diff -u ${self}/docs/content/platform/serving-stack-requirements.md \ + docs/content/platform/serving-stack-requirements.md + diff -ru ${self}/e2e/provided e2e/provided + mkdir -p $out + touch $out/.requirements-doc-current + ''; + # Fail if any hand-written source file is missing its Apache 2.0 license # header. Scoped to the files we author: the composition functions and the # docs manifest validator. Generated models under schemas/python carry their diff --git a/pyproject.toml b/pyproject.toml index d808d3ae1..685d70f65 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -86,6 +86,8 @@ allow-star-arg-any = true # The serving stack generator is a CLI script: its classification report # prints to stderr, and its recipe transform is legitimately branchy. "functions/compose-serving-stack/generate.py" = ["T201", "PLR0912", "PLR0915"] +# The requirements docs generator is a CLI script; print is its output. +"functions/compose-serving-stack/requirements_doc.py" = ["T201"] [tool.ty.environment] # The functions target Python 3.12 (see tool.ruff target-version). diff --git a/schemas/.lock.json b/schemas/.lock.json index f6003c682..e94a302e0 100644 --- a/schemas/.lock.json +++ b/schemas/.lock.json @@ -1,6 +1,6 @@ { "packages": { - "fs://apis": "a3c8867fa02ab5c25944ea69b769d69c5b8de1cf1d21ca61fc9c558159c7bf25", + "fs://apis": "5b7ce5b1445fc7587237cd7095423c6d25f55b93a82ebed9a2d66fba9d1fce87", "git://https://github.com/crossplane/crossplane/cluster/crds": "90d8b72ad8b829f0bcd7d7d5a98eaa0d579f244a", "xpkg://xpkg.upbound.io/upbound/provider-aws-ec2:v2.8.1": "sha256:ca2e9e3b2e3a8b6ca44a9700d5abf7abd733cfa388d1afe9bb2bf7c847cd39ef", "xpkg://xpkg.upbound.io/upbound/provider-aws-efs:v2.8.1": "sha256:ebb1bcd8dc9a7e60e97a324652fbd9d1b0609ddbb1107db518ce999cbc1113bd", diff --git a/schemas/python/models/ai/modelplane/inferencecluster/v1alpha1.py b/schemas/python/models/ai/modelplane/inferencecluster/v1alpha1.py index da447c98f..101dc7248 100644 --- a/schemas/python/models/ai/modelplane/inferencecluster/v1alpha1.py +++ b/schemas/python/models/ai/modelplane/inferencecluster/v1alpha1.py @@ -95,6 +95,10 @@ class Existing(BaseModel): """ ModelCache configuration for this cluster. """ + components: Literal['Managed', 'Provided'] | None = 'Managed' + """ + Who supplies the serving substrate on this cluster. Managed (the default) has Modelplane install every serving stack component. Provided installs no substrate: the cluster already runs cert-manager, the gateway stack, Prometheus, the GPU DRA driver and the stack's workload controller, and Modelplane verifies they are present (the RequirementsMet condition reports what's missing) and composes only its own configuration on top. All or nothing; the cluster provides the whole substrate or none of it. Immutable because flipping it would uninstall a live cluster's substrate or adopt one Modelplane doesn't own. + """ identitySecretRef: IdentitySecretRef | None = None """ Optional reference to a Secret containing cloud provider credentials for IAM-based authentication. The type selects which cloud identity the ProviderConfigs authenticate as, and must match the cloud the existing cluster runs on. diff --git a/schemas/python/models/ai/modelplane/infrastructure/servingstack/v1alpha1.py b/schemas/python/models/ai/modelplane/infrastructure/servingstack/v1alpha1.py index c7aab827a..a467dcdaa 100644 --- a/schemas/python/models/ai/modelplane/infrastructure/servingstack/v1alpha1.py +++ b/schemas/python/models/ai/modelplane/infrastructure/servingstack/v1alpha1.py @@ -115,6 +115,10 @@ class Spec(BaseModel): """ The cloud the target cluster runs on. Selects the fixed set of components and versions this stack installs there, which is resolved per cloud at build time and changes only with a Modelplane release. Mirrors InferenceCluster.spec.cluster.source; the cluster composition sets it. """ + components: Literal['Managed', 'Provided'] | None = 'Managed' + """ + Who supplies the serving substrate. Managed (the default) has Modelplane install every component. Provided, valid only on an Existing cluster, installs no substrate charts or vendored CRDs: Modelplane verifies the cluster supplies them - reported through the RequirementsMet condition - and composes only its own configuration on top. All or nothing. Mirrors InferenceCluster.spec.cluster.existing.components; the cluster composition sets it. + """ crossplane: Crossplane | None = None """ Configures how Crossplane will reconcile this composite resource