diff --git a/apis/metricmappings/composition.yaml b/apis/metricmappings/composition.yaml new file mode 100644 index 000000000..8359dc414 --- /dev/null +++ b/apis/metricmappings/composition.yaml @@ -0,0 +1,13 @@ +apiVersion: apiextensions.crossplane.io/v1 +kind: Composition +metadata: + name: metricmappings.modelplane.ai +spec: + compositeTypeRef: + apiVersion: modelplane.ai/v1alpha1 + kind: MetricMapping + mode: Pipeline + pipeline: + - functionRef: + name: modelplane-modelplanecompose-metric-mapping + step: compose-metric-mapping diff --git a/apis/metricmappings/definition.yaml b/apis/metricmappings/definition.yaml new file mode 100644 index 000000000..63b512801 --- /dev/null +++ b/apis/metricmappings/definition.yaml @@ -0,0 +1,179 @@ +apiVersion: apiextensions.crossplane.io/v2 +kind: CompositeResourceDefinition +metadata: + name: metricmappings.modelplane.ai +spec: + group: modelplane.ai + names: + categories: [crossplane, modelplane, platform] + kind: MetricMapping + plural: metricmappings + shortNames: [mm] + scope: Cluster + versions: + - name: v1alpha1 + served: true + referenceable: true + additionalPrinterColumns: + - name: CLUSTERS + type: integer + jsonPath: .status.clusters + - name: AGE + type: date + jsonPath: .metadata.creationTimestamp + schema: + openAPIV3Schema: + type: object + required: [spec] + properties: + spec: + type: object + required: [metrics] + description: >- + How one component's metrics become part of the modelplane_* + surface. Modelplane renders every MetricMapping into each + inference cluster's collector, so a mapping is written once on + the control plane and reaches the whole fleet. + + A mapping naming a component Modelplane already provides + renames for is additive: its renames run after the built-in + ones, on whatever those left behind. A metric a built-in + already renamed no longer answers to the name it was emitted + under, so a second mapping selecting on that name matches + nothing and the built-in stands. Select on the `modelplane_` + name instead to rename one of Modelplane's own. + properties: + metrics: + type: array + minItems: 1 + maxItems: 128 + description: >- + The metrics to rename. Give two components' metrics the same + name only if they measure the same thing, and histograms only + if their buckets match too. A quantile across mismatched + buckets is wrong. + items: + type: object + required: [from, to] + properties: + from: + type: string + maxLength: 255 + pattern: '^[a-zA-Z_:][a-zA-Z0-9_:]*$' + description: >- + The metric's name as the component exposes it, matched + exactly wherever it appears in the fleet. + + A histogram is named by its base name, without the + _count, _sum or _bucket a Prometheus query would use: + the collector holds it as one metric, and `part` is + what reaches into it. + to: + type: string + maxLength: 255 + pattern: '^modelplane_[a-z0-9_]*[a-z0-9]$' + description: >- + The name to export the metric under. Only modelplane_* + metrics leave a cluster, so a metric no mapping renames + never leaves its cluster. + + Name it in base units - seconds, bytes, joules, a ratio + from nought to one - because that is what `fromUnit` + converts to. + part: + type: string + enum: [Count, Sum] + description: >- + Take a part of a histogram as a counter of its own, + rather than the histogram itself. Count is how many + observations it holds, which is a request count where + the histogram measures request duration. Sum is their + total. + + The extraction leaves the histogram alone, but the + collector exports only what a mapping renames, so the + histogram itself is dropped unless another mapping + gives it a `modelplane_` name of its own. Write that + second mapping to keep both. + labels: + type: array + maxItems: 16 + description: >- + Labels to set on this metric's series. To fold several + metrics into one name, give each its own entry with + the same `to` and a different fixed value, such as + direction: input and direction: output on + modelplane_tokens_total. + items: + type: object + required: [name] + x-kubernetes-validations: + - rule: "has(self.value) != has(self.from)" + message: set either value, for a fixed label, or from, to carry one the component already emits. + - rule: "!has(self.values) || has(self.from)" + message: values remaps what from carries, so it needs from. A fixed value has nothing to remap. + properties: + name: + type: string + maxLength: 63 + pattern: '^[a-zA-Z_][a-zA-Z0-9_]*$' + description: The label to set. + value: + type: string + maxLength: 253 + description: >- + A fixed value, the same on every series this + mapping produces. This is what tells two folded + metrics apart. + from: + type: string + maxLength: 63 + pattern: '^[a-zA-Z_][a-zA-Z0-9_]*$' + description: >- + A label the component already emits. Modelplane + copies its value into this label and removes the + original. + + Naming the label itself keeps it: that is how + `values` rewrites what a component writes without + renaming the label. + values: + type: object + maxProperties: 32 + additionalProperties: + type: string + maxLength: 253 + description: >- + What each of that label's values becomes, for + putting an engine's own vocabulary into + Modelplane's. A value with no entry here is left + as the component wrote it. + fromUnit: + type: string + enum: [Millijoules, Mebibytes, Milliseconds, Nanoseconds, Percent] + description: >- + What the component measures this in, when that isn't + the unit the name claims. Modelplane converts to the + base unit: millijoules and milliseconds are divided by + a thousand, nanoseconds by a billion, percent by a + hundred, and mebibytes multiplied out to bytes. A + histogram is converted whole - its sum, its bounds and + its bucket boundaries - so its quantiles come out in + the target unit too. + + Percent is for a component that counts a saturation + from nought to a hundred where the name says a ratio. + Check rather than assume: vLLM publishes + kv_cache_usage_perc and the value is a fraction, so a + name is no guide. + status: + type: object + properties: + clusters: + type: integer + description: >- + How many inference clusters apply this mapping. + conditions: + type: array + items: + type: object diff --git a/apis/telemetrydestinations/composition.yaml b/apis/telemetrydestinations/composition.yaml new file mode 100644 index 000000000..b345a3d7b --- /dev/null +++ b/apis/telemetrydestinations/composition.yaml @@ -0,0 +1,13 @@ +apiVersion: apiextensions.crossplane.io/v1 +kind: Composition +metadata: + name: telemetrydestinations.modelplane.ai +spec: + compositeTypeRef: + apiVersion: modelplane.ai/v1alpha1 + kind: TelemetryDestination + mode: Pipeline + pipeline: + - functionRef: + name: modelplane-modelplanecompose-telemetry-destination + step: compose-telemetry-destination diff --git a/apis/telemetrydestinations/definition.yaml b/apis/telemetrydestinations/definition.yaml new file mode 100644 index 000000000..0d24e9a83 --- /dev/null +++ b/apis/telemetrydestinations/definition.yaml @@ -0,0 +1,160 @@ +apiVersion: apiextensions.crossplane.io/v2 +kind: CompositeResourceDefinition +metadata: + name: telemetrydestinations.modelplane.ai +spec: + group: modelplane.ai + names: + categories: [crossplane, modelplane, platform] + kind: TelemetryDestination + plural: telemetrydestinations + shortNames: [td] + scope: Cluster + versions: + - name: v1alpha1 + served: true + referenceable: true + additionalPrinterColumns: + - name: AGE + type: date + jsonPath: .metadata.creationTimestamp + schema: + openAPIV3Schema: + type: object + required: [spec] + properties: + spec: + type: object + required: [sinks] + description: >- + Where the fleet's metrics go. Modelplane runs no collectors + until a TelemetryDestination exists, and creating one turns on + collection on every inference cluster. + + Several can exist. Their sinks are concatenated into one + collector configuration, so adding a backend is a new object + rather than an edit to one somebody else owns. A sink's name is + the collector's name for its exporter, so it has to be unique + across destinations. + properties: + sinks: + type: array + minItems: 1 + maxItems: 16 + description: >- + Where to send it. Every sink gets the whole stream, so two + sinks is two copies of the fleet's metrics, billed twice. + x-kubernetes-list-type: map + x-kubernetes-list-map-keys: [name] + items: + type: object + required: [name, type] + x-kubernetes-validations: + - rule: "!has(self.auth) || has(self.secretRef)" + message: secretRef is required when auth is set, because the credential lives in it. + properties: + name: + type: string + maxLength: 63 + pattern: '^[a-z0-9]([-a-z0-9]*[a-z0-9])?$' + description: >- + This sink's name, unique within the destination. It + names the collector's exporter instance, the + authenticator Modelplane composes for it, and the + directory its credential mounts at, so renaming one + restarts the collector. + type: + type: string + maxLength: 63 + description: >- + The collector exporter to send with, by the name + OpenTelemetry gives it: otlphttp, otlp, + prometheus_remote_write, kafka, and every other one the + collector provides. + + Not an enum: the collector already refuses to start + on a name it doesn't have, so repeating the list here + would only add a second place for it to go stale. + endpoint: + type: string + maxLength: 2048 + description: >- + Where this sink writes. Leave it unset for an exporter + that doesn't take an endpoint, such as kafka or debug, + and configure it in `config` instead. + + Set here, it wins: Modelplane applies it over + `config`, so a sink cannot be quietly redirected by + the configuration passed through beside it. + auth: + type: object + description: >- + Authentication Modelplane sets up for this sink, using + a credential from `secretRef`. For another scheme, + define an authenticator under spec.extensions and + reference it from `config`. + + Set here, it wins: Modelplane applies it over an + `auth` block in `config`, so a sink's credential + cannot be quietly unpicked. + properties: + bearerTokenKey: + type: string + maxLength: 253 + description: >- + The key in this sink's Secret holding the bearer + token. Modelplane mounts it as a file and points + the authenticator at it, so a rotated token is + picked up without restarting the collector. + config: + type: object + x-kubernetes-preserve-unknown-fields: true + description: >- + The rest of the exporter's configuration, passed + through as written: TLS, retries, queueing, + compression, headers. + + Modelplane sets one default, for the exporters that + flatten a series into labels: prometheus and + prometheus_remote_write get + resource_to_telemetry_conversion, or they would + receive every series stripped of the cluster, + deployment, engine and role it belongs to. Setting it + here overrides that. + secretRef: + type: object + required: [name] + description: >- + A Secret holding this sink's credentials. Modelplane + mounts each key as a file under + /etc/modelplane/telemetry// and sets it as + an environment variable for ${env:KEY} references in + `config`. All sinks share one environment, so where + two Secrets have the same key, refer to the file. + properties: + name: + type: string + maxLength: 253 + description: >- + Name of the Secret, in modelplane-system on the + control plane. Modelplane copies it to every + cluster running a collector, so it doesn't have + to exist on each of them already. + extensions: + type: object + x-kubernetes-preserve-unknown-fields: true + description: >- + Collector extensions, passed through as written. Use it to + define an authenticator that a sink's `config` references. + + An exporter needs a client authenticator - basicauth, + oauth2client, sigv4auth, headers_setter. The oidc extension + authenticates callers of a receiver, so it is not one of + these. + status: + type: object + properties: + conditions: + type: array + items: + type: object diff --git a/crossplane-project.yaml b/crossplane-project.yaml index b4eb40d39..532b15c58 100644 --- a/crossplane-project.yaml +++ b/crossplane-project.yaml @@ -75,6 +75,14 @@ spec: tarball: name: compose-model-service pathPrefix: _output/functions/compose-model-service + - source: Tarball + tarball: + name: compose-metric-mapping + pathPrefix: _output/functions/compose-metric-mapping + - source: Tarball + tarball: + name: compose-telemetry-destination + pathPrefix: _output/functions/compose-telemetry-destination - source: Tarball tarball: name: compose-usages diff --git a/docs/content/guides/collecting-engine-metrics.md b/docs/content/guides/collecting-engine-metrics.md deleted file mode 100644 index ae6ecff51..000000000 --- a/docs/content/guides/collecting-engine-metrics.md +++ /dev/null @@ -1,67 +0,0 @@ ---- -title: Collecting engine metrics -weight: 20 -description: Scrape a vLLM engine's Prometheus metrics through the in-cluster Prometheus. ---- - -Scraping an inference engine's Prometheus metrics, shown on the smallest serving -shape: a 0.5B Qwen chat model on one NVIDIA L4. vLLM publishes metrics at -`/metrics` on its serving port with no extra flag, and Modelplane runs a -Prometheus on every workload cluster with `PodMonitor` discovery open across -namespaces, so scraping the engine is a `PodMonitor` plus a `port-forward`. The -model is only the subject; the same wiring fits any engine, with the SGLang, -leader/worker, and prefill/decode differences noted at the end. - -This was run end to end on GKE. The `InferenceClass` and `ModelDeployment` are the -exact manifests from that run, and the `PodMonitor` below scraped this deployment. -Apply the platform side first, then the ML side. - -## Platform - -{{< manifests "guides/collecting-engine-metrics/inference-class.yaml" >}} - -{{< manifests "guides/collecting-engine-metrics/inference-cluster.yaml" >}} - -## Deployment - -{{< manifests "guides/collecting-engine-metrics/model-deployment.yaml" >}} - -{{< manifests "guides/collecting-engine-metrics/model-service.yaml" >}} - -## Scraping the metrics - -The `PodMonitor` selects engine pods by the `modelplane.ai/serving` label -Modelplane stamps on them, and the `monitoring` namespace Prometheus discovers any -`PodMonitor`, so this is the whole config. The engine container port is unnamed, -so reference it by number with `targetPort`: - -{{< manifests "guides/collecting-engine-metrics/podmonitor.yaml" >}} - -The engine pods and the `PodMonitor` CRD live on the workload cluster, not the -control plane, so apply it there. The pods run in the namespace Modelplane -mirrors `ml-team` into, which `kubectl get ns -l modelplane.ai/namespace=ml-team` -finds. Then read the metrics from the in-cluster Prometheus over a -`port-forward`: - -```bash -kubectl -n monitoring port-forward svc/prometheus-prometheus 9090:9090 # workload cluster -# open http://localhost:9090, Status > Targets to confirm the scrape, then query -# e.g. vllm:num_requests_running or vllm:gpu_cache_usage_perc -``` - -### Other engine shapes - -The `PodMonitor` above fits a single-pod vLLM engine. The selector and port shift -by shape: - -- **SGLang**: exposes `/metrics` only when the engine runs with - `--enable-metrics`; otherwise it's identical (same selector, `targetPort: 8000`). -- **Leader/worker**: only the leader serves the API and carries - `modelplane.ai/serving`, so the selector above already scrapes the leader alone; - the workers expose nothing. -- **prefill/decode**: two engines, labelled `llm-d.ai/role: prefill` and - `llm-d.ai/role: decode`. The prefill engine serves on `8000`; the decode engine - sits behind the routing sidecar that takes `8000` and listens on `8001`, so - scrape decode with `targetPort: 8001`. Select each by its role label to keep - them apart. - diff --git a/docs/content/models/model-service.md b/docs/content/models/model-service.md index fec06d836..a366694c8 100644 --- a/docs/content/models/model-service.md +++ b/docs/content/models/model-service.md @@ -240,7 +240,7 @@ fails, so don't mix one into a service that OpenAI callers use. Scrape an engine's own operational paths like `/metrics` and `/health` from the replica directly. See -[Collecting engine metrics]({{< ref "/guides/collecting-engine-metrics" >}}). +[Monitor the Fleet]({{< ref "/platform/telemetry" >}}). ## Example diff --git a/docs/content/platform/telemetry.md b/docs/content/platform/telemetry.md new file mode 100644 index 000000000..cc5d0d560 --- /dev/null +++ b/docs/content/platform/telemetry.md @@ -0,0 +1,367 @@ +--- +title: Monitor the Fleet +weight: 37 +aliases: +- /guides/collecting-engine-metrics/ +- /guides/telemetry/ +description: Collect normalized metrics across the fleet and send them to any backend the collector can export to. +--- + + +Modelplane runs an OpenTelemetry collector on every inference cluster. It +collects from every component Modelplane installs. This includes the inference +server engine, inference gateway and Envoy proxy, router, and the GPU exporter +your cloud provides. It renames each component's series to a single +`modelplane_*` vocabulary and exports them to wherever you say - any backend the +collector has an exporter for, not only OTLP. + +Modelplane allows you to write one destination for your metrics. You don't need +to manage per-deployment configurations or update your configuration when a +deployment changes. The collector finds pods itself, so a leader/worker split or a +prefill/decode pair is collected the same as a single pod. + +## Telemetry workflow + +Every series carries `cluster`, `job`, and `instance` labels of the target +resource. A series about a deployment also carries `deployment`, `replica`, +`namespace`, `engine`, and `role` labels. + +Each replica publishes its own series, so combine them in your query. Which +combination is right follows from what the metric measures: + + - `sum by (deployment)`, for anything counted, such as requests, tokens, or queue depth. + - `avg by (deployment)`, for a ratio. + - `max by (deployment)`, for a saturation figure an alert fires on. + +To combine: + +```promql +sum by (deployment) (rate(modelplane_frontend_request_duration_seconds_count[5m])) +``` + +The replica is an index rather than a pod, so it's bounded by the replica count +and survives a restart and a rolling update. Group by `replica`, not by `instance`: + +```promql +# One line per replica, stable across rolling updates +max by (deployment, replica) (modelplane_kv_cache_utilization_ratio) +``` +`instance` is the pod's address, which keeps two pods of the same replica apart. It +turns over on every rolling update, so group by `replica` rather than by `instance`. + +Some examples of the available metrics: + +| Metric | Means | +| --- | --- | +| `modelplane_frontend_ttft_seconds` | Time to the first token, measured at the gateway | +| `modelplane_frontend_tpot_seconds` | Time per output token, measured at the gateway | +| `modelplane_frontend_request_duration_seconds` | What the caller waited, end to end | +| `modelplane_request_queue_seconds` | How long a request waited before the engine started | +| `modelplane_requests_waiting` | Queue depth per engine | +| `modelplane_kv_cache_utilization_ratio` | KV-cache occupancy, averaged over replicas | +| `modelplane_request_input_tokens` | Prompt size, as a histogram | +| `modelplane_request_output_tokens` | Generated length, as a histogram | +| `modelplane_gpu_memory_used_bytes` | Framebuffer memory in use, per GPU | +| `modelplane_energy_joules_total` | Energy drawn since the driver last reloaded | + +Latency appears twice on purpose. The `frontend_` series are what your caller experienced, +measured at the gateway. The engine's own series are what the engine spent. For +example, if the frontend metric is slow and the engine isn't, you can +troubleshoot routing, queueing, or networking issues instead of the model. + + +Saturation gauges are per replica, so how you combine them decides what you see. Average +across a deployment to plan capacity, and take the maximum to alert: three replicas at 0.3 +and one at 0.99 average to something comfortable while the fourth evicts and recomputes. A +high maximum beside `modelplane_requests_preempted_total` climbing is one replica thrashing. To alert on it: + +```promql +max by (deployment) (modelplane_kv_cache_utilization_ratio) > 0.95 +``` + +## Send telemetry to a destination + +Create a `TelemetryDestination` for your OpenTelemetry-compatible endpoint: + +```yaml +apiVersion: modelplane.ai/v1alpha1 +kind: TelemetryDestination +metadata: + name: default +spec: + sinks: + - name: primary + type: otlphttp + endpoint: https://otel.example.internal +``` + +`type` names a collector exporter, by the name OpenTelemetry gives it. + +To authenticate with a bearer token, store the token in a Secret in +`modelplane-system` on your control plane. Create it once: Modelplane copies it to +every cluster running a collector, so you don't put the credential on each GPU +cluster yourself. + +```shell +kubectl create secret generic telemetry-credentials \ + --namespace modelplane-system \ + --from-literal=token= +``` + +Reference the Secret from the sink with `secretRef`, and set `auth.bearerTokenKey` to +the key that holds the token: + +```yaml +spec: + sinks: + - name: primary + type: otlphttp + endpoint: https://otel.example.internal + secretRef: + name: telemetry-credentials + auth: + bearerTokenKey: token +``` + +Modelplane configures the collector to send the token with every export. Rotating the +token needs no restart. + +If you run Prometheus, export to your Prometheus endpoint instead and query the fleet there: + +```yaml +spec: + sinks: + - name: prometheus + type: prometheus_remote_write + endpoint: https://prom.example.internal/api/v1/write +``` + +If you create more than one sink, all get the entire stream. Each sink carries its +own credential, so a vendor and your own Prometheus don't have to share a Secret. + +```yaml +spec: + sinks: + - name: vendor + type: otlphttp + endpoint: https://otel.vendor.example + secretRef: + name: vendor-token + auth: + bearerTokenKey: token + - name: prometheus + type: prometheus_remote_write + endpoint: https://prom.example.internal/api/v1/write +``` + +That is two copies of the fleet's metrics, so a vendor charging per sample charges for +both. + +Sinks can also come from more than one `TelemetryDestination`. Modelplane concatenates +them, so a team adding an export creates its own object rather than editing one somebody +else owns. Sink names are what the collector calls its exporters, so they have to be +unique across destinations; where two collide, the destination whose name sorts first +keeps it and Modelplane says so on the `ServingStack`. + +Anything else the exporter takes goes under `config`, passed through as you wrote it: + +```yaml + - name: vendor + type: otlphttp + endpoint: https://otel.vendor.example + config: + compression: gzip + sending_queue: + queue_size: 10000 + tls: + ca_file: /etc/ssl/certs/internal.pem +``` +Modelplane doesn't define a schema for an exporter's settings so anything under +the `config` is passed to the collector exactly as written. TLS, retries, +queueing, compression and headers all work and any new settings in the collector +are respected and the sink keeps working. + +To use an authentication scheme Modelplane doesn't compose, define the extension +yourself under `spec.extensions`. Then reference it by its key from the sink's +`config.auth.authenticator`. The `auth` block does the same wiring for you when +you use a bearer token. + +```yaml +spec: + sinks: + - name: vendor + type: otlphttp + endpoint: https://otel.vendor.example + secretRef: + name: vendor-oauth + config: + auth: + authenticator: oauth2client/vendor + extensions: + oauth2client/vendor: + client_id: modelplane + client_secret: ${env:CLIENT_SECRET} + token_url: https://issuer.example/oauth2/token +``` + +Modelplane doesn't run any collectors until you create a +`TelemetryDestination`. + +Creating a destination turns collection on everywhere at once, and there's no per-deployment +opt-out. + +Your clusters reach the control plane, and only the control plane reaches your backend. A +cluster with no route to your observability stack still reports, and the backend's +credential lives in one place instead of on every GPU cluster. + +## Computing rates, quantiles, and ratios + +A collector transforms each measurement as it passes it on. It holds no history, so it +produces no rates and no quantiles. Your backend does that. A fleet-wide p99: + +```promql +histogram_quantile(0.99, sum by (le) ( + rate(modelplane_frontend_ttft_seconds_bucket{deployment="qwen3-8b"}[5m]))) +``` + +Modelplane has no dashboards of its own. What it exports is counters and histogram buckets, and +your backend derives the rates and quantiles at query time. To precompute them instead, +export to Prometheus and write recording rules there. + +## Engines + +Modelplane already knows vLLM's and SGLang's metric names and renames them for you, so +neither needs anything from you here. SGLang needs two flags: `--enable-metrics` to publish `/metrics` at all, and +`--collect-tokens-histogram` for the prompt and generation histograms behind +`modelplane_request_input_tokens` and `modelplane_request_output_tokens`. Without the +second it publishes those as plain counters and both series stay empty. vLLM needs +nothing. + +```yaml +apiVersion: modelplane.ai/v1alpha1 +kind: ModelDeployment +metadata: + name: my-sglang-model + namespace: ml-team +spec: + template: + spec: + engines: + - name: engine + members: + - role: Standalone + template: + spec: + containers: + - name: engine + image: lmsysorg/sglang:v0.5.10.post1-runtime + command: + - /bin/sh + - -c + - >- + exec python3 -m sglang.launch_server + --model-path + --host 0.0.0.0 + --port 8000 + --enable-metrics + --collect-tokens-histogram +``` + +SGLang publishes no queue time per request and no preemption counters, so +`modelplane_request_queue_seconds` and `modelplane_requests_preempted_total` carry vLLM +only. + +Any other OpenAI-compatible engine reports its frontend numbers with no configuration. +The gateway measures those, not the engine, so `modelplane_frontend_*` works for an +engine Modelplane has never seen. + +To normalize that engine's own metrics as well, create a `MetricMapping`: + +```yaml +apiVersion: modelplane.ai/v1alpha1 +kind: MetricMapping +metadata: + name: my-engine +spec: + metrics: + - from: my_engine_queued_requests + to: modelplane_requests_waiting + - from: my_engine_kv_transfer_ms + fromUnit: Milliseconds + to: modelplane_request_kv_transfer_seconds +``` + +Modelplane renders every mapping into every cluster's collector, so you write one once. +`from` is the name your engine emits and `to` is what Modelplane calls it. + +Modelplane leaves the combining to your backend. A scrape of one replica is one batch, so +a collector that added them up would be summing readings taken at different moments, and +two readings of one cumulative counter come to twice the traffic that happened. Your +backend holds every replica's series and combines them at query time. + +Say `fromUnit` whenever the engine measures in something other than the unit the name +claims, and Modelplane converts to the base one. Skipping this is the expensive mistake here: +a series named `_seconds` that holds milliseconds reads a thousand times fast, and nothing +downstream can tell. + +Rename only where the measurements agree. Two engines' histograms under one name are worth +less than nothing if their buckets disagree, because a quantile over them is wrong rather +than approximate. + +### Examples + +```yaml +apiVersion: modelplane.ai/v1alpha1 +kind: MetricMapping +metadata: + name: my-engine +spec: + metrics: + # A plain rename. + - from: my_engine_queued_requests + to: modelplane_requests_waiting + + # A unit conversion. The engine reports milliseconds; the name says seconds. + - from: my_engine_kv_transfer_ms + to: modelplane_request_kv_transfer_seconds + fromUnit: Milliseconds + + # A request count taken out of a duration histogram. The histogram + # keeps its own name; this adds a counter beside it. + - from: my_engine_request_duration_seconds + part: Count + to: modelplane_requests_total + + # Two counters folded into one name, told apart by a fixed label. + - from: my_engine_prompt_tokens_total + to: modelplane_tokens_total + labels: + - name: direction + value: input + - from: my_engine_generated_tokens_total + to: modelplane_tokens_total + labels: + - name: direction + value: output + + # A label the engine already emits, renamed and its values translated. + - from: my_engine_finished_requests_total + to: modelplane_responses_total + labels: + - name: reason + from: finish_reason + values: + eos: stop + max_tokens: length +``` + +## Why engine latency and gateway latency differ + +`modelplane_request_ttft_seconds` comes from the engine, and engines bucket their +histograms differently. vLLM resolves down to a millisecond. SGLang resolves to a hundred +of them. A quantile across both is wrong, not approximate. Use the engine series to compare +one engine against itself, and the `frontend_` series for anything fleet-wide. + +Some measurements don't translate at all. SGLang's inter-token latency isn't vLLM's time per +output token, so neither is renamed onto a shared name. The gateway measures time per output +token for both. diff --git a/docs/content/recipes/qwen2.5-72b.md b/docs/content/recipes/qwen2.5-72b.md index ca0fa63c8..e9bee702f 100644 --- a/docs/content/recipes/qwen2.5-72b.md +++ b/docs/content/recipes/qwen2.5-72b.md @@ -69,7 +69,7 @@ under the same model name: Both GPUs now serve the same workload, so their engine metrics give a direct performance comparison: scrape each replica's latency and throughput as in -[Collecting engine metrics]({{< ref "/guides/collecting-engine-metrics.md" >}}) and +[Monitor the Fleet]({{< ref "/platform/telemetry.md" >}}) and read the two side by side. Weights are relative, so once one platform wins, shift the 50/50 toward it - 80/20, and as far as 100/0 - without touching the deployment. diff --git a/docs/data/apigroups.yaml b/docs/data/apigroups.yaml index 9687c9303..fc508a746 100644 --- a/docs/data/apigroups.yaml +++ b/docs/data/apigroups.yaml @@ -31,6 +31,8 @@ concepts: InferenceGateway: /platform/inference-gateway InferenceClass: /platform/inference-class InferenceCluster: /platform/inference-cluster + MetricMapping: /platform/telemetry + TelemetryDestination: /platform/telemetry ModelDeployment: /models/model-deployment ModelService: /models/model-service ModelCache: /models/model-cache diff --git a/docs/utils/vale/styles/config/vocabularies/Modelplane/accept.txt b/docs/utils/vale/styles/config/vocabularies/Modelplane/accept.txt index c0efe996f..a5fc7bb87 100644 --- a/docs/utils/vale/styles/config/vocabularies/Modelplane/accept.txt +++ b/docs/utils/vale/styles/config/vocabularies/Modelplane/accept.txt @@ -21,6 +21,8 @@ EKSCluster ServingStack ModelCache ModelCaches +MetricMapping +MetricMappings ModelDeployment ModelDeployments ModelEndpoint @@ -29,6 +31,8 @@ ModelReplica ModelReplicas ModelService ModelServices +TelemetryDestination +TelemetryDestinations # Spec fields and config keys clusterSelector @@ -145,6 +149,8 @@ Baseten WekaIO NetApp minikube +Monitor +Aggregate # AI assistants and agent tooling MCP @@ -580,3 +586,15 @@ DGX BasePOD SuperPOD Run:ai +OTLP +OpenTelemetry +Grafana +quantile +quantiles +precompute +top-line +p99 +OTTL +per-deployment +opt-out +Framebuffer diff --git a/e2e/README.md b/e2e/README.md index d8bac4ee5..ff7aa6751 100644 --- a/e2e/README.md +++ b/e2e/README.md @@ -47,8 +47,25 @@ server exposes both, so the pod goes Ready without a real model or GPU. | `ModelDeployment` → `ModelReplica` → `ModelEndpoint` → `ModelService` wiring | Real GPU drivers / CUDA (fake DRA devices only) | | DRA `ResourceClaim` → fake device binding (the real allocation path) | Multi-node / disaggregated (`PrefillDecode`) serving | | Serving-stack install on a real (BYO) workload cluster | Cloud provisioning (EKS/GKE/Nebius) | -| `InferenceGateway` + cross-cluster routing to the replica | | +| `InferenceGateway` + cross-cluster routing to the replica | Export to a real metrics backend (debug sink only) | | Status propagation and foreground-deletion ordering | | +| Telemetry: collector composed, engine discovered, series renamed and attributed | | + +### Telemetry + +`60-telemetry.yaml` creates a `TelemetryDestination` with the collector's +**debug** exporter, which prints what reached it to the collector's own log. So +the whole path is assertable with `kubectl logs` and needs no metrics backend: +service discovery finds the engine by the labels `compose-model-replica` stamps +on serving pods, the built-in `MetricMapping`s rename its series, the unit +conversion runs, and the identity comes off the pod. + +The mock engine serves `/metrics` with two real names — `vllm:num_requests_waiting` +and `DCGM_FI_DEV_FB_USED` — so `--verify` asserts they arrive as +`modelplane_requests_waiting` and `modelplane_gpu_memory_used_bytes`, that 1024 +MiB became 1073741824 bytes, that each carries its deployment, engine, role and +cluster, and that the engine's own `vllm:` names did *not* leave the cluster. + ### Why cloud provisioning cannot be tested here diff --git a/e2e/manifests/40-model-deployment.yaml b/e2e/manifests/40-model-deployment.yaml index 2b99dc72e..e32a3508c 100644 --- a/e2e/manifests/40-model-deployment.yaml +++ b/e2e/manifests/40-model-deployment.yaml @@ -53,6 +53,19 @@ spec: def do_GET(self): if self.path == "/health": self._s({"status": "ok"}) + elif self.path == "/metrics": + # Two real vLLM names, so the collector's built-in + # renames have something to act on: one gauge and + # one the fleet converts the unit of. + b = (b'# TYPE vllm:num_requests_waiting gauge\n' + b'vllm:num_requests_waiting 3\n' + b'# TYPE DCGM_FI_DEV_FB_USED gauge\n' + b'DCGM_FI_DEV_FB_USED{gpu="0"} 1024\n') + self.send_response(200) + self.send_header("content-type", "text/plain") + self.send_header("content-length", str(len(b))) + self.end_headers() + self.wfile.write(b) elif self.path.startswith("/v1/models"): self._s({"object": "list", "data": [{"id": served, "object": "model"}]}) else: diff --git a/e2e/manifests/60-telemetry.yaml b/e2e/manifests/60-telemetry.yaml new file mode 100644 index 000000000..97c076afd --- /dev/null +++ b/e2e/manifests/60-telemetry.yaml @@ -0,0 +1,16 @@ +# Telemetry for the fleet. Applied with everything else so the collector +# composes while the model rolls out, which costs --verify no extra wait. +# +# The debug exporter prints what reached it to the collector's own log, so the +# whole path - discovery, scrape, rename, unit conversion, identity - is +# assertable with kubectl logs and needs no metrics backend to run. +apiVersion: modelplane.ai/v1alpha1 +kind: TelemetryDestination +metadata: + name: default +spec: + sinks: + - name: debug + type: debug + config: + verbosity: detailed diff --git a/e2e/run.sh b/e2e/run.sh index 84662ebbb..04a050f65 100644 --- a/e2e/run.sh +++ b/e2e/run.sh @@ -494,3 +494,62 @@ log "verify (usage record): ${usage}" cleanup_verify_pods log "End to end OK: ${base} authenticates callers, serves ${model} over OpenAI and Anthropic, rewrites the model, and meters it" + +# Telemetry. The TelemetryDestination went in with the rest of the manifests, so +# the collector composed while the model rolled out and there's nothing to wait +# for beyond the first scrape. Its debug sink prints what reached it to its own +# log, which is the whole path in one assertion: discovery found the engine by +# the labels compose-model-replica stamps, the built-in mappings renamed its +# series, the unit conversion ran, and the identity came off the pod. +log "Verifying the fleet's telemetry" +kubectl --context "$WLCTX" -n modelplane-system rollout status deploy/modelplane-collector --timeout=180s || { + echo "verify: the collector never rolled out on the workload cluster" >&2 + kubectl --context "$WLCTX" -n modelplane-system describe deploy/modelplane-collector >&2 || true + exit 1 +} + +# One log read per attempt, not one per assertion. The window is generous +# because the collector's config arrives by reconcile: on a fresh install it can +# roll out once against the destination and again once the MetricMappings land, +# and the restart the config change triggers starts its log over. In steady +# state the first read already has everything. +telemetry="" +for _ in $(seq 1 30); do + telemetry="$(kubectl --context "$WLCTX" -n modelplane-system logs deploy/modelplane-collector --tail=4000 2>/dev/null || true)" + case "$telemetry" in *modelplane_gpu_memory_used_bytes*) break ;; esac + sleep 10 +done + +missing="" +for want in \ + 'Name: modelplane_requests_waiting' \ + 'Name: modelplane_gpu_memory_used_bytes' \ + 'deployment: Str(mock-demo)' \ + 'engine: Str(mock)' \ + 'role: Str(Standalone)' \ + 'cluster: Str(local)'; do + case "$telemetry" in + *"$want"*) ;; + *) missing="$missing [$want]" ;; + esac +done + +# 1024 MiB as bytes. DCGM reports the framebuffer in MiB and the name says +# bytes, so a mapping that forgot the unit reads 1024 here instead. +case "$telemetry" in +*"Value: 1073741824"*) ;; +*) missing="$missing [DCGM_FI_DEV_FB_USED converted from MiB to bytes]" ;; +esac + +# Nothing the mappings didn't rename leaves a cluster, so the engine's own +# names must not appear downstream. +case "$telemetry" in +*"Name: vllm:num_requests_waiting"*) missing="$missing [vllm: names should not leave the cluster]" ;; +esac + +[ -z "$missing" ] || { + echo "verify: telemetry missing:$missing" >&2 + echo "$telemetry" | grep -E "Name: |-> (cluster|deployment|engine|role): |Value: " | tail -40 >&2 || true + exit 1 +} +log "Telemetry OK: the engine's series arrive renamed, converted, and attributed to its deployment" diff --git a/flake.nix b/flake.nix index 56c8b85ab..02704e802 100644 --- a/flake.nix +++ b/flake.nix @@ -57,6 +57,8 @@ "compose-inference-class" "compose-inference-cluster" "compose-inference-gateway" + "compose-metric-mapping" + "compose-telemetry-destination" "compose-nebius-cluster" "compose-serving-stack" "compose-vultr-cluster" diff --git a/functions/compose-inference-gateway/function/fn.py b/functions/compose-inference-gateway/function/fn.py index dbf36caae..135d872db 100644 --- a/functions/compose-inference-gateway/function/fn.py +++ b/functions/compose-inference-gateway/function/fn.py @@ -836,7 +836,7 @@ def compose_cluster_name(self, hostname: str, address: str) -> None: ), ) return - # Selectorless and headless: cluster DNS answers with the EndpointSlice's + # Selectorless and headless: cluster DNS answers with the endpoint's # address directly, so Envoy connects to the load balancer rather than # hairpinning through a ClusterIP. resource.update( @@ -869,6 +869,42 @@ def compose_cluster_name(self, hostname: str, address: str) -> None: }, ), ) + # The same address again, as the Endpoints this supersedes. kube-dns + # reads only that API and was never taught the EndpointSlice one, so on + # a cluster running it - which is every GKE cluster, where it is still + # the default - the name answers NXDOMAIN with the slice alone. Envoy + # then resolves no address for the backend and every cross-cluster + # request is refused with no healthy upstream. + # + # Deprecated since 1.33 and still the only thing kube-dns reads. It + # costs one object; getting it wrong costs every request to the cluster. + # + # Mirroring is turned off on it. The endpointslice mirroring controller + # copies a hand-written Endpoints into an EndpointSlice of its own, and + # the slice above already is that slice: left on, two controllers write + # the same address for the same Service and each keeps correcting the + # other's object. + resource.update( + self.rsp.desired.resources[f"cluster-name-endpoints-{label}"], + _k8s_object( + self.pc, + { + "apiVersion": "v1", + "kind": "Endpoints", + "metadata": { + "name": label, + "namespace": REMOTE_NAMESPACE, + "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, + }, + "subsets": [ + { + "addresses": [{"ip": address}], + "ports": [{"name": "https", "port": _CLUSTER_GATEWAY_PORT}], + } + ], + }, + ), + ) def compose_caller_auth(self) -> None: """A SecurityPolicy authenticating callers against the selected Secrets. diff --git a/functions/compose-inference-gateway/tests/test_fn.py b/functions/compose-inference-gateway/tests/test_fn.py index c19ce6379..74703a0a9 100644 --- a/functions/compose-inference-gateway/tests/test_fn.py +++ b/functions/compose-inference-gateway/tests/test_fn.py @@ -773,8 +773,10 @@ async def test_resolves_each_cluster_gateway_name(self) -> None: compose-inference-cluster derived, and this gateway's Envoy resolves it, so its cluster needs a Service of that name. An IP is served by a headless Service and an EndpointSlice; a hostname, which is how a cloud - load balancer names itself, by an ExternalName Service. A cluster that - hasn't published both an address and a name gets neither. + load balancer names itself, by an ExternalName Service. An IP also gets + the Endpoints the slice supersedes, because kube-dns reads only that and + is what GKE runs. A cluster that hasn't published both an address and a + name gets neither. """ ipv4 = "prod-ipv4-gateway-aaaaa.modelplane-system.svc.cluster.local" ipv6 = "prod-ipv6-gateway-bbbbb.modelplane-system.svc.cluster.local" @@ -814,6 +816,18 @@ async def test_resolves_each_cluster_gateway_name(self) -> None: "metadata": {"name": "prod-ipv4-gateway-aaaaa", "namespace": fn.REMOTE_NAMESPACE}, "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, }, + "cluster-name-endpoints-prod-ipv4-gateway-aaaaa": { + "apiVersion": "v1", + "kind": "Endpoints", + "metadata": { + "name": "prod-ipv4-gateway-aaaaa", + "namespace": fn.REMOTE_NAMESPACE, + # Off, or the mirroring controller writes a second + # EndpointSlice over the one composed beside this. + "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, + }, + "subsets": [{"addresses": [{"ip": "203.0.113.7"}], "ports": [{"name": "https", "port": 443}]}], + }, "cluster-name-slice-prod-ipv4-gateway-aaaaa": { "apiVersion": "discovery.k8s.io/v1", "kind": "EndpointSlice", @@ -832,6 +846,18 @@ async def test_resolves_each_cluster_gateway_name(self) -> None: "metadata": {"name": "prod-ipv6-gateway-bbbbb", "namespace": fn.REMOTE_NAMESPACE}, "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, }, + "cluster-name-endpoints-prod-ipv6-gateway-bbbbb": { + "apiVersion": "v1", + "kind": "Endpoints", + "metadata": { + "name": "prod-ipv6-gateway-bbbbb", + "namespace": fn.REMOTE_NAMESPACE, + # Off, or the mirroring controller writes a second + # EndpointSlice over the one composed beside this. + "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, + }, + "subsets": [{"addresses": [{"ip": "2001:db8::1"}], "ports": [{"name": "https", "port": 443}]}], + }, "cluster-name-slice-prod-ipv6-gateway-bbbbb": { "apiVersion": "discovery.k8s.io/v1", "kind": "EndpointSlice", diff --git a/functions/compose-metric-mapping/function/__init__.py b/functions/compose-metric-mapping/function/__init__.py new file mode 100644 index 000000000..ebf4b2ad4 --- /dev/null +++ b/functions/compose-metric-mapping/function/__init__.py @@ -0,0 +1,13 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/functions/compose-metric-mapping/function/fn.py b/functions/compose-metric-mapping/function/fn.py new file mode 100644 index 000000000..66940dac3 --- /dev/null +++ b/functions/compose-metric-mapping/function/fn.py @@ -0,0 +1,94 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Compose a MetricMapping. + +A MetricMapping names the metrics one component emits and what Modelplane +calls them. compose-serving-stack compiles every mapping into the transform +processor of the collector on each inference cluster. + +This function composes nothing. The collector is composed by +compose-serving-stack, which reads every MetricMapping. What this function +does is tell an operator whether the mapping reaches anything, because a +mapping that reaches no cluster looks identical to one that works. +""" + +import grpc +from crossplane.function import logging, request, resource, response +from crossplane.function.proto.v1 import run_function_pb2 as fnv1 +from crossplane.function.proto.v1 import run_function_pb2_grpc as grpcv1 +from models.ai.modelplane.metricmapping import v1alpha1 + +CONDITION_TYPE_ACCEPTED = "Accepted" +CONDITION_REASON_AVAILABLE = "Available" +CONDITION_REASON_WAITING_FOR_CLUSTERS = "WaitingForClusters" +CONDITION_REASON_NO_CLUSTERS = "NoClusters" + + +class FunctionRunner(grpcv1.FunctionRunnerServiceServicer): + """A FunctionRunner handles gRPC RunFunctionRequests.""" + + def __init__(self) -> None: + """Create a new FunctionRunner.""" + self.log = logging.get_logger() + + async def RunFunction( + self, req: fnv1.RunFunctionRequest, _: grpc.aio.ServicerContext | None + ) -> fnv1.RunFunctionResponse: # ty: ignore[invalid-method-override] # the generated grpc servicer base is untyped + """Run the function.""" + log = self.log.bind(tag=req.meta.tag) + log.info("Running function") + + rsp = response.to(req) + xr = v1alpha1.MetricMapping(**resource.struct_to_dict(req.observed.composite.resource)) + + # Every inference cluster renders every mapping, so the count of + # clusters is the count that took these renames. + response.require_resources( + rsp, + name="clusters", + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + ) + if "clusters" not in req.required_resources: + _not_ready(rsp, CONDITION_REASON_WAITING_FOR_CLUSTERS, "Waiting for the inference clusters to resolve") + return rsp + + clusters = len(list(request.get_required_resources(req, "clusters"))) + resource.update_status(rsp.desired.composite, v1alpha1.Status(clusters=clusters)) + + if clusters == 0: + _not_ready(rsp, CONDITION_REASON_NO_CLUSTERS, "No inference cluster to render these renames into") + return rsp + + response.set_conditions( + rsp, + resource.Condition( + typ=CONDITION_TYPE_ACCEPTED, + status="True", + reason=CONDITION_REASON_AVAILABLE, + message=f"Renaming {len(xr.spec.metrics)} metric(s) on {clusters} inference cluster(s)", + ), + ) + rsp.desired.composite.ready = fnv1.READY_TRUE + return rsp + + +def _not_ready(rsp: fnv1.RunFunctionResponse, reason: str, message: str) -> None: + """Report a mapping that isn't reaching anything, and why.""" + response.set_conditions( + rsp, + resource.Condition(typ=CONDITION_TYPE_ACCEPTED, status="False", reason=reason, message=message), + ) + rsp.desired.composite.ready = fnv1.READY_FALSE diff --git a/functions/compose-metric-mapping/function/main.py b/functions/compose-metric-mapping/function/main.py new file mode 100644 index 000000000..2e8441dac --- /dev/null +++ b/functions/compose-metric-mapping/function/main.py @@ -0,0 +1,55 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The composition function's main CLI.""" + +import click +from crossplane.function import logging, runtime + +from function import fn + + +@click.command() +@click.option("--debug", "-d", is_flag=True, help="Emit debug logs.") +@click.option( + "--address", + default="0.0.0.0:9443", + show_default=True, + help="Address at which to listen for gRPC connections", +) +@click.option("--tls-certs-dir", help="Serve using mTLS certificates.", envvar="TLS_SERVER_CERTS_DIR") +@click.option( + "--insecure", + is_flag=True, + help="Run without mTLS credentials. If you supply this flag --tls-certs-dir will be ignored.", +) +def cli(debug: bool, address: str, tls_certs_dir: str, insecure: bool) -> None: + """A Crossplane composition function.""" + try: + level = logging.Level.INFO + if debug: + level = logging.Level.DEBUG + logging.configure(level=level) + runtime.serve( + fn.FunctionRunner(), + address, + creds=runtime.load_credentials(tls_certs_dir), + insecure=insecure, + ) + except Exception as e: + click.echo(f"Cannot run function: {e}") + + +if __name__ == "__main__": + cli() diff --git a/functions/compose-metric-mapping/pyproject.toml b/functions/compose-metric-mapping/pyproject.toml new file mode 100644 index 000000000..526492339 --- /dev/null +++ b/functions/compose-metric-mapping/pyproject.toml @@ -0,0 +1,26 @@ +[build-system] +requires = ["uv_build>=0.11.0,<0.12"] +build-backend = "uv_build" + +[project] +name = "compose-metric-mapping" +version = "0.0.0" +description = "Mark a MetricMapping as ready." +requires-python = ">=3.11,<3.14" +license = "Apache-2.0" +dependencies = [ + "crossplane-function-sdk-python>=0.14.0", + "click>=8.1.0", + "grpcio>=1.73.1", + "crossplane-models", +] + +[tool.uv.sources] +crossplane-models = { workspace = true } + +[project.scripts] +function = "function.main:cli" + +[tool.uv.build-backend] +module-name = "function" +module-root = "" diff --git a/functions/compose-metric-mapping/tests/__init__.py b/functions/compose-metric-mapping/tests/__init__.py new file mode 100644 index 000000000..ebf4b2ad4 --- /dev/null +++ b/functions/compose-metric-mapping/tests/__init__.py @@ -0,0 +1,13 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/functions/compose-metric-mapping/tests/test_fn.py b/functions/compose-metric-mapping/tests/test_fn.py new file mode 100644 index 000000000..5399e5dd1 --- /dev/null +++ b/functions/compose-metric-mapping/tests/test_fn.py @@ -0,0 +1,138 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests for the compose-metric-mapping function.""" + +import dataclasses +import unittest + +from crossplane.function import logging, resource +from crossplane.function.proto.v1 import run_function_pb2 as fnv1 +from function import fn +from google.protobuf import duration_pb2 as durationpb +from google.protobuf import json_format +from google.protobuf import struct_pb2 as structpb + + +@dataclasses.dataclass +class Case: + """A test case for compose-metric-mapping.""" + + name: str + req: fnv1.RunFunctionRequest + want: fnv1.RunFunctionResponse + + +def setUpModule() -> None: + logging.configure(level=logging.Level.DISABLED) + + +class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): + """Tests for FunctionRunner.RunFunction.""" + + @classmethod + def setUpClass(cls) -> None: + cls.runner = fn.FunctionRunner() + + async def test_compose(self) -> None: + """The function reports whether a mapping reaches any cluster.""" + mapping = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "MetricMapping", + "metadata": {"name": "my-engine"}, + "spec": { + "metrics": [{"from": "my_engine_queued", "to": "modelplane_requests_waiting"}], + }, + } + cluster = resource.dict_to_struct( + {"apiVersion": "modelplane.ai/v1alpha1", "kind": "InferenceCluster", "metadata": {"name": "prod-us-east"}} + ) + + def req(xr: dict, clusters: list | None) -> fnv1.RunFunctionRequest: + r = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(xr))), + ) + if clusters is not None: + r.required_resources["clusters"].items.extend([fnv1.Resource(resource=c) for c in clusters]) + return r + + def want(ready: fnv1.Ready, status: dict | None, cond: fnv1.Condition) -> fnv1.RunFunctionResponse: + composite = fnv1.Resource(ready=ready) + if status is not None: + composite.resource.CopyFrom(resource.dict_to_struct(status)) + return fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=composite), + conditions=[cond], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster") + } + ), + ) + + cases = [ + Case( + name="ready, and says how many clusters took the renames", + req=req(mapping, [cluster, cluster]), + want=want( + fnv1.READY_TRUE, + {"status": {"clusters": 2}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Renaming 1 metric(s) on 2 inference cluster(s)", + ), + ), + ), + Case( + name="not ready when no cluster exists to render into", + req=req(mapping, []), + want=want( + fnv1.READY_FALSE, + {"status": {"clusters": 0}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoClusters", + message="No inference cluster to render these renames into", + ), + ), + ), + Case( + name="waits for the clusters to resolve", + req=req(mapping, None), + want=want( + fnv1.READY_FALSE, + None, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForClusters", + message="Waiting for the inference clusters to resolve", + ), + ), + ), + ] + + for case in cases: + with self.subTest(case.name): + got = await self.runner.RunFunction(case.req, None) + self.assertEqual( + json_format.MessageToDict(case.want), + json_format.MessageToDict(got), + "-want, +got", + ) diff --git a/functions/compose-model-replica/function/backends/base.py b/functions/compose-model-replica/function/backends/base.py index 6129039f7..ebe9cd1cb 100644 --- a/functions/compose-model-replica/function/backends/base.py +++ b/functions/compose-model-replica/function/backends/base.py @@ -228,6 +228,11 @@ def remote_namespace(replica: v1alpha1.ModelReplica) -> str: # the ModelEndpoint URLs, so it must not diverge between backends. ENGINE_PORT = 8000 +# The name given to that port. Named because the collector's engine scrape job +# selects on it: matching by number instead would find the pd-sidecar's port on +# a disaggregated pod rather than the engine behind it. +ENGINE_PORT_NAME = "http" + # Pod label carrying the serving identity (the replica name). The replica's one # shared Service selects on it, so every engine's serving pods carry it - a # Standalone pod, or a gang's leader (a LeaderWorkerSet leader or a Grove leader @@ -241,8 +246,75 @@ def remote_namespace(replica: v1alpha1.ModelReplica) -> str: # Deployment selectors fighting over each other's pods. LABEL_WORKLOAD = "modelplane.ai/workload" +# Pod labels carrying the identity every metric off this pod is attributed to. +# The collector reads them off the pod during service discovery and stamps them +# on each series, so a MetricMapping needs no labels of its own and a series +# merged across replicas keeps the identity all of them share. Metrics are the +# only reason these exist; nothing selects on them. +LABEL_DEPLOYMENT = "modelplane.ai/deployment" +LABEL_REPLICA = "modelplane.ai/replica" +LABEL_ENGINE = "modelplane.ai/engine" +LABEL_ROLE = "modelplane.ai/role" + + +# Set on the ModelReplica by the composite that scheduled it. +_LABEL_REPLICA_INDEX = "modelplane.ai/replica-index" + -def pod_metadata(member: v1alpha1.Member, labels: dict[str, str] | None = None) -> dict: +def telemetry_labels( + replica: v1alpha1.ModelReplica, + engine: v1alpha1.Engine, + member: v1alpha1.Member, +) -> dict[str, str]: + """What a metric off this pod is attributed to. + + The deployment comes off the replica, which the composite already labels + with it. There is no model here on purpose: a ModelReplica doesn't know + which ModelService fronts it, and a model name carries a slash, which a + label value can't. + """ + labels: dict[str, str] = {} + own = (replica.metadata.labels if replica.metadata else None) or {} + if name := own.get(LABEL_DEPLOYMENT): + labels[LABEL_DEPLOYMENT] = name + # Which replica of that deployment. Two replicas reporting the same metric + # need something to tell them apart or they are one series downstream and + # one of them is simply lost. The index rather than the pod: it is bounded + # by the replica count, and it survives a restart and a rolling update, + # where a pod name is minted afresh each time. + if index := own.get(_LABEL_REPLICA_INDEX): + labels[LABEL_REPLICA] = index + if engine.name: + labels[LABEL_ENGINE] = engine.name + if member.role: + labels[LABEL_ROLE] = member.role + return labels + + +def fleet_labels(replica: v1alpha1.ModelReplica, role: str) -> dict[str, str]: + """What a metric off a non-serving pod of this replica is attributed to. + + The picker is one of these: it belongs to a replica of a deployment and + carries no engine, because it serves no model. Same labels as a serving + pod otherwise, so the collector reads it off the pod with the rules it + already has and a series off the picker joins the deployment's. + """ + labels: dict[str, str] = {LABEL_ROLE: role} + own = (replica.metadata.labels if replica.metadata else None) or {} + if name := own.get(LABEL_DEPLOYMENT): + labels[LABEL_DEPLOYMENT] = name + if index := own.get(_LABEL_REPLICA_INDEX): + labels[LABEL_REPLICA] = index + return labels + + +def pod_metadata( + member: v1alpha1.Member, + labels: dict[str, str] | None = None, + *, + replica: v1alpha1.ModelReplica, + engine: v1alpha1.Engine, +) -> dict: """Pod template metadata for a member: its template.metadata plus managed labels. The member's template.metadata.labels and .annotations propagate to the pod @@ -255,6 +327,7 @@ def pod_metadata(member: v1alpha1.Member, labels: dict[str, str] | None = None) """ user = member.template.metadata merged = dict((user.labels if user else None) or {}) + merged.update(telemetry_labels(replica, engine, member)) merged.update(labels or {}) meta: dict = {} if merged: diff --git a/functions/compose-model-replica/function/backends/grove.py b/functions/compose-model-replica/function/backends/grove.py index 70a4cfad8..500e2ecbe 100644 --- a/functions/compose-model-replica/function/backends/grove.py +++ b/functions/compose-model-replica/function/backends/grove.py @@ -106,7 +106,7 @@ def container(member: v1alpha1.Member, *, serving: bool) -> dict: if security_context: c["securityContext"] = security_context if serving: - c["ports"] = [{"containerPort": base.ENGINE_PORT}] + c["ports"] = [{"name": base.ENGINE_PORT_NAME, "containerPort": base.ENGINE_PORT}] c["readinessProbe"] = { "httpGet": {"path": "/health", "port": base.ENGINE_PORT}, "initialDelaySeconds": 30, @@ -151,6 +151,8 @@ def pod_spec(member: v1alpha1.Member, c: dict) -> dict: base.GROVE_QUEUE_LABEL: base.GROVE_QUEUE, _LABEL_CLIQUE_ROLE: "leader", }, + replica=replica, + engine=engine, ), "spec": { "roleName": base.GROVE_LEADER_CLIQUE, @@ -170,7 +172,7 @@ def pod_spec(member: v1alpha1.Member, c: dict) -> dict: # stable DNS name until it's listening. worker_clique = { "name": base.GROVE_WORKER_CLIQUE, - **base.pod_metadata(worker, {base.GROVE_QUEUE_LABEL: base.GROVE_QUEUE}), + **base.pod_metadata(worker, {base.GROVE_QUEUE_LABEL: base.GROVE_QUEUE}, replica=replica, engine=engine), "spec": { "roleName": base.GROVE_WORKER_CLIQUE, "replicas": worker_replicas, diff --git a/functions/compose-model-replica/function/backends/llmd.py b/functions/compose-model-replica/function/backends/llmd.py index d5ce24fb0..b1ad11348 100644 --- a/functions/compose-model-replica/function/backends/llmd.py +++ b/functions/compose-model-replica/function/backends/llmd.py @@ -100,7 +100,7 @@ def container(member: v1alpha1.Member, *, serving: bool) -> dict: env.extend(e.model_dump(exclude_none=True) for e in engine_container.env) c["env"] = env if serving: - c["ports"] = [{"containerPort": base.ENGINE_PORT}] + c["ports"] = [{"name": base.ENGINE_PORT_NAME, "containerPort": base.ENGINE_PORT}] c["readinessProbe"] = { "httpGet": {"path": "/health", "port": base.ENGINE_PORT}, "initialDelaySeconds": 30, @@ -133,7 +133,9 @@ def pod_spec(member: v1alpha1.Member, c: dict) -> dict: # serving port, and the readiness probe. The leader member's own # template.metadata merges in underneath them. leader_pod = { - "metadata": base.pod_metadata(leader, {base.LABEL_SERVING: serving_label, _LABEL_ROLE: "leader"}), + "metadata": base.pod_metadata( + leader, {base.LABEL_SERVING: serving_label, _LABEL_ROLE: "leader"}, replica=replica, engine=engine + ), "spec": pod_spec(leader, container(leader, serving=True)), } # The worker followers don't serve the OpenAI API, so they carry no @@ -145,7 +147,7 @@ def pod_spec(member: v1alpha1.Member, c: dict) -> dict: worker_pod = { "spec": pod_spec(worker, container(worker, serving=False)), } - worker_metadata = base.pod_metadata(worker) + worker_metadata = base.pod_metadata(worker, replica=replica, engine=engine) if worker_metadata: worker_pod["metadata"] = worker_metadata diff --git a/functions/compose-model-replica/function/backends/native.py b/functions/compose-model-replica/function/backends/native.py index 14ffbcc52..9568d1e3e 100644 --- a/functions/compose-model-replica/function/backends/native.py +++ b/functions/compose-model-replica/function/backends/native.py @@ -61,7 +61,7 @@ def build( "name": "engine", "image": engine_container.image, "args": list(engine_container.args or []), - "ports": [{"containerPort": base.ENGINE_PORT}], + "ports": [{"name": base.ENGINE_PORT_NAME, "containerPort": base.ENGINE_PORT}], # vLLM tensor parallelism needs a large /dev/shm. "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}, *cache_volume_mounts], "readinessProbe": { @@ -115,7 +115,10 @@ def build( "spec": { "replicas": int(engine.copies or 1), "selector": {"matchLabels": selector}, - "template": {"metadata": base.pod_metadata(member, pod_labels), "spec": pod_spec}, + "template": { + "metadata": base.pod_metadata(member, pod_labels, replica=replica, engine=engine), + "spec": pod_spec, + }, }, } diff --git a/functions/compose-model-replica/function/routing.py b/functions/compose-model-replica/function/routing.py index 68bdd05a5..2be71ad6b 100644 --- a/functions/compose-model-replica/function/routing.py +++ b/functions/compose-model-replica/function/routing.py @@ -66,6 +66,12 @@ def _namespace(meta: metav1.ObjectMeta | None) -> str: # config, so the mount path never needs to encode which one. _EPP_CONFIG_FILE = "epp-config.yaml" +# The picker's Prometheus endpoint, and what a series off it is attributed to. +# 9090 is the picker's own default; it is named here because the port has to be +# declared on the pod for the collector to discover it either way. +_EPP_METRICS_PORT = 9090 +_EPP_ROLE = "picker" + # The pd-sidecar takes ENGINE_PORT (8000), so the decode engine listens here. _DECODE_ENGINE_PORT = 8001 @@ -279,7 +285,7 @@ def _unified( ns = base.remote_namespace(replica) out["inference-pool"] = base.wrap_object(provider_config, _inference_pool(name, ns, selector)) out[base.ROUTE_KEY] = base.wrap_object(provider_config, _http_route(replica, name)) - out.update(_epp_objects(name, ns, provider_config, _unified_epp_config_yaml(block_size))) + out.update(_epp_objects(name, ns, provider_config, _unified_epp_config_yaml(block_size), replica)) return out @@ -328,7 +334,7 @@ def _disaggregated( ns = base.remote_namespace(replica) out["inference-pool"] = base.wrap_object(provider_config, _inference_pool(name, ns, selector)) out[base.ROUTE_KEY] = base.wrap_object(provider_config, _http_route(replica, name)) - out.update(_epp_objects(name, ns, provider_config, _disaggregated_epp_config_yaml(block_size))) + out.update(_epp_objects(name, ns, provider_config, _disaggregated_epp_config_yaml(block_size), replica)) return out @@ -392,7 +398,12 @@ def _add_sidecar_to_decode(obj: k8sobjv1alpha1.Object) -> None: containers = tmpl["spec"]["containers"] engine = next(c for c in containers if c["name"] == "engine") port = _decode_port(engine) - engine["ports"] = [{"containerPort": port}] + # Named, because the collector's engine scrape job selects on the port + # name: dropping it here would leave a disaggregated decode pod with no + # named port at all, and nothing would scrape the engine behind the + # sidecar. The sidecar's own port stays unnamed for the same reason - + # it serves inference, not /metrics. + engine["ports"] = [{"name": base.ENGINE_PORT_NAME, "containerPort": port}] engine["readinessProbe"] = { "httpGet": {"path": "/health", "port": port}, "initialDelaySeconds": 30, @@ -490,7 +501,13 @@ def _http_route(replica: v1alpha1.ModelReplica, name: str) -> dict: } -def _epp_objects(name: str, ns: str, provider_config: str, config_yaml: str) -> dict[str, k8sobjv1alpha1.Object]: +def _epp_objects( + name: str, + ns: str, + provider_config: str, + config_yaml: str, + replica: v1alpha1.ModelReplica, +) -> dict[str, k8sobjv1alpha1.Object]: """The endpoint picker: ServiceAccount, RBAC, ConfigMap, Deployment, Service. config_yaml is the rendered EndpointPickerConfig the picker runs with; it @@ -545,7 +562,11 @@ def _epp_objects(name: str, ns: str, provider_config: str, config_yaml: str) -> "selector": {"matchLabels": {"app": epp}}, "template": { "metadata": { - "labels": {"app": epp}, + # The identity labels beside the selector label: they are + # how the collector attributes a series off this pod to the + # deployment and replica it routes for. Nothing selects on + # them, so they can't collide with the selector above. + "labels": {"app": epp, **base.fleet_labels(replica, _EPP_ROLE)}, "annotations": {"modelplane.ai/epp-config-checksum": config_checksum}, }, "spec": { @@ -560,10 +581,27 @@ def _epp_objects(name: str, ns: str, provider_config: str, config_yaml: str) -> "--pool-group=inference.networking.k8s.io", f"--config-file=/config/{_EPP_CONFIG_FILE}", "--grpc-port=9002", + f"--metrics-port={_EPP_METRICS_PORT}", + # The picker's metrics endpoint authenticates + # its callers by TokenReview by default, which + # needs the system:auth-delegator ClusterRole + # its ServiceAccount does not hold and cannot + # be given namespace-scoped. Left on, every + # scrape is rejected. Off, the endpoint is + # plain HTTP on the pod network, which is how + # every engine already serves /metrics. + # + # --secure-serving is a different thing and + # stays on: it is the TLS of the ext-proc gRPC + # server Envoy calls, not of this. + "--metrics-endpoint-auth=false", ], "ports": [ {"name": "grpc", "containerPort": 9002}, {"name": "grpc-health", "containerPort": 9003}, + # Named for the collector's engine scrape job, + # which keeps a pod on this name. + {"name": base.ENGINE_PORT_NAME, "containerPort": _EPP_METRICS_PORT}, ], "volumeMounts": [{"name": "config", "mountPath": "/config"}], } diff --git a/functions/compose-model-replica/tests/test_backends.py b/functions/compose-model-replica/tests/test_backends.py index e00f7cef6..79728fa0b 100644 --- a/functions/compose-model-replica/tests/test_backends.py +++ b/functions/compose-model-replica/tests/test_backends.py @@ -35,6 +35,8 @@ from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 _SERVING = "modelplane.ai/serving" +_ENGINE = "modelplane.ai/engine" +_ROLE = "modelplane.ai/role" _WORKLOAD = "modelplane.ai/workload" _CLIQUE_ROLE = "modelplane.ai/clique-role" _QUEUE_LABEL = "kai.scheduler/queue" @@ -206,14 +208,16 @@ def _claim_template(count: int, *, replica: str = "r", engine: str = "main", rol "replicas": 1, "selector": {"matchLabels": {_WORKLOAD: _WORKLOAD_NAME}}, "template": { - "metadata": {"labels": {_SERVING: "r", _WORKLOAD: _WORKLOAD_NAME}}, + "metadata": { + "labels": {_ENGINE: "main", _ROLE: "Standalone", _SERVING: "r", _WORKLOAD: _WORKLOAD_NAME} + }, "spec": { "containers": [ { "name": "engine", "image": "vllm/vllm-openai:latest", "args": ["--model=Qwen/Qwen3-0.6B"], - "ports": [{"containerPort": 8000}], + "ports": [{"name": "http", "containerPort": 8000}], "resources": {"claims": [{"name": "devices"}]}, "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], "readinessProbe": { @@ -283,7 +287,13 @@ def pod_spec(container: dict, role: str) -> dict: "cliques": [ { "name": "leader", - "labels": {_SERVING: "r", _QUEUE_LABEL: _QUEUE, _CLIQUE_ROLE: "leader"}, + "labels": { + _ENGINE: "main", + _ROLE: "Leader", + _SERVING: "r", + _QUEUE_LABEL: _QUEUE, + _CLIQUE_ROLE: "leader", + }, "spec": { "roleName": "leader", "replicas": 1, @@ -293,7 +303,7 @@ def pod_spec(container: dict, role: str) -> dict: }, { "name": "worker", - "labels": {_QUEUE_LABEL: _QUEUE}, + "labels": {_ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE}, "spec": { "roleName": "worker", "replicas": worker_replicas, @@ -336,7 +346,7 @@ def _engine( if env is not None: c["env"] = env if serving: - c["ports"] = [{"containerPort": 8000}] + c["ports"] = [{"name": "http", "containerPort": 8000}] c["readinessProbe"] = { "httpGet": {"path": "/health", "port": 8000}, "initialDelaySeconds": 30, @@ -481,7 +491,13 @@ def test_member_metadata_propagates_to_native_pod_template(self) -> None: meta = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["metadata"] self.assertEqual( meta["labels"], - {"example.com/role": "standalone", _SERVING: "r", _WORKLOAD: _WORKLOAD_NAME}, + { + "example.com/role": "standalone", + _ENGINE: "main", + _ROLE: "Standalone", + _SERVING: "r", + _WORKLOAD: _WORKLOAD_NAME, + }, ) self.assertEqual(meta["annotations"], {"example.com/config": "standalone"}) @@ -502,11 +518,20 @@ def test_member_metadata_propagates_to_cliques_independently(self) -> None: leader = _clique(manifest, "leader") self.assertEqual( leader["labels"], - {"example.com/role": "leader", _SERVING: "r", _QUEUE_LABEL: _QUEUE, _CLIQUE_ROLE: "leader"}, + { + "example.com/role": "leader", + _ENGINE: "main", + _ROLE: "Leader", + _SERVING: "r", + _QUEUE_LABEL: _QUEUE, + _CLIQUE_ROLE: "leader", + }, ) self.assertEqual(leader["annotations"], {"example.com/config": "leader"}) worker = _clique(manifest, "worker") - self.assertEqual(worker["labels"], {"example.com/role": "worker", _QUEUE_LABEL: _QUEUE}) + self.assertEqual( + worker["labels"], {"example.com/role": "worker", _ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE} + ) self.assertEqual(worker["annotations"], {"example.com/config": "worker"}) def test_worker_without_metadata_composes_only_managed_labels(self) -> None: @@ -517,7 +542,7 @@ def test_worker_without_metadata_composes_only_managed_labels(self) -> None: out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") manifest = out["model-serving-main"].spec.forProvider.manifest worker = _clique(manifest, "worker") - self.assertEqual(worker["labels"], {_QUEUE_LABEL: _QUEUE}) + self.assertEqual(worker["labels"], {_ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE}) self.assertNotIn("annotations", worker) @staticmethod @@ -668,8 +693,13 @@ def test_only_leader_carries_serving_label(self) -> None: leader_labels = lwt["leaderTemplate"]["metadata"]["labels"] self.assertEqual(leader_labels[_SERVING], "r") self.assertEqual(leader_labels[self._LWS_ROLE], "leader") - # The worker followers never serve, so they carry no metadata at all. - self.assertNotIn("metadata", lwt["workerTemplate"]) + # The worker followers never serve, so they carry no serving label and + # the replica's Service can't route to them. They do carry the + # telemetry identity: a worker holds GPUs, and its metrics are the + # deployment's. + worker_labels = lwt["workerTemplate"]["metadata"]["labels"] + self.assertNotIn(_SERVING, worker_labels) + self.assertEqual(worker_labels, {_ENGINE: "main", _ROLE: "Worker"}) def test_leader_address_and_rank_env_injected(self) -> None: # Every gang container leads with the backend-neutral coordination vars @@ -925,6 +955,44 @@ def test_replaces_unified_service_with_pool_and_epp(self) -> None: self.assertEqual(pool["kind"], "InferencePool") self.assertEqual(pool["spec"]["endpointPickerRef"]["name"], "r-epp") + def test_the_picker_is_scrapeable_and_attributed(self) -> None: + """A built-in MetricMapping renames the picker's scheduling latency. + + Nothing can match it unless something scrapes the picker, and the + collector's engine job keeps a pod on two things: the deployment + label, and a container port named `http`. Without both, the mapping + is config that matches nothing for the life of the fleet. + + Its metrics endpoint authenticates callers by TokenReview by default, + which needs a ClusterRole the picker's namespaced ServiceAccount + cannot hold, so every scrape would be rejected. --secure-serving is + left alone: that one is the ext-proc gRPC server Envoy calls. + """ + replica = _replica() + replica.metadata = metav1.ObjectMeta( + name="r", + namespace="ml-team", + labels={base.LABEL_DEPLOYMENT: "qwen3-8b", "modelplane.ai/replica-index": "2"}, + ) + replica.spec.serving = v1alpha1.Serving(mode="Unified") + composed = {} + for engine in replica.spec.engines: + composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) + template = routing.apply(composed, replica, _PC)["epp"].spec.forProvider.manifest["spec"]["template"] + + labels = template["metadata"]["labels"] + self.assertEqual(labels[base.LABEL_DEPLOYMENT], "qwen3-8b") + self.assertEqual(labels[base.LABEL_REPLICA], "2") + self.assertEqual(labels[base.LABEL_ROLE], "picker") + # Still the Deployment's own selector label, which must not move. + self.assertEqual(labels["app"], "r-epp") + + container = next(c for c in template["spec"]["containers"] if c["name"] == "epp") + self.assertIn({"name": base.ENGINE_PORT_NAME, "containerPort": 9090}, container["ports"]) + self.assertIn("--metrics-port=9090", container["args"]) + self.assertIn("--metrics-endpoint-auth=false", container["args"]) + self.assertNotIn("--secure-serving=false", container["args"]) + def test_injects_nixl_plumbing(self) -> None: """Both disagg engines get the NIXL plumbing the schema can't express: a Memory /dev/shm and VLLM_NIXL_SIDE_CHANNEL_HOST = pod IP.""" @@ -1029,6 +1097,19 @@ def test_decode_gets_sidecar_and_moves_engine_port(self) -> None: self.assertEqual(sidecar["readinessProbe"]["timeoutSeconds"], 5) self.assertIn("--secure-proxy=false", sidecar["args"]) + def test_a_decode_engine_keeps_a_scrapeable_port(self) -> None: + """The collector's engine job keeps a pod on the port named `http`. + + Moving the decode engine off 8000 for the sidecar drops the name with + it, and the sidecar takes the port unnamed because it serves inference + rather than /metrics. A decode pod with no named port anywhere is one + nothing scrapes, so a disaggregated deployment reports half its + engines and the shortfall looks like idle capacity. + """ + containers = self._serving_pod(self._apply(), "decode")["spec"]["containers"] + engine = next(c for c in containers if c["name"] == "engine") + self.assertEqual(engine["ports"], [{"name": base.ENGINE_PORT_NAME, "containerPort": 8001}]) + def test_prefill_has_no_sidecar(self) -> None: containers = self._serving_pod(self._apply(), "prefill")["spec"]["containers"] self.assertEqual([c["name"] for c in containers], ["engine"]) @@ -1099,7 +1180,7 @@ def test_decode_can_be_a_grove_gang(self) -> None: worker_clique = _clique(manifest, "worker") worker = worker_clique["spec"]["podSpec"] self.assertEqual([c["name"] for c in worker["containers"]], ["engine"]) - self.assertEqual(worker_clique["labels"], {_QUEUE_LABEL: _QUEUE}) + self.assertEqual(worker_clique["labels"], {_ENGINE: "decode", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE}) class TestUnifiedRouting(unittest.TestCase): diff --git a/functions/compose-model-replica/tests/test_fn.py b/functions/compose-model-replica/tests/test_fn.py index 74ab85151..ad6d524ec 100644 --- a/functions/compose-model-replica/tests/test_fn.py +++ b/functions/compose-model-replica/tests/test_fn.py @@ -197,6 +197,9 @@ async def test_compose(self) -> None: "template": { "metadata": { "labels": { + "modelplane.ai/deployment": "my-deployment", + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", "modelplane.ai/serving": "test-replica", "modelplane.ai/workload": resource.child_name( "test-replica", "main" @@ -209,7 +212,7 @@ async def test_compose(self) -> None: "name": "engine", "image": "vllm/vllm-openai:latest", "args": ["--model=Qwen/Qwen3-0.6B"], - "ports": [{"containerPort": 8000}], + "ports": [{"name": "http", "containerPort": 8000}], "resources": {"claims": [{"name": "devices"}]}, "volumeMounts": [ {"name": "dshm", "mountPath": "/dev/shm"}, diff --git a/functions/compose-serving-stack/function/collector.py b/functions/compose-serving-stack/function/collector.py new file mode 100644 index 000000000..706f2fa47 --- /dev/null +++ b/functions/compose-serving-stack/function/collector.py @@ -0,0 +1,738 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The OpenTelemetry collector each inference cluster runs. + +It scrapes everything Modelplane installs, renames what it scraped onto the +modelplane_* surface, merges each deployment's replicas into one series, and +exports to the destination. + +Rendered here rather than through the OpenTelemetry Operator: its +OpenTelemetryCollector CRD would be one more operator on every GPU cluster to +install, own and upgrade, for a Deployment and a ConfigMap that this composes +directly. The receiver does its own Kubernetes service discovery, so nothing +here needs the operator's target allocator either. +""" + +import hashlib +from typing import Any + +import yaml +from models.ai.modelplane.metricmapping import v1alpha1 as mmv1alpha1 +from models.ai.modelplane.telemetrydestination import v1alpha1 as tdv1alpha1 + +NAMESPACE = "modelplane-system" +NAME = "modelplane-collector" +_CREDENTIALS_DIR = "/etc/modelplane/telemetry" + +IMAGE = "otel/opentelemetry-collector-contrib:0.161.0" + +# Pod labels as Kubernetes service discovery spells them: modelplane.ai/x +# arrives as __meta_kubernetes_pod_label_modelplane_ai_x. +# +# On every serving pod, workers included. The engines job selects on it and the +# substrate job drops on it, so the two partition the same set rather than +# leaving a worker to be collected by both. +_DEPLOYMENT_LABEL = "modelplane_ai_deployment" + +# The name compose-model-replica gives the engine's port. Selecting by name +# rather than number is what keeps this off the pd-sidecar's port on a +# disaggregated pod, which would answer and serve the wrong thing. +_METRICS_PORT = "http" + +# The port the AI gateway's ext-proc sidecar serves its own metrics on, which +# is where the GenAI semantic-convention series live. Envoy's admin port +# carries Envoy's own statistics and none of these. +_GENAI_PORT = "aigw-admin" + +# What a series is attributed to, and the only resource attributes that survive +# to the exporter. The merge across replicas groups on exactly these, so a +# series differing in nothing else is one series. +# +# node is here for the GPU job, whose series belong to hardware rather than to +# a deployment; an engine's series carry no node, which is what lets replicas on +# different nodes merge. +_IDENTITY = ( + "cluster", + "namespace", + "deployment", + "replica", + "engine", + "role", + "node", + # What the scrape came from, as the Prometheus receiver names it, which an + # exporter renders as job and instance. + # + # Kept because a series has to be unique to whatever produced it or one + # producer's numbers silently replace another's. The identity above covers + # an engine; it covers nothing else. Two gateway pods carry no identity at + # all, two replicas of a substrate controller share a namespace, and a + # ModelReplica with copies greater than one runs several pods under one + # replica index. Each of those is a collision, and a collision is a wrong + # number that looks right. + # + # It is the pod's address, so it does churn on a rolling update, which is + # the cost. Aggregate it away in the query, grouping by the identity above. + "service.name", + "service.instance.id", +) + +# OTTL quotes strings with double quotes; a Python list renders single ones and +# the collector refuses to start on it. +_IDENTITY_OTTL = ", ".join(f'"{k}"' for k in sorted(_IDENTITY)) + +_SCRAPE_INTERVAL = "15s" +_SUBSTRATE_INTERVAL = "30s" + + +def _relabel_pod_identity() -> list[dict[str, Any]]: + """Carry Modelplane's identity from the pod's labels onto every series. + + A series inherits these, so a MetricMapping needs no labels of its own. + compose-model-replica stamps them; a pod carrying none is one no deployment + owns. + + No model: a ModelReplica doesn't know which ModelService fronts it, and a + model name carries a slash, which a label value can't. Deployment is finer + grained anyway - a deployment serves one model, a model may have several. + """ + return [ + {"source_labels": [f"__meta_kubernetes_pod_label_{src}"], "target_label": dst} + for src, dst in ( + ("modelplane_ai_deployment", "deployment"), + ("modelplane_ai_replica", "replica"), + ("modelplane_ai_engine", "engine"), + ("modelplane_ai_role", "role"), + ) + ] + [{"source_labels": ["__meta_kubernetes_namespace"], "target_label": "namespace"}] + + +def _dcgm_selector(action: str) -> dict[str, Any]: + """Match a GPU exporter, however it was packaged. + + The two spell it differently - gke-managed-dcgm-exporter and dcgm-exporter - + and put the name on different labels depending on who packaged it, so both + are read and the match is on what they share. One definition, used by the + job that keeps them and the job that has to leave them alone, because the + two drifting apart is how a series gets collected twice. + """ + return { + "source_labels": [ + "__meta_kubernetes_pod_label_app_kubernetes_io_name", + "__meta_kubernetes_pod_label_app", + ], + "action": action, + "regex": ".*dcgm.*", + } + + +def _annotated_path() -> list[dict[str, Any]]: + """Take the metrics path from the pod's own annotation, where it sets one.""" + return [ + { + "source_labels": ["__meta_kubernetes_pod_annotation_prometheus_io_path"], + "action": "replace", + "target_label": "__metrics_path__", + "regex": "(.+)", + } + ] + + +def _annotated_port() -> list[dict[str, Any]]: + """Take the port from the pod's own annotation, where it sets one. + + Service discovery makes a target of every declared container port, so + without this a pod is scraped on whichever it declared first. Rewriting + them all to the annotated one leaves identical targets, which discovery + then collapses to one. + """ + return [ + { + "source_labels": ["__address__", "__meta_kubernetes_pod_annotation_prometheus_io_port"], + "action": "replace", + "target_label": "__address__", + # A bracketed IPv6 host or a bare one: [^:]+ alone never matches + # [2001:db8::1]:9090, so an IPv6 pod would keep whichever port + # discovery happened to pick. + "regex": r"(\[.+\]|[^:]+)(?::\d+)?;(\d+)", + "replacement": "$1:$2", + } + ] + + +def _scrape_configs() -> list[dict[str, Any]]: + """What to scrape on an inference cluster. + + Jobs over disjoint sets of pods, so nothing is scraped twice: the engines + Modelplane runs, the gateways in front of them, the GPU exporter the + cluster came with, and everything else the serving stack installs. + + The front door needs two of them. Envoy publishes its own statistics on its + admin port, and the GenAI metrics a caller's experience is measured by come + from the AI gateway's ext-proc on a different port, at a different path. + """ + return [ + { + "job_name": "modelplane-engines", + "scrape_interval": _SCRAPE_INTERVAL, + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + # Every serving pod carries the deployment it belongs to, + # workers included: a worker holds GPUs, and the GPU series are + # the deployment's. Matched on presence, because the value is + # the deployment's name. + { + "source_labels": [f"__meta_kubernetes_pod_label_{_DEPLOYMENT_LABEL}"], + "action": "keep", + "regex": ".+", + }, + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": _METRICS_PORT, + }, + *_relabel_pod_identity(), + ], + }, + { + "job_name": "modelplane-gateway", + "scrape_interval": _SCRAPE_INTERVAL, + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": ["__meta_kubernetes_pod_label_gateway_envoyproxy_io_owning_gateway_name"], + "action": "keep", + "regex": ".+", + }, + {"source_labels": ["__meta_kubernetes_pod_container_port_name"], "action": "keep", "regex": "metrics"}, + # Envoy publishes Prometheus on its admin port, not at + # /metrics, and says so in its own annotation. Without this the + # front door 404s every interval and the modelplane_frontend_* + # series - the ones an SLO is written against - never arrive. + *_annotated_path(), + ], + }, + { + # The GenAI metrics, which are the SLO ones: what a caller waited, + # measured the same way whatever engine served it. + # + # They come from the AI gateway's ext-proc, not from the proxy. It + # runs as a native sidecar - an initContainer with restartPolicy + # Always - so it is easy to miss when reading the pod, and it + # serves its own admin port rather than Envoy's. A job of its own + # because the two ports want different paths: the proxy publishes + # at the path its annotation names, the ext-proc at /metrics. + "job_name": "modelplane-gateway-genai", + "scrape_interval": _SCRAPE_INTERVAL, + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": ["__meta_kubernetes_pod_label_gateway_envoyproxy_io_owning_gateway_name"], + "action": "keep", + "regex": ".+", + }, + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": _GENAI_PORT, + }, + ], + }, + { + "job_name": "modelplane-gpu", + "scrape_interval": _SCRAPE_INTERVAL, + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + # A job of its own because nobody annotates DCGM for scraping + # and Modelplane doesn't install it: GKE runs a managed one in + # gke-managed-system, the NVIDIA GPU operator installs its own + # elsewhere, and the substrate job sees neither. Without this + # every modelplane_gpu_* series is empty on a cloud that + # provides its own - which is every cloud. + # + # Matched on the name rather than an exact label, because the + # two spell it differently: gke-managed-dcgm-exporter and + # dcgm-exporter. Both labels are read, since which one carries + # the name depends on who packaged it. + _dcgm_selector("keep"), + {"source_labels": ["__meta_kubernetes_pod_container_port_name"], "action": "keep", "regex": "metrics"}, + # The node, because a GPU series belongs to hardware rather + # than to a deployment. DCGM names the card itself. + {"source_labels": ["__meta_kubernetes_pod_node_name"], "target_label": "node"}, + ], + }, + { + "job_name": "modelplane-substrate", + "scrape_interval": _SUBSTRATE_INTERVAL, + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": ["__meta_kubernetes_pod_annotation_prometheus_io_scrape"], + "action": "keep", + "regex": "true", + }, + # Everything a job above already names. Each of these can + # annotate itself for scraping - the gateway does, and a + # GPU operator's DCGM usually does - and collecting one here + # as well would carry it twice under two job names. + { + "source_labels": ["__meta_kubernetes_pod_label_gateway_envoyproxy_io_owning_gateway_name"], + "action": "drop", + "regex": ".+", + }, + { + "source_labels": [f"__meta_kubernetes_pod_label_{_DEPLOYMENT_LABEL}"], + "action": "drop", + "regex": ".+", + }, + _dcgm_selector("drop"), + # The rest of the same convention, not just the first line of + # it. A pod that says scrape me generally also says where: the + # cert-manager webhook declares 10250 first and annotates 9402, + # so honouring only the keep scrapes its TLS port over plain + # HTTP and logs a 400 every interval. + *_annotated_path(), + *_annotated_port(), + {"source_labels": ["__meta_kubernetes_namespace"], "target_label": "namespace"}, + ], + }, + ] + + +# Taking a part of a histogram is a function that mints a new metric beside it, +# named for the part. The rename then applies to that. +_PART_FUNCTION = {"Count": "extract_count_metric(true)", "Sum": "extract_sum_metric(true)"} +_PART_SUFFIX = {"Count": "_count", "Sum": "_sum"} + +# What one of the source unit is worth in the base unit the target name claims. +# +# Applied with scale_metric in the metric context rather than by setting a +# datapoint's value, because a histogram has no single value to set: its sum, +# its minimum and maximum, and every one of its bucket boundaries are all in +# the source unit, and a conversion that reached only the value would rename a +# histogram to seconds with its buckets still in milliseconds. scale_metric +# carries all of them, and an integer datapoint as well, which reading +# value_double would have read as nought. +# +# Written out rather than as a Python float so nothing has to render one: +# 1e-09 is not an OTTL literal. Every one carries a decimal point, because +# scale_metric takes a float and the collector refuses to start on an integer +# literal in that position. +_UNIT_FACTOR = { + "Millijoules": "0.001", + "Milliseconds": "0.001", + "Nanoseconds": "0.000000001", + "Mebibytes": "1048576.0", + "Percent": "0.01", +} + + +def statements(mappings: list[mmv1alpha1.MetricMapping]) -> tuple[list[str], list[str], list[str], list[str]]: + """Compile the mappings to OTTL, as (extract, scale, datapoint, metric) statements. + + OTTL is rendered here rather than written in a MetricMapping so the kind + stays a description of what a component emits, and the collector's own + configuration language stays Modelplane's problem. It is also the only + place that knows a unit conversion has to run somewhere a rename cannot. + """ + extract: list[str] = [] + scale: list[str] = [] + datapoint: list[str] = [] + metric: list[str] = [] + for mapping in mappings: + for m in mapping.spec.metrics: + source = m.from_ + if m.part: + # In a block of its own, ahead of the datapoint statements. The + # extraction mints a new metric, and a label or a unit + # conversion for it selects on the name that extraction gives + # it - a name that does not exist until the extraction has run. + # The extraction leaves the histogram itself alone, though the + # filter downstream drops it unless a mapping renames it too. + extract.append(f'{_PART_FUNCTION[m.part]} where metric.name == "{source}"') + source = f"{source}{_PART_SUFFIX[m.part]}" + if m.fromUnit: + scale.append(f'scale_metric({_UNIT_FACTOR[m.fromUnit]}) where metric.name == "{source}"') + # Labels before the rename, while the series still answers to the + # name this mapping selected on. After it, two folded mappings share + # one name and a statement could no longer tell them apart. + datapoint.extend(_label_statements(source, m.labels or [])) + metric.append(f'set(metric.name, "{m.to}") where metric.name == "{source}"') + return extract, scale, datapoint, metric + + +def _quote(value: str) -> str: + """One OTTL string literal, with anything that would end it escaped. + + A metric name and a label name are pattern-constrained by the XRD, but a + label's value is free text, and so is every key and value of a `values` + remap - they have to be, because they carry whatever vocabulary the + component already emits. An unescaped quote in one of them would close the + literal early and leave the rest of it as OTTL, which at best stops the + collector from starting and at worst runs. + """ + escaped = value.replace("\\", "\\\\").replace('"', '\\"') + return f'"{escaped}"' + + +def _label_statements(source: str, labels: list[Any]) -> list[str]: + """Set this mapping's labels on the datapoints of one source metric.""" + out: list[str] = [] + for label in labels: + if label.value is not None: + out.append( + f'set(datapoint.attributes["{label.name}"], {_quote(label.value)}) where metric.name == "{source}"' + ) + continue + carried = f'datapoint.attributes["{label.from_}"]' + out.append(f'set(datapoint.attributes["{label.name}"], {carried}) where metric.name == "{source}"') + for old, new in sorted((label.values or {}).items()): + out.append( + f'set(datapoint.attributes["{label.name}"], {_quote(new)}) ' + f'where metric.name == "{source}" and {carried} == {_quote(old)}' + ) + # The component's own name for it goes, or the series carries the same + # fact twice under two labels and costs twice the cardinality. Unless + # the mapping carried the label onto its own name, which is how a pure + # value remap is written: there the delete would take the label the + # statements above just set. + if label.from_ != label.name: + out.append(f'delete_key(datapoint.attributes, "{label.from_}") where metric.name == "{source}"') + return out + + +def _transform(mappings: list[mmv1alpha1.MetricMapping]) -> dict[str, Any]: + """Parts extracted, then units converted, then labels, then every rename. + + Four blocks rather than one list. Each block finishes over every metric + before the next one starts, and each step here depends on the one before + having finished everywhere: an extraction mints the metric a conversion + scales, a conversion has to reach a datapoint the rename would otherwise + have stranded in the source unit, and a label has to be set while the + series still answers to the name its mapping selected on. A statement + reaching a datapoint's attributes also can't run in the metric context. + + The conversions ignore their own errors. scale_metric refuses an + exponential histogram, and under the default error mode that one refusal + would drop the whole batch - every metric from every pod in the scrape, + not just the one it could not convert. + """ + extract, scale, datapoint, metric = statements(mappings) + blocks: list[dict[str, Any]] = [] + if extract: + blocks.append({"context": "metric", "statements": extract}) + if scale: + blocks.append({"context": "metric", "error_mode": "ignore", "statements": scale}) + if datapoint: + blocks.append({"context": "datapoint", "statements": datapoint}) + blocks.append({"context": "metric", "statements": metric}) + return {"metric_statements": blocks} + + +# Exporters that flatten a series into labels, losing anything held as a +# resource attribute unless told otherwise. Modelplane's identity - the +# cluster, deployment, engine and role a series belongs to - is all held there, +# because that is what the merge across replicas groups on, so without this a +# Prometheus backend receives every series stripped of everything that says +# what it measures. +# +# Applied under an operator's own config rather than over it: this is a default, +# and a sink that sets it wins. +# +# Keyed under both spellings of the remote-write exporter. The component +# registers as prometheus_remote_write and still answers to the older +# prometheusremotewrite, so a sink writing the name the collector's own +# documentation gives would otherwise match nothing here and export every +# series stripped of its identity - a silent loss, since the sink itself works. +_RESOURCE_TO_LABELS = {"resource_to_telemetry_conversion": {"enabled": True}} +_SINK_DEFAULTS = { + "prometheus_remote_write": _RESOURCE_TO_LABELS, + "prometheusremotewrite": _RESOURCE_TO_LABELS, + "prometheus": _RESOURCE_TO_LABELS, +} + + +def _sink_defaults(exporter: str) -> dict[str, Any]: + return {k: dict(v) for k, v in _SINK_DEFAULTS.get(exporter, {}).items()} + + +def _credential_path(sink: tdv1alpha1.Sink, key: str) -> str: + return f"{_CREDENTIALS_DIR}/{sink.name}/{key}" + + +def authenticators(sinks: list[tdv1alpha1.Sink]) -> dict[str, Any]: + """The extensions Modelplane composes for the sinks that asked for one. + + The collector carries no credential on an exporter: it authenticates + through an extension the exporter names. A typed auth block is therefore + an extension plus a reference, not a field, and composing both is what + keeps `auth` from being something an operator has to assemble by hand. + """ + composed: dict[str, Any] = {} + for sink in sinks: + if sink.auth and sink.auth.bearerTokenKey: + composed[f"bearertokenauth/{sink.name}"] = { + # A file rather than the environment: an environment variable + # is fixed for the life of the process, so a rotated token + # would need a restart to be read. + "filename": _credential_path(sink, sink.auth.bearerTokenKey), + } + return composed + + +def exporters(sinks: list[tdv1alpha1.Sink]) -> dict[str, Any]: + """The sinks as the collector's exporters block. + + A sink renders under `/`, which is how the collector names a + second instance of one component, so two sinks of the same type don't + collide. + + The endpoint and the authenticator reference are Modelplane's, and go on + last: an operator's own config can carry anything the exporter takes, but + not quietly redirect the sink somewhere else or unpick its credential. + """ + rendered: dict[str, Any] = {} + for sink in sinks: + cfg: dict[str, Any] = _sink_defaults(sink.type) | dict(sink.config or {}) + if sink.endpoint: + cfg["endpoint"] = sink.endpoint + if sink.auth and sink.auth.bearerTokenKey: + cfg["auth"] = {"authenticator": f"bearertokenauth/{sink.name}"} + rendered[f"{sink.type}/{sink.name}"] = cfg + return rendered + + +def config( + cluster: str, + mappings: list[mmv1alpha1.MetricMapping], + sinks: list[tdv1alpha1.Sink], + extensions: dict[str, Any], +) -> str: + """The collector's configuration, as YAML. + + Only modelplane_* leaves the cluster: a series no mapping renamed is one + whose meaning Modelplane cannot vouch for across engines, and it costs the + same to carry as one that was renamed. + """ + sink_exporters = exporters(sinks) + # Modelplane's authenticators last: an operator's extensions can define + # anything, but not replace the one composed for a sink's own auth block. + extensions = {**extensions, **authenticators(sinks)} + processors: dict[str, Any] = { + # First in the pipeline, so it refuses work before anything allocates + # for it. Nothing bounds what one interval brings off a fleet of + # engines, and the kill that follows a spike takes the whole cluster's + # telemetry with it until the pod is back. + "memory_limiter": {"check_interval": "1s", "limit_percentage": 80, "spike_limit_percentage": 25}, + # cluster is stamped here rather than downstream: one receiver on the + # control plane sees a merged stream and cannot tell senders apart. + "resource/cluster": {"attributes": [{"key": "cluster", "value": cluster, "action": "upsert"}]}, + # Everything a replica carries that is the replica's rather than the + # deployment's. Discovery attaches the pod's name, uid and replicaset, + # and the scrape its address, and all of it lands on the resource - + # where it does two kinds of damage. + # + # The merge below groups on the resource, so a pod name on it keeps + # every replica in a resource of its own and nothing merges. And an + # exporter that flattens resources into labels then publishes a pod + # label, which a rolling update mints afresh on every deploy and a + # billing backend counts as active for half an hour after it dies. + # + # An allowlist rather than a list of what to drop: what discovery + # attaches grows, and a series carrying something nobody chose is the + # failure this prevents. + "transform/identity": { + "metric_statements": [ + {"context": "resource", "statements": [f"keep_keys(resource.attributes, [{_IDENTITY_OTTL}])"]} + ] + }, + "transform/modelplane": _transform(mappings), + # Discovery writes the identity onto each datapoint; this lifts it onto + # the resource, which is where an exporter that flattens a series into + # labels looks for it. Without it a series arrives carrying only the + # cluster. + # + # It does not merge a deployment's replicas, whatever its name + # suggests. Each replica is scraped separately, so each is its own + # batch, and there is never a second replica here to merge with. They + # stay separate series, told apart by the replica label, and a query + # over the deployment combines them - which is the only place the + # arithmetic can be right, because adding two cumulative readings taken + # at different moments is not the traffic that happened. + "groupbyattrs/identity": {"keys": list(_IDENTITY)}, + "filter/modelplane": {"metrics": {"metric": ['not IsMatch(name, "^modelplane_.*")']}}, + "batch": {"timeout": "10s"}, + } + pipeline = [ + "memory_limiter", + "resource/cluster", + "transform/identity", + "transform/modelplane", + "groupbyattrs/identity", + "filter/modelplane", + "batch", + ] + + service: dict[str, Any] = { + "pipelines": { + "metrics": { + "receivers": ["prometheus"], + "processors": pipeline, + "exporters": sorted(sink_exporters), + } + }, + } + cfg: dict[str, Any] = { + "receivers": {"prometheus": {"config": {"scrape_configs": _scrape_configs()}}}, + "processors": processors, + "exporters": sink_exporters, + "service": service, + } + if extensions: + cfg["extensions"] = extensions + # An authenticator the service does not list is one the collector will + # not load, and an exporter referencing it then fails at startup. + service["extensions"] = sorted(extensions) + return yaml.safe_dump(cfg, sort_keys=False) + + +def _digest(rendered: str) -> str: + """A stable hash of the rendered config, so the pod restarts when it changes.""" + return hashlib.sha256(rendered.encode()).hexdigest()[:16] + + +def objects( + cluster: str, + mappings: list[mmv1alpha1.MetricMapping], + sinks: list[tdv1alpha1.Sink], + extensions: dict[str, Any], +) -> list[tuple[str, dict[str, Any], str | None]]: + """The collector as (key, manifest, readiness CEL) triples. + + Composed here rather than as a Manifests entry in the stack because the + stack is fixed at build time and every object here depends on request-time + data: the ConfigMap holds config rendered from the TelemetryDestination and + the MetricMappings, the Deployment carries that config's digest and the + destination's optional Secret mounts, and none of it exists at all until a + TelemetryDestination does. The ServiceAccount and RBAC would fit the stack, + but splitting one component across two mechanisms would put a collector's + permissions on clusters running no collector. + """ + labels = {"app.kubernetes.io/name": NAME, "app.kubernetes.io/managed-by": "modelplane"} + rendered = config(cluster, mappings, sinks, extensions) + volumes: list[dict[str, Any]] = [{"name": "config", "configMap": {"name": NAME}}] + mounts: list[dict[str, Any]] = [{"name": "config", "mountPath": "/conf"}] + env_from: list[dict[str, Any]] = [] + for sink in sinks: + if not sink.secretRef: + continue + # Mounted both ways. An environment variable is fixed for the life of a + # process, so a rotated credential would need a restart to be read; a + # mounted file is refreshed in place and an authenticator reading one + # picks the new credential up without one. The file is under the sink's + # own directory, so two sinks can both hold a key called `token`. + volume = f"credentials-{sink.name}" + volumes.append({"name": volume, "secret": {"secretName": sink.secretRef.name}}) + mounts.append({"name": volume, "mountPath": f"{_CREDENTIALS_DIR}/{sink.name}", "readOnly": True}) + env_from.append({"secretRef": {"name": sink.secretRef.name}}) + + return [ + ( + "collector-serviceaccount", + { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": {"name": NAME, "namespace": NAMESPACE, "labels": labels}, + }, + None, + ), + ( + "collector-clusterrole", + { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "ClusterRole", + "metadata": {"name": NAME, "labels": labels}, + "rules": [ + { + "apiGroups": [""], + "resources": ["pods", "services", "endpoints", "nodes", "nodes/metrics"], + "verbs": ["get", "list", "watch"], + }, + {"nonResourceURLs": ["/metrics"], "verbs": ["get"]}, + ], + }, + None, + ), + ( + "collector-clusterrolebinding", + { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "ClusterRoleBinding", + "metadata": {"name": NAME, "labels": labels}, + "roleRef": {"apiGroup": "rbac.authorization.k8s.io", "kind": "ClusterRole", "name": NAME}, + "subjects": [{"kind": "ServiceAccount", "name": NAME, "namespace": NAMESPACE}], + }, + None, + ), + ( + "collector-config", + { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": NAME, "namespace": NAMESPACE, "labels": labels}, + "data": {"collector.yaml": rendered}, + }, + None, + ), + ( + "collector", + { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": NAME, "namespace": NAMESPACE, "labels": labels}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"app.kubernetes.io/name": NAME}}, + "template": { + "metadata": { + "labels": {"app.kubernetes.io/name": NAME}, + # A changed ConfigMap doesn't restart the pod on + # its own. sha256 rather than hash(), whose string + # seed is randomised per process and would redeploy + # the collector on every reconcile. + "annotations": {"modelplane.ai/config-hash": _digest(rendered)}, + }, + "spec": { + "serviceAccountName": NAME, + "containers": [ + { + "name": "collector", + "image": IMAGE, + "args": ["--config=/conf/collector.yaml"], + "volumeMounts": mounts, + **({"envFrom": env_from} if env_from else {}), + "resources": { + "requests": {"cpu": "100m", "memory": "256Mi"}, + "limits": {"memory": "512Mi"}, + }, + } + ], + "volumes": volumes, + }, + }, + }, + }, + "object.status.readyReplicas > 0", + ), + ] diff --git a/functions/compose-serving-stack/function/fn.py b/functions/compose-serving-stack/function/fn.py index 9f77439e0..e58cf0b70 100644 --- a/functions/compose-serving-stack/function/fn.py +++ b/functions/compose-serving-stack/function/fn.py @@ -34,11 +34,16 @@ ahead of the Envoy Gateway release. """ +from typing import Any, TypeVar + import grpc -from crossplane.function import logging, resource, response +import pydantic +from crossplane.function import logging, request, resource, response from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from crossplane.function.proto.v1 import run_function_pb2_grpc as grpcv1 from models.ai.modelplane.infrastructure.servingstack import v1alpha1 +from models.ai.modelplane.metricmapping import v1alpha1 as mmv1alpha1 +from models.ai.modelplane.telemetrydestination import v1alpha1 as tdv1alpha1 from models.io.crossplane.m.helm.providerconfig import v1beta1 as helmpcv1beta1 from models.io.crossplane.m.helm.release import v1beta1 as helmv1beta1 from models.io.crossplane.m.kubernetes.object import v1alpha1 as k8sobjv1alpha1 @@ -48,7 +53,10 @@ from models.io.crossplane.protection.usage import v1beta1 as usagev1beta1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -from function import gateway, stacks +from function import collector, gateway, stacks + +# The Pydantic model one required resource is parsed into. +_T = TypeVar("_T", bound=pydantic.BaseModel) # Label key every rendered Release and Object carries, valued with its # composed-resource key, so Usage resourceSelectors can name any @@ -267,6 +275,22 @@ def _ensure_trailing_newline(cert: str) -> str: return cert if cert.endswith("\n") else cert + "\n" +# The label Crossplane stamps on a composed resource, naming the composite that +# claimed it. A ServingStack's own name is generated and carries a suffix, so +# this is what an operator calls the cluster. +_LABEL_COMPOSITE = "crossplane.io/composite" + + +def _cluster_name(xr: v1alpha1.ServingStack) -> str: + """The InferenceCluster this stack serves, as its operator named it. + + Every series the collector exports is stamped with this, and a metric + labelled with a generated name matches nothing an operator would query for. + """ + labels = (xr.metadata.labels if xr.metadata else None) or {} + return labels.get(_LABEL_COMPOSITE) or _name(xr.metadata) + + def _pc_name(xr: v1alpha1.ServingStack) -> str: """Derive the ProviderConfig name from the XR.""" return resource.child_name(_name(xr.metadata), "cluster") @@ -320,6 +344,7 @@ def compose(self) -> None: rendered = self.compose_components(components) rendered += self.compose_gateway() rendered += self.compose_gateway_pki() + self.compose_collector() self.compose_component_usages(components) self.compose_gateway_usages() self.write_status() @@ -744,6 +769,192 @@ def compose_gateway_pki(self) -> list[str]: rendered.append("gateway-client-auth") return rendered + def compose_collector(self) -> None: + """Compose the collector that gathers this cluster's telemetry. + + Nothing until a TelemetryDestination exists. Neither collector stores + anything, so collecting with nowhere to export is GPU-cluster memory and + CPU spent on samples nobody will ever read; a fleet that has not said + where its telemetry goes gets none composed. + + Ready on arrival, unlike everything else the stack composes. The + collector observes the fleet; nothing serving depends on it. Gating the + stack on it would put the fleet's ability to place a replica behind its + ability to export a metric, so one TelemetryDestination naming an + endpoint that has gone away would leave every InferenceCluster in the + fleet not Ready and stop the scheduler placing anything, anywhere. A + collector that cannot start reports it on its own objects and on the + TelemetryDestination, which is where that failure belongs. + """ + response.require_resources( + self.rsp, + name="destinations", + api_version="modelplane.ai/v1alpha1", + kind="TelemetryDestination", + ) + response.require_resources( + self.rsp, + name="mappings", + api_version="modelplane.ai/v1alpha1", + kind="MetricMapping", + ) + if "destinations" not in self.req.required_resources or "mappings" not in self.req.required_resources: + return + + # Sorted, not whichever the API server listed first: the collector + # restarts on a change to its rendered config, so an unstable order + # would redeploy it on alternate reconciles. + destinations = sorted( + self._parse("TelemetryDestination", tdv1alpha1.TelemetryDestination, "destinations"), + key=lambda d: _name(d.metadata), + ) + if not destinations: + return + sinks, extensions = self.merge_destinations(destinations) + + # The collector mounts each sink's credential as a Secret in its own + # namespace on this cluster, and the operator wrote one Secret, on the + # control plane. Resolve it here so it can be composed out there: + # without this the Deployment mounts a Secret nobody creates and the + # pod never starts, which is the one failure mode a destination with a + # credential would always hit. + secrets: dict[str, tuple[str, dict]] = {} + for sink in sinks: + if sink.secretRef is None: + continue + key = f"collector-secret-{sink.name}" + response.require_resources( + self.rsp, + name=key, + api_version="v1", + kind="Secret", + match_name=sink.secretRef.name, + namespace=_CLUSTER_RESOURCE_NAMESPACE, + ) + if key not in self.req.required_resources: + return + found = request.get_required_resource(self.req, key) + # Resolved and absent is not a reason to withhold the collector. + # The TelemetryDestination reports a Secret that isn't there, and + # the pod waits for it the way any pod waits for a Secret, which is + # recoverable the moment the operator creates it. + if found and found.get("data"): + secrets[key] = (sink.secretRef.name, found["data"]) + + # Modelplane's own mappings first, then the operator's, which add to + # them rather than replacing them. + mappings = list(stacks.BUILTIN_MAPPINGS) + mappings += self._parse("MetricMapping", mmv1alpha1.MetricMapping, "mappings") + + pc_observed = self.provider_configs_observed() + pc = _pc_name(self.xr) + for key, manifest, cel in collector.objects( + cluster=_cluster_name(self.xr), + mappings=mappings, + sinks=sinks, + extensions=extensions, + ): + if not (pc_observed or key in self.req.observed.resources): + continue + resource.update( + self.rsp.desired.resources[key], + _k8s_object( + pc, + manifest, + metadata=metav1.ObjectMeta(labels={_LABEL_RESOURCE: key}), + ready_when=cel, + ), + ) + self.rsp.desired.resources[key].ready = fnv1.READY_TRUE + + for key, (secret_name, data) in secrets.items(): + if not (pc_observed or key in self.req.observed.resources): + continue + # Same name, same namespace, other cluster, so the Deployment's + # mount needs nothing rewritten. The base64 `data` is copied + # verbatim: decoding and re-encoding would corrupt a credential + # that isn't text. + resource.update( + self.rsp.desired.resources[key], + _k8s_object( + pc, + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": secret_name, "namespace": collector.NAMESPACE}, + "data": data, + }, + metadata=metav1.ObjectMeta(labels={_LABEL_RESOURCE: key}), + ), + ) + self.rsp.desired.resources[key].ready = fnv1.READY_TRUE + + def _parse(self, kind: str, model: type[_T], key: str) -> list[_T]: + """Parse the required resources under `key`, skipping what won't. + + An object the API server stored under an older schema still comes back + on read - a CRD's validation runs on write, not on what is already + there - so one MetricMapping written before a field was required is + enough to raise here. Raising fails the whole pipeline step, which + takes down the serving stack: the fleet stops placing replicas because + a telemetry object is out of date. + + So a parse failure drops that object and says so, the same reasoning + that keeps the collector out of the stack's readiness. The rest of the + fleet's telemetry carries on without it. + """ + out: list[_T] = [] + for obj in request.get_required_resources(self.req, key): + try: + out.append(model.model_validate(obj)) + except pydantic.ValidationError as err: + name = (obj.get("metadata") or {}).get("name", "") + response.warning( + self.rsp, + f"Ignoring {kind} {name}: it does not match the current schema " + f"({err.error_count()} problems), so nothing it asks for is collected.", + ) + return out + + def merge_destinations( + self, destinations: list[tdv1alpha1.TelemetryDestination] + ) -> tuple[list[tdv1alpha1.Sink], dict[str, Any]]: + """Every destination's sinks and extensions, as one collector config. + + Concatenated rather than one of them chosen, so a second backend is a + second object rather than an edit to a singleton somebody else owns. + Every sink gets the whole stream either way, so the fleet exports to + all of them. + + A sink names the collector's exporter instance, and two destinations + naming a sink the same way would be one exporter with two meanings. + The destination that sorts first keeps the name and the other is + dropped with a warning, because taking the whole fleet's telemetry + down over a name collision is the worse failure. Extensions collide + the same way and resolve the same way. + """ + sinks: dict[str, tdv1alpha1.Sink] = {} + extensions: dict[str, Any] = {} + clashes: list[str] = [] + for dest in destinations: + name = _name(dest.metadata) + for sink in dest.spec.sinks: + if sink.name in sinks: + clashes.append(f"sink {sink.name} in TelemetryDestination {name}") + continue + sinks[sink.name] = sink + for key, value in (dest.spec.extensions or {}).items(): + if key in extensions: + clashes.append(f"extension {key} in TelemetryDestination {name}") + continue + extensions[key] = value + if clashes: + response.warning( + self.rsp, + f"Ignored {', '.join(clashes)}: already defined by a TelemetryDestination sorting earlier.", + ) + return list(sinks.values()), extensions + def compose_gateway_usages(self) -> None: """Compose Usages ordering the hand-rendered gateway teardown. diff --git a/functions/compose-serving-stack/function/stacks/__init__.py b/functions/compose-serving-stack/function/stacks/__init__.py index 6e8d240a6..5e803c978 100644 --- a/functions/compose-serving-stack/function/stacks/__init__.py +++ b/functions/compose-serving-stack/function/stacks/__init__.py @@ -21,12 +21,14 @@ stack's own file. See design/serving-stack-generation.md. """ -from function.stacks import common, components, dynamo, standard +from function.stacks import common, components, dynamo, metrics, standard from function.stacks.clouds import civo, existing, nebius, vultr from function.stacks.clouds.generated.aicr import aks, eks, gke from function.stacks.components import Chart, Cloud, Component, Manifests, Stack +from function.stacks.metrics import BUILTIN_MAPPINGS __all__ = [ + "BUILTIN_MAPPINGS", "Chart", "Cloud", "Component", @@ -35,6 +37,7 @@ "clouds", "components", "join", + "metrics", "stacks", ] diff --git a/functions/compose-serving-stack/function/stacks/metrics.py b/functions/compose-serving-stack/function/stacks/metrics.py new file mode 100644 index 000000000..66dac9f22 --- /dev/null +++ b/functions/compose-serving-stack/function/stacks/metrics.py @@ -0,0 +1,136 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""What Modelplane calls each metric the components it installs emit. + +Reviewed, pinned data, the same discipline as the component lists. A +MetricMapping is the extension point for a component Modelplane ships no +statements for; these are the ones it does. + +Renaming is only safe where the measurements agree. SGLang's +inter_token_latency is not vLLM's time per output token, so neither is +renamed onto a shared name and the gateway supplies that measurement for +both. A histogram is renamed only where its bucket boundaries match, which is +why SGLang's latency histograms are absent here: they resolve to a hundred +milliseconds where vLLM's resolve to one, and a quantile across the two is +wrong rather than approximate. +""" + +from models.ai.modelplane.metricmapping import v1alpha1 +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 + +# The front door, which measures every request it proxies under the +# OpenTelemetry GenAI conventions, for whatever engine is behind it. These are +# the SLO metrics: one component, one bucket layout, so a fleet quantile over +# them is sound. +_GATEWAY = { + "gen_ai_server_request_duration_seconds": "modelplane_frontend_request_duration_seconds", + "gen_ai_server_time_to_first_token_seconds": "modelplane_frontend_ttft_seconds", + "gen_ai_server_time_per_output_token_seconds": "modelplane_frontend_tpot_seconds", +} + +# The engines, which explain what the gateway measured. +_VLLM = { + "vllm:time_to_first_token_seconds": "modelplane_request_ttft_seconds", + "vllm:e2e_request_latency_seconds": "modelplane_request_duration_seconds", + "vllm:request_queue_time_seconds": "modelplane_request_queue_seconds", + "vllm:request_prefill_time_seconds": "modelplane_request_prefill_seconds", + "vllm:request_decode_time_seconds": "modelplane_request_decode_seconds", + "vllm:request_prompt_tokens": "modelplane_request_input_tokens", + "vllm:request_generation_tokens": "modelplane_request_output_tokens", + "vllm:num_requests_running": "modelplane_requests_running", + "vllm:num_requests_waiting": "modelplane_requests_waiting", + "vllm:kv_cache_usage_perc": "modelplane_kv_cache_utilization_ratio", + "vllm:num_preemptions_total": "modelplane_requests_preempted_total", + "vllm:prefix_cache_hits_total": "modelplane_prefix_cache_hits_total", + "vllm:prefix_cache_queries_total": "modelplane_prefix_cache_lookups_total", +} + +# SGLang publishes none of the queue-latency or preemption measurements vLLM +# does, so there is nothing here to rename onto those names. +# sglang:avg_request_queue_latency is the nearest thing to a queue time and is +# not the same measurement - a gauge holding the mean over the last batch, +# where modelplane_request_queue_seconds is a per-request histogram - and one +# name holding both would make a quantile over the fleet meaningless. +# The two token histograms need --collect-tokens-histogram as well as +# --enable-metrics; the engine publishes only the _total counters without it. +_SGLANG = { + "sglang:num_running_reqs": "modelplane_requests_running", + "sglang:num_queue_reqs": "modelplane_requests_waiting", + "sglang:token_usage": "modelplane_kv_cache_utilization_ratio", + "sglang:prompt_tokens_histogram": "modelplane_request_input_tokens", + "sglang:generation_tokens_histogram": "modelplane_request_output_tokens", +} + +# The endpoint picker. A router's queue is a different measurement from an +# engine's, so it keeps a name of its own. +_PICKER = { + "llm_d_epp_scheduler_e2e_duration_seconds": "modelplane_route_decision_seconds", +} + +# The GPUs, through whichever vendor's exporter the stack installed. DCGM +# reports framebuffer memory in MiB and energy in millijoules, and both are +# renamed onto a name that claims a different unit, so both say so. +_GPU = { + "DCGM_FI_DEV_FB_USED": "modelplane_gpu_memory_used_bytes", + "DCGM_FI_PROF_GR_ENGINE_ACTIVE": "modelplane_gpu_compute_active_ratio", + "DCGM_FI_PROF_PIPE_TENSOR_ACTIVE": "modelplane_gpu_tensor_active_ratio", + "DCGM_FI_PROF_DRAM_ACTIVE": "modelplane_gpu_memory_bandwidth_ratio", + "DCGM_FI_DEV_GPU_TEMP": "modelplane_gpu_temperature_celsius", + "DCGM_FI_DEV_POWER_USAGE": "modelplane_gpu_power_watts", + "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION": "modelplane_energy_joules_total", +} + +_GPU_UNITS = { + "DCGM_FI_DEV_FB_USED": "Mebibytes", + "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION": "Millijoules", +} + + +def _mapping( + name: str, + pairs: dict[str, str], + units: dict[str, str] | None = None, +) -> v1alpha1.MetricMapping: + units = units or {} + return v1alpha1.MetricMapping( + metadata=metav1.ObjectMeta(name=name), + spec=v1alpha1.Spec( + metrics=[ + v1alpha1.Metric.model_validate( + {"from": src, "to": dst} | ({"fromUnit": units[src]} if src in units else {}) + ) + for src, dst in pairs.items() + ] + ), + ) + + +def mappings() -> list[v1alpha1.MetricMapping]: + """The mappings Modelplane provides, as the kind an operator would write. + + Built as MetricMappings rather than as collector configuration so that + Modelplane's own renames reach the collector down the same path an + operator's do, and a built-in that breaks breaks the path everyone uses. + """ + return [ + _mapping("modelplane-gateway", _GATEWAY), + _mapping("modelplane-vllm", _VLLM), + _mapping("modelplane-sglang", _SGLANG), + _mapping("modelplane-picker", _PICKER), + _mapping("modelplane-gpu", _GPU, _GPU_UNITS), + ] + + +BUILTIN_MAPPINGS = mappings() diff --git a/functions/compose-serving-stack/tests/test_collector.py b/functions/compose-serving-stack/tests/test_collector.py new file mode 100644 index 000000000..f1b8217c6 --- /dev/null +++ b/functions/compose-serving-stack/tests/test_collector.py @@ -0,0 +1,493 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests for the collector this stack composes.""" + +import re +import typing +import unittest + +import yaml +from function import collector, stacks +from models.ai.modelplane.metricmapping import v1alpha1 as mmv1alpha1 +from models.ai.modelplane.telemetrydestination import v1alpha1 as tdv1alpha1 +from pydantic import ValidationError + +# A client authenticator: an exporter needs one of those, not the oidc +# extension, which authenticates callers of a receiver. +_EXTENSIONS = {"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}} + + +def _sink(name: str = "primary", type_: str = "otlphttp", secret: str | None = None) -> tdv1alpha1.Sink: + return tdv1alpha1.Sink.model_validate( + { + "name": name, + "type": type_, + "endpoint": "https://otel.acme.example", + **({"secretRef": {"name": secret}, "auth": {"bearerTokenKey": "token"}} if secret else {}), + } + ) + + +_SINKS = [_sink()] + + +def _metric_statements() -> list[str]: + """The rename statements, which are the last of the four blocks.""" + return collector.statements(list(stacks.BUILTIN_MAPPINGS))[-1] + + +def _config(*, extensions: dict | None = None, sinks: list | None = None) -> dict: + return yaml.safe_load( + collector.config( + "prod-us-east", + list(stacks.BUILTIN_MAPPINGS), + _SINKS if sinks is None else sinks, + _EXTENSIONS if extensions is None else extensions, + ) + ) + + +class TestConfig(unittest.TestCase): + """The collector configuration this renders.""" + + def test_pipeline_order(self) -> None: + """The rename runs before the identity is lifted onto the resource. + + Discovery writes the identity onto each datapoint and groupbyattrs + lifts it; a statement matching on a metric's name has to run while the + datapoints are still where the rename can reach them. + """ + procs = _config()["service"]["pipelines"]["metrics"]["processors"] + self.assertLess(procs.index("transform/modelplane"), procs.index("groupbyattrs/identity")) + self.assertEqual(procs[-1], "batch") + + def test_only_modelplane_leaves_the_cluster(self) -> None: + """A series the statements didn't rename is dropped.""" + self.assertIn("filter/modelplane", _config()["processors"]) + self.assertIn("filter/modelplane", _config()["service"]["pipelines"]["metrics"]["processors"]) + + def test_cluster_is_stamped_here(self) -> None: + """One receiver downstream sees a merged stream and can't tell senders apart.""" + attrs = _config()["processors"]["resource/cluster"]["attributes"] + self.assertEqual(attrs, [{"key": "cluster", "value": "prod-us-east", "action": "upsert"}]) + + def test_the_jobs_cover_disjoint_pods(self) -> None: + """A pod two jobs both collect arrives twice, under two job names.""" + jobs = { + j["job_name"]: j["relabel_configs"] + for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"] + } + substrate = jobs["modelplane-substrate"] + + def predicate(rules: list[dict], action: str) -> set[tuple]: + return {(tuple(r["source_labels"]), r["regex"]) for r in rules if r.get("action") == action} + + # Everything another job keeps, the substrate job drops on the same terms. + for job in ("modelplane-engines", "modelplane-gateway", "modelplane-gpu"): + for kept in predicate(jobs[job], "keep"): + if kept[0] == ("__meta_kubernetes_pod_container_port_name",): + continue # a port filter, not a pod filter + self.assertIn(kept, predicate(substrate, "drop"), f"{job} keeps {kept}, substrate does not drop it") + + def test_only_the_identity_survives_to_the_exporter(self) -> None: + """Discovery attaches the pod's name and uid; neither is the deployment's.""" + blocks = _config()["processors"]["transform/identity"]["metric_statements"] + statement = blocks[0]["statements"][0] + # OTTL quotes with double quotes. A Python list renders single ones and + # the collector refuses to start, which a unit test on shape won't catch. + self.assertNotIn("'", statement) + self.assertIn('keep_keys(resource.attributes, ["cluster"', statement) + pipeline = _config()["service"]["pipelines"]["metrics"]["processors"] + self.assertLess(pipeline.index("transform/identity"), pipeline.index("groupbyattrs/identity")) + + def test_the_identity_is_lifted_onto_the_resource(self) -> None: + """Without this a series arrives carrying only the cluster. + + Discovery writes the identity onto each datapoint. An exporter that + flattens a series into labels reads the resource, so something has to + move it, and this is the only processor that does. Removing it as a + no-op strips every series of what says who it belongs to - verified on + a cluster, where the resource came back carrying `cluster` alone. + """ + cfg = _config() + self.assertEqual(cfg["processors"]["groupbyattrs/identity"]["keys"], list(collector._IDENTITY)) + self.assertIn("groupbyattrs/identity", cfg["service"]["pipelines"]["metrics"]["processors"]) + + def test_a_part_is_extracted_before_anything_selects_on_it(self) -> None: + """A label or a unit for an extracted part names a metric that must exist. + + The extraction mints `_count`, and a datapoint statement for it + selects on that name. Run the datapoint block first and it matches + nothing, silently. + """ + mapping = mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_duration_ms", + "to": "modelplane_requests_total", + "part": "Count", + "fromUnit": "Milliseconds", + "labels": [{"name": "status", "value": "ok"}], + } + ] + } + } + ) + blocks = collector._transform([mapping])["metric_statements"] + contexts = [b["context"] for b in blocks] + self.assertEqual(contexts, ["metric", "metric", "datapoint", "metric"]) + self.assertIn("extract_count_metric", blocks[0]["statements"][0]) + # Everything selecting on the extracted name comes after the extraction. + for block in blocks[1:]: + for statement in block["statements"]: + self.assertIn("my_engine_duration_ms_count", statement) + + def test_every_job_carries_something_unique_to_its_target(self) -> None: + """Two producers whose series are identical are one series, and one is lost. + + The modelplane identity names an engine and nothing else: a gateway pod + carries none of it, two replicas of a substrate controller share a + namespace, and a ModelReplica with copies > 1 runs several pods under + one replica index. + """ + self.assertIn("service.instance.id", collector._IDENTITY) + statement = _config()["processors"]["transform/identity"]["metric_statements"][0]["statements"][0] + self.assertIn('"service.instance.id"', statement) + + def test_a_scrape_spike_cannot_take_the_collector_down(self) -> None: + """Nothing bounds what one interval brings off a fleet of engines.""" + cfg = _config() + self.assertIn("memory_limiter", cfg["processors"]) + self.assertEqual(cfg["service"]["pipelines"]["metrics"]["processors"][0], "memory_limiter") + + def test_the_port_rewrite_matches_an_ipv6_pod(self) -> None: + """__address__ is [2001:db8::1]:9090 there, which [^:]+ never matches.""" + rule = next( + r + for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"] + if j["job_name"] == "modelplane-substrate" + for r in j["relabel_configs"] + if r.get("target_label") == "__address__" + ) + for address in ("10.1.0.5:8000", "[2001:db8::1]:9090"): + matched = re.fullmatch(rule["regex"], f"{address};9402") + assert matched is not None, address + self.assertTrue(matched.expand(r"\1:\2").endswith(":9402")) + + def test_engine_scrape_selects_the_port_by_name(self) -> None: + """Matching by number would find the pd-sidecar on a disaggregated pod.""" + jobs = {j["job_name"]: j for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"]} + keeps = [r for r in jobs["modelplane-engines"]["relabel_configs"] if r.get("action") == "keep"] + self.assertIn("__meta_kubernetes_pod_container_port_name", [k["source_labels"][0] for k in keeps]) + + def test_gateway_has_a_target_of_its_own(self) -> None: + """Its GenAI metrics are on the ext-proc sidecar, not the proxy's port.""" + jobs = [j["job_name"] for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"]] + self.assertIn("modelplane-gateway", jobs) + + def test_both_spellings_of_remote_write_keep_their_identity(self) -> None: + """The exporter registers as prometheus_remote_write in 0.161.0. + + prometheusremotewrite is the older name it still answers to. A sink + writing the one the collector's own documentation gives would + otherwise match no default here and export every series stripped of + the cluster, deployment, engine and role it belongs to - silently, + because the sink itself works. + """ + for type_ in ("prometheus_remote_write", "prometheusremotewrite", "prometheus"): + with self.subTest(type=type_): + exporters = _config(sinks=[_sink(type_=type_)])["exporters"] + exporter = next(v for k, v in exporters.items() if k.startswith(f"{type_}/")) + self.assertTrue(exporter["resource_to_telemetry_conversion"]["enabled"]) + + def test_extensions_are_declared_to_the_service(self) -> None: + """An authenticator the service doesn't list is one the collector won't load.""" + self.assertEqual(_config()["service"]["extensions"], ["oauth2client/acme"]) + self.assertNotIn("extensions", _config(extensions={})["service"]) + + def test_energy_is_scaled_before_it_is_renamed(self) -> None: + """DCGM counts millijoules, and the name says joules. + + The scale is a block ahead of the renames, not a line ahead. The + processor finishes a block over every metric before the next one + starts, so a rename sharing the block would strand every metric + after the first at millijoules. + """ + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + self.assertEqual([b["context"] for b in blocks], ["metric", "metric"]) + self.assertTrue(any("scale_metric(0.001)" in st for st in blocks[0]["statements"])) + self.assertTrue(any("modelplane_energy_joules_total" in st for st in blocks[-1]["statements"])) + + def test_a_conversion_reaches_a_histogram_bucket(self) -> None: + """Setting value_double converts a gauge and leaves a histogram lying. + + A histogram holds its measurements in its sum, its minimum and maximum + and every bucket boundary, none of which is value_double. Renaming one + to seconds with its buckets still at milliseconds puts every quantile + a thousand times out, and nothing says so. + """ + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + for block in blocks: + for statement in block["statements"]: + self.assertNotIn("value_double", statement) + scales = next(b for b in blocks if any("scale_metric" in st for st in b["statements"])) + self.assertEqual(scales["context"], "metric") + + def test_every_conversion_factor_is_a_float_literal(self) -> None: + """scale_metric takes a float, and 1048576 is an integer to OTTL. + + The collector refuses to start on it - "must be a float" - which + takes the whole cluster's telemetry down, and nothing short of + running the collector catches it. + """ + for unit, factor in collector._UNIT_FACTOR.items(): + with self.subTest(unit=unit): + self.assertIn(".", factor, "an OTTL float literal needs a decimal point") + float(factor) + + def test_a_conversion_cannot_drop_the_batch_it_rides_in(self) -> None: + """scale_metric refuses an exponential histogram. + + Under the default error mode that one refusal fails the whole batch: + every metric from every pod in the scrape is lost, not the one it + could not convert. + """ + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + scales = next(b for b in blocks if any("scale_metric" in st for st in b["statements"])) + self.assertEqual(scales["error_mode"], "ignore") + + def test_dcgm_units_are_converted_to_the_unit_the_name_claims(self) -> None: + """DCGM reports mJ and MiB; the names say joules and bytes.""" + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + scales = " ".join(blocks[0]["statements"]) + self.assertIn("DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION", scales) + self.assertIn("DCGM_FI_DEV_FB_USED", scales) + + def test_every_unit_the_api_offers_has_a_conversion(self) -> None: + """A unit the API accepts with no conversion here is a KeyError at render time.""" + annotation = mmv1alpha1.Metric.model_fields["fromUnit"].annotation + literal = next(a for a in typing.get_args(annotation) if typing.get_origin(a) is typing.Literal) + self.assertEqual(set(typing.get_args(literal)), set(collector._UNIT_FACTOR)) + + def test_a_percentage_is_divided_into_a_ratio(self) -> None: + """A component counting 0 to 100 under a name that says a ratio is 100x out. + + vLLM and SGLang both publish a fraction, so no built-in needs this, but + vLLM's is called kv_cache_usage_perc - the name is no guide, and an + engine that means it has to be able to say so. + """ + mapping = mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_cache_percent", + "to": "modelplane_kv_cache_utilization_ratio", + "fromUnit": "Percent", + } + ] + } + } + ) + _, scale, _, _ = collector.statements([mapping]) + self.assertEqual(scale, ['scale_metric(0.01) where metric.name == "my_engine_cache_percent"']) + + def test_a_metric_name_cannot_end_the_comparison_early(self) -> None: + """A quote in `from` would rename whatever the rest of the line matched.""" + with self.assertRaises(ValidationError): + mmv1alpha1.Metric.model_validate({"from": 'x" or true or name == "y', "to": "modelplane_x"}) + for mapping in stacks.BUILTIN_MAPPINGS: + for m in mapping.spec.metrics: + round_tripped = mmv1alpha1.Metric.model_validate({"from": m.from_, "to": m.to}) + self.assertEqual(round_tripped.from_, m.from_) + + def test_a_label_value_cannot_end_the_string_it_sits_in(self) -> None: + """`from` is pattern-constrained; a label's value cannot be. + + A value and a `values` remap carry whatever vocabulary the component + already writes, so the schema has to take free text. A quote in one + would close the OTTL literal early and leave the remainder of the + value as OTTL - at best the collector refuses to start. + """ + mapping = mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_finish", + "to": "modelplane_requests_total", + "labels": [{"name": "reason", "from": "finish", "values": {'ab"c': 'x"y'}}], + } + ] + } + } + ) + _, _, datapoint, _ = collector.statements([mapping]) + joined = " ".join(datapoint) + self.assertIn(r'"ab\"c"', joined) + self.assertIn(r'"x\"y"', joined) + + def test_carrying_a_label_onto_itself_keeps_it(self) -> None: + """`from` equal to `name` is how a mapping remaps values in place. + + The delete that stops a carried label costing twice the cardinality + would otherwise take the label the statements before it just set, and + the series would lose the label entirely. + """ + mapping = mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_finish", + "to": "modelplane_requests_total", + "labels": [{"name": "reason", "from": "reason", "values": {"eos": "stop"}}], + } + ] + } + } + ) + _, _, datapoint, _ = collector.statements([mapping]) + self.assertFalse([st for st in datapoint if st.startswith("delete_key")]) + + def test_a_value_rewrite_never_lands_in_the_metric_context(self) -> None: + """value_double is a datapoint path; the collector refuses to start on it here.""" + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + metric_block = next(b for b in blocks if b["context"] == "metric") + self.assertFalse([st for st in metric_block["statements"] if "value_double" in st]) + for block in blocks: + for st in block["statements"]: + self.assertNotIn("set(name,", st) + self.assertNotIn("set(value_double,", st) + + def test_sglang_carries_no_queue_time_or_preemption(self) -> None: + """SGLang publishes neither, so there is nothing to rename onto them. + + Checked against a running SGLang v0.4.9.post2: it has no per-request + queue-time metric and no retraction counters at all. The nearest + thing, sglang:avg_request_queue_latency, is a gauge of the mean over + the last batch - a different measurement from vLLM's per-request + histogram, and one name holding both makes a fleet quantile + meaningless. + """ + sglang = { + m.from_: m.to + for mapping in stacks.BUILTIN_MAPPINGS + for m in mapping.spec.metrics + if m.from_.startswith("sglang:") + } + self.assertTrue(sglang, "the SGLang built-in went missing") + self.assertNotIn("modelplane_request_queue_seconds", sglang.values()) + self.assertNotIn("modelplane_requests_preempted_total", sglang.values()) + self.assertFalse([k for k in sglang if "retracted" in k or "queue_time" in k]) + + def test_sglang_latency_histograms_are_not_renamed(self) -> None: + """Their buckets resolve to 100ms where vLLM's resolve to 1ms.""" + joined = " ".join(_metric_statements()) + self.assertNotIn("sglang:time_to_first_token_seconds", joined) + self.assertNotIn("sglang:inter_token_latency", joined) + + +class TestObjects(unittest.TestCase): + """The manifests this composes.""" + + def _objects(self, secret: str | None = None) -> dict: + sinks = [_sink(secret=secret)] if secret else _SINKS + return { + k: m for k, m, _ in collector.objects("prod-us-east", list(stacks.BUILTIN_MAPPINGS), sinks, _EXTENSIONS) + } + + def test_config_hash_is_stable_across_processes(self) -> None: + """hash() is seeded per process, so it would redeploy on every reconcile.""" + first = self._objects()["collector"]["spec"]["template"]["metadata"]["annotations"] + second = self._objects()["collector"]["spec"]["template"]["metadata"]["annotations"] + self.assertEqual(first, second) + self.assertRegex(first["modelplane.ai/config-hash"], r"^[0-9a-f]{16}$") + + def test_credentials_mount_as_a_file_and_an_environment_variable(self) -> None: + """A rotated token in an environment variable needs a restart to be read.""" + pod = self._objects(secret="telemetry-credentials")["collector"]["spec"]["template"]["spec"] + self.assertIn("credentials-primary", [v["name"] for v in pod["volumes"]]) + self.assertEqual(pod["containers"][0]["envFrom"], [{"secretRef": {"name": "telemetry-credentials"}}]) + + def test_each_sink_gets_its_own_credential_directory(self) -> None: + """Two sinks can both hold a key called token, and neither reads the other's.""" + sinks = [ + _sink(name="vendor", secret="vendor-token"), + _sink(name="prometheus", type_="prometheusremotewrite", secret="prom-token"), + ] + pod = { + k: m for k, m, _ in collector.objects("prod-us-east", list(stacks.BUILTIN_MAPPINGS), sinks, _EXTENSIONS) + }["collector"]["spec"]["template"]["spec"] + mounts = {m["name"]: m["mountPath"] for m in pod["containers"][0]["volumeMounts"]} + self.assertEqual(mounts["credentials-vendor"], "/etc/modelplane/telemetry/vendor") + self.assertEqual(mounts["credentials-prometheus"], "/etc/modelplane/telemetry/prometheus") + + def test_two_sinks_of_one_type_do_not_collide(self) -> None: + """The collector names a second instance of a component /.""" + rendered = collector.exporters([_sink(name="a"), _sink(name="b")]) + self.assertEqual(sorted(rendered), ["otlphttp/a", "otlphttp/b"]) + + def test_a_sink_that_addresses_its_destination_another_way(self) -> None: + """Kafka takes brokers, the debug exporter nothing; neither has an endpoint.""" + sinks = [ + tdv1alpha1.Sink.model_validate( + {"name": "bus", "type": "kafka", "config": {"brokers": ["kafka.acme.example:9092"]}} + ), + tdv1alpha1.Sink.model_validate({"name": "seen", "type": "debug"}), + ] + rendered = collector.exporters(sinks) + self.assertNotIn("endpoint", rendered["kafka/bus"]) + self.assertEqual(rendered["kafka/bus"]["brokers"], ["kafka.acme.example:9092"]) + self.assertEqual(rendered["debug/seen"], {}) + + def test_auth_composes_its_own_authenticator(self) -> None: + """The collector carries no credential on an exporter, only a reference.""" + sink = _sink(secret="telemetry-credentials") + self.assertEqual( + collector.authenticators([sink]), + {"bearertokenauth/primary": {"filename": "/etc/modelplane/telemetry/primary/token"}}, + ) + self.assertEqual( + collector.exporters([sink])["otlphttp/primary"]["auth"], + {"authenticator": "bearertokenauth/primary"}, + ) + + def test_a_sinks_own_config_cannot_redirect_it(self) -> None: + """The endpoint is Modelplane's, and goes on after the operator's config.""" + sink = tdv1alpha1.Sink.model_validate( + { + "name": "primary", + "type": "otlphttp", + "endpoint": "https://otel.acme.example", + "config": {"endpoint": "https://elsewhere.example", "compression": "gzip"}, + } + ) + rendered = collector.exporters([sink])["otlphttp/primary"] + self.assertEqual(rendered["endpoint"], "https://otel.acme.example") + self.assertEqual(rendered["compression"], "gzip") + + def test_no_secret_mounts_nothing(self) -> None: + pod = self._objects()["collector"]["spec"]["template"]["spec"] + self.assertEqual([v["name"] for v in pod["volumes"]], ["config"]) + self.assertNotIn("envFrom", pod["containers"][0]) + + def test_rbac_is_read_only(self) -> None: + """Service discovery needs to list pods, and nothing needs to write.""" + rules = self._objects()["collector-clusterrole"]["rules"] + verbs = {v for r in rules for v in r["verbs"]} + self.assertEqual(verbs, {"get", "list", "watch"}) diff --git a/functions/compose-serving-stack/tests/test_fn.py b/functions/compose-serving-stack/tests/test_fn.py index f041b2edb..70fcbef17 100644 --- a/functions/compose-serving-stack/tests/test_fn.py +++ b/functions/compose-serving-stack/tests/test_fn.py @@ -93,6 +93,28 @@ def _crds(filename: str) -> list[dict]: ] +class TestClusterName(unittest.TestCase): + """The name every exported series is stamped with.""" + + def _stack(self, labels: dict[str, str] | None) -> v1alpha1.ServingStack: + return v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="local-serving-stack-d4206", labels=labels), + spec=v1alpha1.Spec( + cloud="Existing", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), + ), + ) + + def test_it_is_the_composite_an_operator_named(self) -> None: + """A ServingStack's own name is generated and carries a suffix.""" + self.assertEqual(fn._cluster_name(self._stack({"crossplane.io/composite": "local"})), "local") + + def test_it_falls_back_to_the_stack(self) -> None: + """Better a generated name on the series than none at all.""" + self.assertEqual(fn._cluster_name(self._stack(None)), "local-serving-stack-d4206") + + def _request(cloud: str, stack: str, observed: dict | None = None) -> fnv1.RunFunctionRequest: """Build a RunFunctionRequest for a test-backend ServingStack.""" return fnv1.RunFunctionRequest( @@ -829,7 +851,12 @@ def _existing_dynamo_stack() -> dict[str, fnv1.Resource]: def _response(resources: dict[str, fnv1.Resource], status: dict | None = None) -> fnv1.RunFunctionResponse: - """A whole expected response: 60s TTL, empty context, the XR status.""" + """A whole expected response: 60s TTL, empty context, the XR status. + + Every response asks for the telemetry kinds, because the collector is + composed from them and the function cannot know whether any exist until + they resolve. + """ return fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( @@ -837,6 +864,14 @@ def _response(resources: dict[str, fnv1.Resource], status: dict | None = None) - resources=resources, ), context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), ) @@ -1455,3 +1490,217 @@ async def test_composed_resource_keys(self) -> None: ) got = await self.runner.RunFunction(_request(cloud, stack, observed=observed), None) self.assertEqual(expected, set(got.desired.resources.keys())) + + +class TestCollectorReadiness(unittest.IsolatedAsyncioTestCase): + """The collector is composed, but the stack never waits on it.""" + + maxDiff = None + + @classmethod + def setUpClass(cls) -> None: + cls.runner = fn.FunctionRunner() + + @staticmethod + def _with_destination(req: fnv1.RunFunctionRequest, *, secret: str | None = None) -> fnv1.RunFunctionRequest: + sink: dict = {"name": "primary", "type": "otlphttp", "endpoint": "https://otel.acme.example"} + if secret: + sink |= {"secretRef": {"name": secret}, "auth": {"bearerTokenKey": "token"}} + req.required_resources["destinations"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "TelemetryDestination", + "metadata": {"name": "acme"}, + "spec": {"sinks": [sink]}, + } + ) + ) + ) + req.required_resources["mappings"].items.extend([]) + if secret: + req.required_resources["collector-secret-primary"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": secret, "namespace": "modelplane-system"}, + "data": {"token": "c2hoaGg="}, + } + ) + ) + ) + return req + + async def test_the_collector_does_not_gate_the_stack(self) -> None: + """A collector nothing has observed yet is still Ready. + + Everything else the stack composes is Ready only once its observed + Ready condition says so, because the fleet cannot serve without it. + The collector only watches, so gating on it would put placing a + replica behind exporting a metric: one destination pointing at an + endpoint that has gone away would take every InferenceCluster in the + fleet out of Ready and stop the scheduler. + """ + req = self._with_destination(_request("GKE", "Standard", observed=_observed_pcs())) + got = await self.runner.RunFunction(req, None) + collector_keys = [k for k in got.desired.resources if k == "collector" or k.startswith("collector-")] + # The Deployment, which is the one with a readiness CEL of its own and + # so the one that would have gated the stack. + self.assertIn("collector", collector_keys) + for key in collector_keys: + with self.subTest(key=key): + self.assertNotIn(key, req.observed.resources) + self.assertEqual(got.desired.resources[key].ready, fnv1.READY_TRUE) + + async def test_the_credential_reaches_the_cluster_that_mounts_it(self) -> None: + """The operator writes one Secret; the collector mounts it elsewhere. + + A TelemetryDestination is cluster-scoped on the control plane and the + collector runs on every workload cluster in the fleet. Resolving the + Secret and stopping there leaves the Deployment mounting a name + nothing out there creates, so the pod never starts and the fleet + exports nothing - the failure every destination with a credential + would hit, which is every destination that reaches a real backend. + """ + req = self._with_destination( + _request("GKE", "Standard", observed=_observed_pcs()), secret="telemetry-credentials" + ) + got = await self.runner.RunFunction(req, None) + self.assertIn("collector-secret-primary", got.desired.resources) + composed = resource.struct_to_dict(got.desired.resources["collector-secret-primary"].resource) + manifest = composed["spec"]["forProvider"]["manifest"] + self.assertEqual(manifest["kind"], "Secret") + self.assertEqual(manifest["metadata"]["name"], "telemetry-credentials") + self.assertEqual(manifest["metadata"]["namespace"], "modelplane-system") + # Copied verbatim: re-encoding would corrupt a credential that is not + # text, and the mount reads the same key the sink's auth names. + self.assertEqual(manifest["data"], {"token": "c2hoaGg="}) + + async def test_a_destination_asks_for_its_credential_in_one_namespace(self) -> None: + """Unqualified, the requirement matches a Secret of that name anywhere.""" + req = self._with_destination( + _request("GKE", "Standard", observed=_observed_pcs()), secret="telemetry-credentials" + ) + got = await self.runner.RunFunction(req, None) + selector = got.requirements.resources["collector-secret-primary"] + self.assertEqual(selector.match_name, "telemetry-credentials") + self.assertEqual(selector.namespace, "modelplane-system") + + async def test_every_destination_contributes_its_sinks(self) -> None: + """A second backend is a second object, not an edit to a singleton. + + Picking one destination and warning about the rest means a team + adding an export has to edit an object another team owns, and gets + silence if they create their own instead. + """ + req = _request("GKE", "Standard", observed=_observed_pcs()) + for name, sink in ( + ("acme", {"name": "vendor", "type": "otlphttp", "endpoint": "https://otel.vendor.example"}), + ("zeta", {"name": "prom", "type": "prometheus_remote_write", "endpoint": "https://p.example/w"}), + ): + req.required_resources["destinations"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "TelemetryDestination", + "metadata": {"name": name}, + "spec": {"sinks": [sink]}, + } + ) + ) + ) + req.required_resources["mappings"].items.extend([]) + got = await self.runner.RunFunction(req, None) + config = yaml.safe_load( + resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"][ + "manifest" + ]["data"]["collector.yaml"] + ) + self.assertEqual( + sorted(config["exporters"]), + ["otlphttp/vendor", "prometheus_remote_write/prom"], + ) + self.assertEqual( + sorted(config["service"]["pipelines"]["metrics"]["exporters"]), + ["otlphttp/vendor", "prometheus_remote_write/prom"], + ) + + async def test_two_destinations_cannot_name_one_exporter(self) -> None: + """A sink names the collector's exporter instance. + + Two of them under one name is one exporter with two meanings. The + destination sorting first keeps it and the other is dropped with a + warning, rather than failing the whole fleet's telemetry over a name. + """ + req = _request("GKE", "Standard", observed=_observed_pcs()) + for name, endpoint in (("acme", "https://a.example"), ("zeta", "https://z.example")): + req.required_resources["destinations"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "TelemetryDestination", + "metadata": {"name": name}, + "spec": {"sinks": [{"name": "primary", "type": "otlphttp", "endpoint": endpoint}]}, + } + ) + ) + ) + req.required_resources["mappings"].items.extend([]) + got = await self.runner.RunFunction(req, None) + config = yaml.safe_load( + resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"][ + "manifest" + ]["data"]["collector.yaml"] + ) + self.assertEqual(list(config["exporters"]), ["otlphttp/primary"]) + self.assertEqual(config["exporters"]["otlphttp/primary"]["endpoint"], "https://a.example") + self.assertTrue([r for r in got.results if "zeta" in r.message]) + + async def test_a_stale_mapping_does_not_break_the_stack(self) -> None: + """A CRD validates on write, not on what it already stored. + + A MetricMapping written against an older schema comes back on read + exactly as it was stored, so a value the enum no longer carries + reaches the parser. Parsing it raises, and raising fails the whole + pipeline step - so the serving stack composes nothing and the fleet + stops placing replicas, because one telemetry object is out of date. + Seen on a real cluster, where a mapping predating a required field + did it. + """ + req = self._with_destination(_request("GKE", "Standard", observed=_observed_pcs())) + req.required_resources["mappings"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "MetricMapping", + "metadata": {"name": "stale"}, + # A unit the enum no longer carries. + "spec": { + "metrics": [ + { + "from": "old_engine_transfer", + "to": "modelplane_request_kv_transfer_seconds", + "fromUnit": "Centiseconds", + } + ] + }, + } + ) + ) + ) + got = await self.runner.RunFunction(req, None) + # The stack still composes, and says what it dropped. + self.assertIn("collector", got.desired.resources) + self.assertTrue([r for r in got.results if "stale" in r.message]) + config = yaml.safe_load( + resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"][ + "manifest" + ]["data"]["collector.yaml"] + ) + self.assertNotIn("old_engine_waiting", yaml.safe_dump(config)) diff --git a/functions/compose-telemetry-destination/function/__init__.py b/functions/compose-telemetry-destination/function/__init__.py new file mode 100644 index 000000000..ebf4b2ad4 --- /dev/null +++ b/functions/compose-telemetry-destination/function/__init__.py @@ -0,0 +1,13 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/functions/compose-telemetry-destination/function/fn.py b/functions/compose-telemetry-destination/function/fn.py new file mode 100644 index 000000000..89d237822 --- /dev/null +++ b/functions/compose-telemetry-destination/function/fn.py @@ -0,0 +1,136 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Compose a TelemetryDestination. + +A TelemetryDestination names the sinks the fleet's metrics go to, each +carrying an exporter's own configuration verbatim, and compose-serving-stack +renders them into the collector it composes. Modelplane does not model what an +exporter is, so there is little here to validate and the little there is +matters: a sink naming an authenticator that nothing defines makes a collector +refuse to start, and that failure surfaces as telemetry silently never +arriving. +""" + +import grpc +from crossplane.function import logging, request, resource, response +from crossplane.function.proto.v1 import run_function_pb2 as fnv1 +from crossplane.function.proto.v1 import run_function_pb2_grpc as grpcv1 +from models.ai.modelplane.telemetrydestination import v1alpha1 + +CONDITION_TYPE_ACCEPTED = "Accepted" +CONDITION_REASON_AVAILABLE = "Available" +CONDITION_REASON_UNKNOWN_AUTHENTICATOR = "UnknownAuthenticator" +CONDITION_REASON_WAITING_FOR_SECRET = "WaitingForSecret" +CONDITION_REASON_SECRET_NOT_FOUND = "SecretNotFound" + +_SECRET_PREFIX = "secret-" + +# Where a sink's credential lives on the control plane. Unqualified, the +# requirement resolves a Secret of that name in any namespace, so a +# destination would accept a credential that happens to share a name with one +# in some unrelated namespace while the one it meant is absent. +_NAMESPACE = "modelplane-system" + + +class FunctionRunner(grpcv1.FunctionRunnerServiceServicer): + """A FunctionRunner handles gRPC RunFunctionRequests.""" + + def __init__(self) -> None: + """Create a new FunctionRunner.""" + self.log = logging.get_logger() + + async def RunFunction( + self, req: fnv1.RunFunctionRequest, _: grpc.aio.ServicerContext | None + ) -> fnv1.RunFunctionResponse: # ty: ignore[invalid-method-override] # the generated grpc servicer base is untyped + """Run the function.""" + log = self.log.bind(tag=req.meta.tag) + log.info("Running function") + + rsp = response.to(req) + xr = v1alpha1.TelemetryDestination(**resource.struct_to_dict(req.observed.composite.resource)) + + sinks = list(xr.spec.sinks) + + # A sink's auth block names an authenticator by extension name. The + # collector refuses to start when it names one no extension defines, and + # a collector that never starts looks exactly like a fleet that produces + # nothing, so it is worth catching on the object instead. + # An authenticator is defined either by the operator, under extensions, + # or by Modelplane, for a sink that set auth. Both count. + defined = set((xr.spec.extensions or {}).keys()) + defined |= {f"bearertokenauth/{s.name}" for s in sinks if s.auth and s.auth.bearerTokenKey} + missing = sorted(_authenticators(sinks) - defined) + if missing: + _not_ready( + rsp, + CONDITION_REASON_UNKNOWN_AUTHENTICATOR, + f"No extension defines {', '.join(missing)}, so the collector would refuse to start", + ) + return rsp + + for sink in sinks: + if sink.secretRef is None: + continue + key = f"{_SECRET_PREFIX}{sink.name}" + response.require_resources( + rsp, + name=key, + api_version="v1", + kind="Secret", + match_name=sink.secretRef.name, + namespace=_NAMESPACE, + ) + if key not in req.required_resources: + _not_ready(rsp, CONDITION_REASON_WAITING_FOR_SECRET, "Waiting for the credential Secret to resolve") + return rsp + if not list(request.get_required_resources(req, key)): + _not_ready( + rsp, + CONDITION_REASON_SECRET_NOT_FOUND, + f"Secret {sink.secretRef.name} does not exist, so sink {sink.name} has no credential to send with", + ) + return rsp + + resource.update_status(rsp.desired.composite, v1alpha1.Status()) + response.set_conditions( + rsp, + resource.Condition( + typ=CONDITION_TYPE_ACCEPTED, + status="True", + reason=CONDITION_REASON_AVAILABLE, + message=f"Exporting through {', '.join(f'{s.type}/{s.name}' for s in sinks)}", + ), + ) + rsp.desired.composite.ready = fnv1.READY_TRUE + return rsp + + +def _authenticators(sinks: list[v1alpha1.Sink]) -> set[str]: + """Every authenticator a sink's own config references, by extension name.""" + names: set[str] = set() + for sink in sinks: + auth = (sink.config or {}).get("auth") + if isinstance(auth, dict) and isinstance(auth.get("authenticator"), str): + names.add(auth["authenticator"]) + return names + + +def _not_ready(rsp: fnv1.RunFunctionResponse, reason: str, message: str) -> None: + """Report a destination nothing can send through, and why.""" + response.set_conditions( + rsp, + resource.Condition(typ=CONDITION_TYPE_ACCEPTED, status="False", reason=reason, message=message), + ) + rsp.desired.composite.ready = fnv1.READY_FALSE diff --git a/functions/compose-telemetry-destination/function/main.py b/functions/compose-telemetry-destination/function/main.py new file mode 100644 index 000000000..2e8441dac --- /dev/null +++ b/functions/compose-telemetry-destination/function/main.py @@ -0,0 +1,55 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The composition function's main CLI.""" + +import click +from crossplane.function import logging, runtime + +from function import fn + + +@click.command() +@click.option("--debug", "-d", is_flag=True, help="Emit debug logs.") +@click.option( + "--address", + default="0.0.0.0:9443", + show_default=True, + help="Address at which to listen for gRPC connections", +) +@click.option("--tls-certs-dir", help="Serve using mTLS certificates.", envvar="TLS_SERVER_CERTS_DIR") +@click.option( + "--insecure", + is_flag=True, + help="Run without mTLS credentials. If you supply this flag --tls-certs-dir will be ignored.", +) +def cli(debug: bool, address: str, tls_certs_dir: str, insecure: bool) -> None: + """A Crossplane composition function.""" + try: + level = logging.Level.INFO + if debug: + level = logging.Level.DEBUG + logging.configure(level=level) + runtime.serve( + fn.FunctionRunner(), + address, + creds=runtime.load_credentials(tls_certs_dir), + insecure=insecure, + ) + except Exception as e: + click.echo(f"Cannot run function: {e}") + + +if __name__ == "__main__": + cli() diff --git a/functions/compose-telemetry-destination/pyproject.toml b/functions/compose-telemetry-destination/pyproject.toml new file mode 100644 index 000000000..d36739309 --- /dev/null +++ b/functions/compose-telemetry-destination/pyproject.toml @@ -0,0 +1,26 @@ +[build-system] +requires = ["uv_build>=0.11.0,<0.12"] +build-backend = "uv_build" + +[project] +name = "compose-telemetry-destination" +version = "0.0.0" +description = "Mark a TelemetryDestination as ready." +requires-python = ">=3.11,<3.14" +license = "Apache-2.0" +dependencies = [ + "crossplane-function-sdk-python>=0.14.0", + "click>=8.1.0", + "grpcio>=1.73.1", + "crossplane-models", +] + +[tool.uv.sources] +crossplane-models = { workspace = true } + +[project.scripts] +function = "function.main:cli" + +[tool.uv.build-backend] +module-name = "function" +module-root = "" diff --git a/functions/compose-telemetry-destination/tests/__init__.py b/functions/compose-telemetry-destination/tests/__init__.py new file mode 100644 index 000000000..ebf4b2ad4 --- /dev/null +++ b/functions/compose-telemetry-destination/tests/__init__.py @@ -0,0 +1,13 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. diff --git a/functions/compose-telemetry-destination/tests/test_fn.py b/functions/compose-telemetry-destination/tests/test_fn.py new file mode 100644 index 000000000..02367c924 --- /dev/null +++ b/functions/compose-telemetry-destination/tests/test_fn.py @@ -0,0 +1,255 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests for the compose-telemetry-destination function.""" + +import dataclasses +import unittest + +from crossplane.function import logging, resource +from crossplane.function.proto.v1 import run_function_pb2 as fnv1 +from function import fn +from google.protobuf import duration_pb2 as durationpb +from google.protobuf import json_format +from google.protobuf import struct_pb2 as structpb + + +@dataclasses.dataclass +class Case: + """A test case for compose-telemetry-destination.""" + + name: str + req: fnv1.RunFunctionRequest + want: fnv1.RunFunctionResponse + + +def setUpModule() -> None: + logging.configure(level=logging.Level.DISABLED) + + +class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): + """Tests for FunctionRunner.RunFunction.""" + + @classmethod + def setUpClass(cls) -> None: + cls.runner = fn.FunctionRunner() + + async def test_compose(self) -> None: + """The function reports whether a destination can actually be sent through.""" + + def sink(name: str = "primary", type_: str = "otlphttp", secret: str | None = None) -> dict: + """A sink wiring its own authenticator, which is the case worth validating.""" + return { + "name": name, + "type": type_, + "endpoint": "https://otel.acme.example", + "config": {"auth": {"authenticator": "oauth2client/acme"}}, + **({"secretRef": {"name": secret}} if secret else {}), + } + + sinks = [sink()] + extensions = {"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}} + + def xr(spec: dict) -> dict: + return { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "TelemetryDestination", + "metadata": {"name": "default"}, + "spec": spec, + } + + def req(spec: dict, secrets: list | None = None) -> fnv1.RunFunctionRequest: + r = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(xr(spec)))), + ) + if secrets is not None: + r.required_resources["secret-primary"].items.extend([fnv1.Resource(resource=s) for s in secrets]) + return r + + def want( + ready: fnv1.Ready, status: dict | None, cond: fnv1.Condition, secret: str | None = None + ) -> fnv1.RunFunctionResponse: + composite = fnv1.Resource(ready=ready) + if status is not None: + composite.resource.CopyFrom(resource.dict_to_struct(status)) + rsp = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=composite), + conditions=[cond], + context=structpb.Struct(), + ) + if secret is not None: + rsp.requirements.resources["secret-primary"].api_version = "v1" + rsp.requirements.resources["secret-primary"].kind = "Secret" + rsp.requirements.resources["secret-primary"].match_name = secret + # Qualified: unqualified it would resolve a Secret of that + # name in any namespace, and accept the wrong credential. + rsp.requirements.resources["secret-primary"].namespace = "modelplane-system" + return rsp + + cases = [ + Case( + name="ready, naming the sinks it sends through", + req=req({"sinks": sinks, "extensions": extensions}), + want=want( + fnv1.READY_TRUE, + {"status": {}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Exporting through otlphttp/primary", + ), + ), + ), + Case( + name="ready with an exporter that references no authenticator at all", + req=req( + { + "sinks": [ + { + "name": "prom", + "type": "prometheusremotewrite", + "endpoint": "https://prom.acme.example/api/v1/write", + } + ] + } + ), + want=want( + fnv1.READY_TRUE, + {"status": {}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Exporting through prometheusremotewrite/prom", + ), + ), + ), + Case( + name="ready with no extensions, because Modelplane composes the authenticator", + req=req( + { + "sinks": [ + { + "name": "primary", + "type": "otlphttp", + "endpoint": "https://otel.acme.example", + "secretRef": {"name": "telemetry-credentials"}, + "auth": {"bearerTokenKey": "token"}, + } + ] + }, + secrets=[ + resource.dict_to_struct( + {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} + ) + ], + ), + want=want( + fnv1.READY_TRUE, + {"status": {}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Exporting through otlphttp/primary", + ), + secret="telemetry-credentials", + ), + ), + Case( + name="ready once the credential Secret exists", + req=req( + {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, + secrets=[ + resource.dict_to_struct( + {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} + ) + ], + ), + want=want( + fnv1.READY_TRUE, + {"status": {}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Exporting through otlphttp/primary", + ), + secret="telemetry-credentials", + ), + ), + Case( + name="waits for the credential Secret to resolve", + req=req( + {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, + ), + want=want( + fnv1.READY_FALSE, + None, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForSecret", + message="Waiting for the credential Secret to resolve", + ), + secret="telemetry-credentials", + ), + ), + Case( + name="not ready when a sink names an authenticator nothing defines", + req=req({"sinks": sinks}), + want=want( + fnv1.READY_FALSE, + None, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="UnknownAuthenticator", + message="No extension defines oauth2client/acme, so the collector would refuse to start", + ), + ), + ), + Case( + name="not ready when the credential Secret is missing", + req=req( + {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, + secrets=[], + ), + want=want( + fnv1.READY_FALSE, + None, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="SecretNotFound", + message=( + "Secret telemetry-credentials does not exist, " + "so sink primary has no credential to send with" + ), + ), + secret="telemetry-credentials", + ), + ), + ] + + for case in cases: + with self.subTest(case.name): + got = await self.runner.RunFunction(case.req, None) + self.assertEqual( + json_format.MessageToDict(case.want), + json_format.MessageToDict(got), + "-want, +got", + ) diff --git a/schemas/.lock.json b/schemas/.lock.json index f6003c682..b8c2f548b 100644 --- a/schemas/.lock.json +++ b/schemas/.lock.json @@ -1,6 +1,6 @@ { "packages": { - "fs://apis": "a3c8867fa02ab5c25944ea69b769d69c5b8de1cf1d21ca61fc9c558159c7bf25", + "fs://apis": "8718ada2c08c9f8bdad4a455f3f05e1c1cc5a83ef633259fd01848d8b5a8be84", "git://https://github.com/crossplane/crossplane/cluster/crds": "90d8b72ad8b829f0bcd7d7d5a98eaa0d579f244a", "xpkg://xpkg.upbound.io/upbound/provider-aws-ec2:v2.8.1": "sha256:ca2e9e3b2e3a8b6ca44a9700d5abf7abd733cfa388d1afe9bb2bf7c847cd39ef", "xpkg://xpkg.upbound.io/upbound/provider-aws-efs:v2.8.1": "sha256:ebb1bcd8dc9a7e60e97a324652fbd9d1b0609ddbb1107db518ce999cbc1113bd", diff --git a/schemas/python/models/ai/modelplane/metricmapping/__init__.py b/schemas/python/models/ai/modelplane/metricmapping/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/schemas/python/models/ai/modelplane/metricmapping/v1alpha1.py b/schemas/python/models/ai/modelplane/metricmapping/v1alpha1.py new file mode 100644 index 000000000..f924b9700 --- /dev/null +++ b/schemas/python/models/ai/modelplane/metricmapping/v1alpha1.py @@ -0,0 +1,167 @@ +# generated by datamodel-codegen: +# filename: workdir/modelplane_ai_v1alpha1_metricmapping.yaml + +from __future__ import annotations + +from typing import Literal + +from pydantic import AwareDatetime, BaseModel, Field, constr + +from ....io.k8s.apimachinery.pkg.apis.meta import v1 + + +class CompositionRef(BaseModel): + name: str + + +class CompositionRevisionRef(BaseModel): + name: str + + +class CompositionRevisionSelector(BaseModel): + matchLabels: dict[str, str] + + +class CompositionSelector(BaseModel): + matchLabels: dict[str, str] + + +class ResourceRef(BaseModel): + apiVersion: str + kind: str + name: str | None = None + namespace: str | None = None + + +class Crossplane(BaseModel): + compositionRef: CompositionRef | None = None + compositionRevisionRef: CompositionRevisionRef | None = None + compositionRevisionSelector: CompositionRevisionSelector | None = None + compositionSelector: CompositionSelector | None = None + compositionUpdatePolicy: Literal['Automatic', 'Manual'] | None = None + resourceRefs: list[ResourceRef] | None = None + + +class Label(BaseModel): + from_: constr(pattern=r'^[a-zA-Z_][a-zA-Z0-9_]*$', max_length=63) | None = Field( + None, alias='from' + ) + """ + A label the component already emits. Modelplane copies its value into this label and removes the original. + Naming the label itself keeps it: that is how `values` rewrites what a component writes without renaming the label. + """ + name: constr(pattern=r'^[a-zA-Z_][a-zA-Z0-9_]*$', max_length=63) + """ + The label to set. + """ + value: constr(max_length=253) | None = None + """ + A fixed value, the same on every series this mapping produces. This is what tells two folded metrics apart. + """ + values: dict[str, constr(max_length=253)] | None = Field(None, max_length=32) + """ + What each of that label's values becomes, for putting an engine's own vocabulary into Modelplane's. A value with no entry here is left as the component wrote it. + """ + + +class Metric(BaseModel): + from_: constr(pattern=r'^[a-zA-Z_:][a-zA-Z0-9_:]*$', max_length=255) = Field( + ..., alias='from' + ) + """ + The metric's name as the component exposes it, matched exactly wherever it appears in the fleet. + A histogram is named by its base name, without the _count, _sum or _bucket a Prometheus query would use: the collector holds it as one metric, and `part` is what reaches into it. + """ + fromUnit: ( + Literal['Millijoules', 'Mebibytes', 'Milliseconds', 'Nanoseconds', 'Percent'] + | None + ) = None + """ + What the component measures this in, when that isn't the unit the name claims. Modelplane converts to the base unit: millijoules and milliseconds are divided by a thousand, nanoseconds by a billion, percent by a hundred, and mebibytes multiplied out to bytes. A histogram is converted whole - its sum, its bounds and its bucket boundaries - so its quantiles come out in the target unit too. + Percent is for a component that counts a saturation from nought to a hundred where the name says a ratio. Check rather than assume: vLLM publishes kv_cache_usage_perc and the value is a fraction, so a name is no guide. + """ + labels: list[Label] | None = Field(None, max_length=16) + """ + Labels to set on this metric's series. To fold several metrics into one name, give each its own entry with the same `to` and a different fixed value, such as direction: input and direction: output on modelplane_tokens_total. + """ + part: Literal['Count', 'Sum'] | None = None + """ + Take a part of a histogram as a counter of its own, rather than the histogram itself. Count is how many observations it holds, which is a request count where the histogram measures request duration. Sum is their total. + The extraction leaves the histogram alone, but the collector exports only what a mapping renames, so the histogram itself is dropped unless another mapping gives it a `modelplane_` name of its own. Write that second mapping to keep both. + """ + to: constr(pattern=r'^modelplane_[a-z0-9_]*[a-z0-9]$', max_length=255) + """ + The name to export the metric under. Only modelplane_* metrics leave a cluster, so a metric no mapping renames never leaves its cluster. + Name it in base units - seconds, bytes, joules, a ratio from nought to one - because that is what `fromUnit` converts to. + """ + + +class Spec(BaseModel): + crossplane: Crossplane | None = None + """ + Configures how Crossplane will reconcile this composite resource + """ + metrics: list[Metric] = Field(..., max_length=128, min_length=1) + """ + The metrics to rename. Give two components' metrics the same name only if they measure the same thing, and histograms only if their buckets match too. A quantile across mismatched buckets is wrong. + """ + + +class Condition(BaseModel): + lastTransitionTime: AwareDatetime + message: str | None = None + observedGeneration: int | None = None + reason: str + status: str + type: str + + +class Status(BaseModel): + clusters: int | None = None + """ + How many inference clusters apply this mapping. + """ + conditions: list[Condition] | None = None + """ + Conditions of the resource. + """ + + +class MetricMapping(BaseModel): + apiVersion: Literal['modelplane.ai/v1alpha1'] | None = 'modelplane.ai/v1alpha1' + """ + APIVersion defines the versioned schema of this representation of an object. Servers should convert recognized schemas to the latest internal value, and may reject unrecognized values. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + """ + kind: Literal['MetricMapping'] | None = 'MetricMapping' + """ + Kind is a string value representing the REST resource this object represents. Servers may infer this from the endpoint the client submits requests to. Cannot be updated. In CamelCase. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + """ + metadata: v1.ObjectMeta | None = None + """ + Standard object's metadata. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#metadata + """ + spec: Spec + """ + How one component's metrics become part of the modelplane_* surface. Modelplane renders every MetricMapping into each inference cluster's collector, so a mapping is written once on the control plane and reaches the whole fleet. + A mapping naming a component Modelplane already provides renames for is additive: its renames run after the built-in ones, on whatever those left behind. A metric a built-in already renamed no longer answers to the name it was emitted under, so a second mapping selecting on that name matches nothing and the built-in stands. Select on the `modelplane_` name instead to rename one of Modelplane's own. + """ + status: Status | None = None + + +class MetricMappingList(BaseModel): + apiVersion: str | None = None + """ + APIVersion defines the versioned schema of this representation of an object. Servers should convert recognized schemas to the latest internal value, and may reject unrecognized values. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + """ + items: list[MetricMapping] + """ + List of metricmappings. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md + """ + kind: str | None = None + """ + Kind is a string value representing the REST resource this object represents. Servers may infer this from the endpoint the client submits requests to. Cannot be updated. In CamelCase. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + """ + metadata: v1.ListMeta | None = None + """ + Standard list metadata. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + """ \ No newline at end of file diff --git a/schemas/python/models/ai/modelplane/telemetrydestination/__init__.py b/schemas/python/models/ai/modelplane/telemetrydestination/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/schemas/python/models/ai/modelplane/telemetrydestination/v1alpha1.py b/schemas/python/models/ai/modelplane/telemetrydestination/v1alpha1.py new file mode 100644 index 000000000..3414851d8 --- /dev/null +++ b/schemas/python/models/ai/modelplane/telemetrydestination/v1alpha1.py @@ -0,0 +1,159 @@ +# generated by datamodel-codegen: +# filename: workdir/modelplane_ai_v1alpha1_telemetrydestination.yaml + +from __future__ import annotations + +from typing import Any, Literal + +from pydantic import AwareDatetime, BaseModel, Field, constr + +from ....io.k8s.apimachinery.pkg.apis.meta import v1 + + +class CompositionRef(BaseModel): + name: str + + +class CompositionRevisionRef(BaseModel): + name: str + + +class CompositionRevisionSelector(BaseModel): + matchLabels: dict[str, str] + + +class CompositionSelector(BaseModel): + matchLabels: dict[str, str] + + +class ResourceRef(BaseModel): + apiVersion: str + kind: str + name: str | None = None + namespace: str | None = None + + +class Crossplane(BaseModel): + compositionRef: CompositionRef | None = None + compositionRevisionRef: CompositionRevisionRef | None = None + compositionRevisionSelector: CompositionRevisionSelector | None = None + compositionSelector: CompositionSelector | None = None + compositionUpdatePolicy: Literal['Automatic', 'Manual'] | None = None + resourceRefs: list[ResourceRef] | None = None + + +class Auth(BaseModel): + bearerTokenKey: constr(max_length=253) | None = None + """ + The key in this sink's Secret holding the bearer token. Modelplane mounts it as a file and points the authenticator at it, so a rotated token is picked up without restarting the collector. + """ + + +class SecretRef(BaseModel): + name: constr(max_length=253) + """ + Name of the Secret, in modelplane-system on the control plane. Modelplane copies it to every cluster running a collector, so it doesn't have to exist on each of them already. + """ + + +class Sink(BaseModel): + auth: Auth | None = None + """ + Authentication Modelplane sets up for this sink, using a credential from `secretRef`. For another scheme, define an authenticator under spec.extensions and reference it from `config`. + Set here, it wins: Modelplane applies it over an `auth` block in `config`, so a sink's credential cannot be quietly unpicked. + """ + config: dict[str, Any] | None = None + """ + The rest of the exporter's configuration, passed through as written: TLS, retries, queueing, compression, headers. + Modelplane sets one default, for the exporters that flatten a series into labels: prometheus and prometheus_remote_write get resource_to_telemetry_conversion, or they would receive every series stripped of the cluster, deployment, engine and role it belongs to. Setting it here overrides that. + """ + endpoint: constr(max_length=2048) | None = None + """ + Where this sink writes. Leave it unset for an exporter that doesn't take an endpoint, such as kafka or debug, and configure it in `config` instead. + Set here, it wins: Modelplane applies it over `config`, so a sink cannot be quietly redirected by the configuration passed through beside it. + """ + name: constr(pattern=r'^[a-z0-9]([-a-z0-9]*[a-z0-9])?$', max_length=63) + """ + This sink's name, unique within the destination. It names the collector's exporter instance, the authenticator Modelplane composes for it, and the directory its credential mounts at, so renaming one restarts the collector. + """ + secretRef: SecretRef | None = None + """ + A Secret holding this sink's credentials. Modelplane mounts each key as a file under /etc/modelplane/telemetry// and sets it as an environment variable for ${env:KEY} references in `config`. All sinks share one environment, so where two Secrets have the same key, refer to the file. + """ + type: constr(max_length=63) + """ + The collector exporter to send with, by the name OpenTelemetry gives it: otlphttp, otlp, prometheus_remote_write, kafka, and every other one the collector provides. + Not an enum: the collector already refuses to start on a name it doesn't have, so repeating the list here would only add a second place for it to go stale. + """ + + +class Spec(BaseModel): + crossplane: Crossplane | None = None + """ + Configures how Crossplane will reconcile this composite resource + """ + extensions: dict[str, Any] | None = None + """ + Collector extensions, passed through as written. Use it to define an authenticator that a sink's `config` references. + An exporter needs a client authenticator - basicauth, oauth2client, sigv4auth, headers_setter. The oidc extension authenticates callers of a receiver, so it is not one of these. + """ + sinks: list[Sink] = Field(..., max_length=16, min_length=1) + """ + Where to send it. Every sink gets the whole stream, so two sinks is two copies of the fleet's metrics, billed twice. + """ + + +class Condition(BaseModel): + lastTransitionTime: AwareDatetime + message: str | None = None + observedGeneration: int | None = None + reason: str + status: str + type: str + + +class Status(BaseModel): + conditions: list[Condition] | None = None + """ + Conditions of the resource. + """ + + +class TelemetryDestination(BaseModel): + apiVersion: Literal['modelplane.ai/v1alpha1'] | None = 'modelplane.ai/v1alpha1' + """ + APIVersion defines the versioned schema of this representation of an object. Servers should convert recognized schemas to the latest internal value, and may reject unrecognized values. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + """ + kind: Literal['TelemetryDestination'] | None = 'TelemetryDestination' + """ + Kind is a string value representing the REST resource this object represents. Servers may infer this from the endpoint the client submits requests to. Cannot be updated. In CamelCase. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + """ + metadata: v1.ObjectMeta | None = None + """ + Standard object's metadata. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#metadata + """ + spec: Spec + """ + Where the fleet's metrics go. Modelplane runs no collectors until a TelemetryDestination exists, and creating one turns on collection on every inference cluster. + Several can exist. Their sinks are concatenated into one collector configuration, so adding a backend is a new object rather than an edit to one somebody else owns. A sink's name is the collector's name for its exporter, so it has to be unique across destinations. + """ + status: Status | None = None + + +class TelemetryDestinationList(BaseModel): + apiVersion: str | None = None + """ + APIVersion defines the versioned schema of this representation of an object. Servers should convert recognized schemas to the latest internal value, and may reject unrecognized values. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + """ + items: list[TelemetryDestination] + """ + List of telemetrydestinations. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md + """ + kind: str | None = None + """ + Kind is a string value representing the REST resource this object represents. Servers may infer this from the endpoint the client submits requests to. Cannot be updated. In CamelCase. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + """ + metadata: v1.ListMeta | None = None + """ + Standard list metadata. More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + """ \ No newline at end of file diff --git a/uv.lock b/uv.lock index b72267a7c..9fcb8dba1 100644 --- a/uv.lock +++ b/uv.lock @@ -15,6 +15,7 @@ members = [ "compose-inference-class", "compose-inference-cluster", "compose-inference-gateway", + "compose-metric-mapping", "compose-model-cache", "compose-model-deployment", "compose-model-endpoint", @@ -23,6 +24,7 @@ members = [ "compose-model-service", "compose-nebius-cluster", "compose-serving-stack", + "compose-telemetry-destination", "compose-usages", "compose-vultr-cluster", "crossplane-models", @@ -208,6 +210,25 @@ requires-dist = [ { name = "grpcio", specifier = ">=1.73.1" }, ] +[[package]] +name = "compose-metric-mapping" +version = "0.0.0" +source = { editable = "functions/compose-metric-mapping" } +dependencies = [ + { name = "click" }, + { name = "crossplane-function-sdk-python" }, + { name = "crossplane-models" }, + { name = "grpcio" }, +] + +[package.metadata] +requires-dist = [ + { name = "click", specifier = ">=8.1.0" }, + { name = "crossplane-function-sdk-python", specifier = ">=0.14.0" }, + { name = "crossplane-models", editable = "schemas/python" }, + { name = "grpcio", specifier = ">=1.73.1" }, +] + [[package]] name = "compose-model-cache" version = "0.0.0" @@ -364,6 +385,25 @@ requires-dist = [ { name = "pyyaml", specifier = ">=6.0" }, ] +[[package]] +name = "compose-telemetry-destination" +version = "0.0.0" +source = { editable = "functions/compose-telemetry-destination" } +dependencies = [ + { name = "click" }, + { name = "crossplane-function-sdk-python" }, + { name = "crossplane-models" }, + { name = "grpcio" }, +] + +[package.metadata] +requires-dist = [ + { name = "click", specifier = ">=8.1.0" }, + { name = "crossplane-function-sdk-python", specifier = ">=0.14.0" }, + { name = "crossplane-models", editable = "schemas/python" }, + { name = "grpcio", specifier = ">=1.73.1" }, +] + [[package]] name = "compose-usages" version = "0.0.0"